andyshu commited on
Commit
1a0a7fb
·
verified ·
1 Parent(s): 8651d62

Back up verified OpenSysOne training snapshot and pinned source

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +1 -0
  2. publications/20260917-expanded-pilot-hf-backup/fe7e48f806a7e947d689b9e966d38bc0dbedcc2e/manifest.json +1542 -0
  3. snapshots/20260917-expanded-pilot-hf-backup/SHA256SUMS +40 -0
  4. snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/best.pt +3 -0
  5. snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/best_validation_selection.json +601 -0
  6. snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/checkpoint.pt +3 -0
  7. snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/correctness_final.json +9 -0
  8. snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/correctness_initial.json +9 -0
  9. snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/data_filter.json +100 -0
  10. snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/exit_code +1 -0
  11. snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/manifest.json +370 -0
  12. snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/summary.json +437 -0
  13. snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/validation_step_000000_predictions.json +0 -0
  14. snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/best.pt +3 -0
  15. snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/best_validation_selection.json +601 -0
  16. snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/correctness_initial.json +9 -0
  17. snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/data_filter.json +96 -0
  18. snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/manifest.json +360 -0
  19. snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/parent-snapshot.json +273 -0
  20. snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/source-data-manifest.json +136 -0
  21. snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/validation_step_001500_predictions.json +0 -0
  22. snapshots/20260917-expanded-pilot-hf-backup/backup-manifest.json +402 -0
  23. snapshots/20260917-expanded-pilot-hf-backup/backup-tools/prepare_expansion_backup.py +331 -0
  24. snapshots/20260917-expanded-pilot-hf-backup/backup-tools/test_expansion_backup.py +175 -0
  25. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v1-20260916-manifest.json +136 -0
  26. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/ATTRIBUTION.md +18 -0
  27. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/calibration.jsonl +0 -0
  28. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/diagnostics/new_sources.jsonl +0 -0
  29. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/holdout.jsonl +0 -0
  30. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/manifest.json +303 -0
  31. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/data_filter.json +100 -0
  32. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/postbuild-audit.json +239 -0
  33. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/prepare_expanded_data.py +343 -0
  34. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/tokenization-proof.json +64 -0
  35. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/tokenize_only.py +79 -0
  36. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/raw/commonsenseqa/README.md +233 -0
  37. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/raw/hellaswag/README.md +218 -0
  38. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/raw/piqa/piqa/README.md +5 -0
  39. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/test.jsonl +0 -0
  40. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/train.jsonl +3 -0
  41. snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/validation.jsonl +0 -0
  42. snapshots/20260917-expanded-pilot-hf-backup/fleet/plan.json +67 -0
  43. snapshots/20260917-expanded-pilot-hf-backup/fleet/reference-validation-predictions.json +0 -0
  44. source/EXPANDED_DATA.md +147 -0
  45. source/FLEET_RUN.md +11 -0
  46. source/HANDOVER.md +20 -11
  47. source/HF_MODEL_CARD.md +27 -8
  48. source/PLAN.md +18 -0
  49. source/RESULTS.md +31 -0
  50. source/data_transition.py +64 -0
.gitattributes CHANGED
@@ -35,3 +35,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
  source/results/20260917-playground/desktop.png filter=lfs diff=lfs merge=lfs -text
37
  source/results/20260917-playground/mobile.png filter=lfs diff=lfs merge=lfs -text
 
 
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
  source/results/20260917-playground/desktop.png filter=lfs diff=lfs merge=lfs -text
37
  source/results/20260917-playground/mobile.png filter=lfs diff=lfs merge=lfs -text
38
+ snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/train.jsonl filter=lfs diff=lfs merge=lfs -text
publications/20260917-expanded-pilot-hf-backup/fe7e48f806a7e947d689b9e966d38bc0dbedcc2e/manifest.json ADDED
@@ -0,0 +1,1542 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "created_utc": "2026-09-17T07:24:12.563442+00:00",
3
+ "files": {
4
+ "snapshots/20260917-expanded-pilot-hf-backup/SHA256SUMS": {
5
+ "sha256": "60eb2bde2dd7005cd845ce9b16b5ed149ad7172def61f678445d378c607c60e9",
6
+ "size": 4664
7
+ },
8
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/best.pt": {
9
+ "sha256": "5b9eccc4c4e2bf1306e2def3e663b66e9f0cd1f3791dff04313039eb07426be5",
10
+ "size": 66203019
11
+ },
12
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/best_validation_selection.json": {
13
+ "sha256": "ced9908b7a6a9d6f682c7afac42d12316467b7a8aa83dcfd4b09b6af8aee6d6e",
14
+ "size": 19700
15
+ },
16
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/checkpoint.pt": {
17
+ "sha256": "cf4cd775165cc2bb02113d17b94facd0db7b94fcff87b27389b3280fbb712469",
18
+ "size": 198820477
19
+ },
20
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/correctness_final.json": {
21
+ "sha256": "d5a9fc3e02e6e5d204d4c5154c21d30dcf92aa7621620f1ea7f64e824a4ecfef",
22
+ "size": 355
23
+ },
24
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/correctness_initial.json": {
25
+ "sha256": "cd6b7a551708c15a09099a58d7863fc0eee17e8b5ff36fd3ed2fdb7147ffbf3a",
26
+ "size": 356
27
+ },
28
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/data_filter.json": {
29
+ "sha256": "f07eef84b3081ad86bb5b48f810bbed569a76ee8e82cba9228beec908232d79e",
30
+ "size": 2462
31
+ },
32
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/exit_code": {
33
+ "sha256": "9a271f2a916b0b6ee6cecb2426f0b3206ef074578be55d9bc94f6f3fe3ab86aa",
34
+ "size": 2
35
+ },
36
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/manifest.json": {
37
+ "sha256": "80dc3efef131f59bc7bb5005bc6d1de46350c405604dd3a710aa1f2d34c3762b",
38
+ "size": 14253
39
+ },
40
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/summary.json": {
41
+ "sha256": "bc5e3ad91ecf3ef13ed3b82e408ebdd7110102a924f8b046e5ac3631f7fcdda7",
42
+ "size": 11417
43
+ },
44
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/validation_step_000000_predictions.json": {
45
+ "sha256": "1e0a2f838251c8eb411b59f4fa061db67ec308ce249d166901760b097c64bd6c",
46
+ "size": 331142
47
+ },
48
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/best.pt": {
49
+ "sha256": "5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024",
50
+ "size": 66202635
51
+ },
52
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/best_validation_selection.json": {
53
+ "sha256": "ced9908b7a6a9d6f682c7afac42d12316467b7a8aa83dcfd4b09b6af8aee6d6e",
54
+ "size": 19700
55
+ },
56
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/correctness_initial.json": {
57
+ "sha256": "f9ad8b0308a09577abc390412f214cc6289b1108ced3907570c6d4cf58539158",
58
+ "size": 354
59
+ },
60
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/data_filter.json": {
61
+ "sha256": "3a514e34a8a6f7e35776bafe5b19015d6c5cec0eaabea5558aa896471aba3b56",
62
+ "size": 2360
63
+ },
64
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/manifest.json": {
65
+ "sha256": "0fe2c1bde4a47879d3ba040b97aa26c3025101dec10192e86188fe2cd98c33cf",
66
+ "size": 13002
67
+ },
68
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/parent-snapshot.json": {
69
+ "sha256": "fa43228a19a32fc2caf5480799a2746b4da619b42406c0f3898d36059107050f",
70
+ "size": 11628
71
+ },
72
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/source-data-manifest.json": {
73
+ "sha256": "adf5a8ca2bab60cf7429a82b1b7a2d1ae7d3cc812de04dd195a478dfd7c6f628",
74
+ "size": 5266
75
+ },
76
+ "snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/validation_step_001500_predictions.json": {
77
+ "sha256": "1e0a2f838251c8eb411b59f4fa061db67ec308ce249d166901760b097c64bd6c",
78
+ "size": 331142
79
+ },
80
+ "snapshots/20260917-expanded-pilot-hf-backup/backup-manifest.json": {
81
+ "sha256": "d23a648a7d73cb8d6f139033985b1b6bb67be3bb510ebbb89f4432138c2ebd52",
82
+ "size": 23085
83
+ },
84
+ "snapshots/20260917-expanded-pilot-hf-backup/backup-tools/prepare_expansion_backup.py": {
85
+ "sha256": "0b546fd6b96e34316fddcb06314fa072d1ddb9c06d7e6979ccd790421b93f419",
86
+ "size": 21154
87
+ },
88
+ "snapshots/20260917-expanded-pilot-hf-backup/backup-tools/test_expansion_backup.py": {
89
+ "sha256": "3845483f6469dc30122ae31a2bcbf6b1ad7ce69663308266f831142cd5d4fd1e",
90
+ "size": 11522
91
+ },
92
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v1-20260916-manifest.json": {
93
+ "sha256": "adf5a8ca2bab60cf7429a82b1b7a2d1ae7d3cc812de04dd195a478dfd7c6f628",
94
+ "size": 5266
95
+ },
96
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/ATTRIBUTION.md": {
97
+ "sha256": "87f4430d84f2a6157593f9af44a356ede87bad18cc51fa9ae3db2fee96830691",
98
+ "size": 1345
99
+ },
100
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/calibration.jsonl": {
101
+ "sha256": "58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4",
102
+ "size": 303868
103
+ },
104
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/diagnostics/new_sources.jsonl": {
105
+ "sha256": "93ec461769c925a7f10de76ff7b04118ab731ed5f49ebb5517b2d448fa1fdafc",
106
+ "size": 318001
107
+ },
108
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/holdout.jsonl": {
109
+ "sha256": "0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3",
110
+ "size": 307068
111
+ },
112
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/manifest.json": {
113
+ "sha256": "fde6ee7ce2eca20cb22cdbbe4db0ddbdb29a8ea9d597906d88e545939a5b602c",
114
+ "size": 12761
115
+ },
116
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/data_filter.json": {
117
+ "sha256": "f07eef84b3081ad86bb5b48f810bbed569a76ee8e82cba9228beec908232d79e",
118
+ "size": 2462
119
+ },
120
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/postbuild-audit.json": {
121
+ "sha256": "a72ee77ba11017679548b06a2b956f7d2e56ca07e95eba165bb48aefbb75da18",
122
+ "size": 6956
123
+ },
124
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/prepare_expanded_data.py": {
125
+ "sha256": "5c05478b29c84218784690f3c7c3ec994615fec1c57826591607189f191c56b4",
126
+ "size": 18386
127
+ },
128
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/tokenization-proof.json": {
129
+ "sha256": "c2c506ad72513d573eec723adb3960a00f9416fb974e9359f7b61b108f1ae2bb",
130
+ "size": 2220
131
+ },
132
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/tokenize_only.py": {
133
+ "sha256": "f30b0dab7c917f74676fd1cffc988ffb175ddb9a843f2029ce715d7e44bc7f96",
134
+ "size": 4027
135
+ },
136
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/raw/commonsenseqa/README.md": {
137
+ "sha256": "172917e887dcc013fe0f0bd6aa8c810aea1e2be67d3bc7ff63e9b9d92075cc34",
138
+ "size": 7395
139
+ },
140
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/raw/hellaswag/README.md": {
141
+ "sha256": "cfe6e26e7e936a447a12f7eee50f2bffb0355c3b97459496d8c7296f65c5b353",
142
+ "size": 7019
143
+ },
144
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/raw/piqa/piqa/README.md": {
145
+ "sha256": "f7bcb808a64959151698f2bca621572c4809459733d04b95d53f3c62588f1593",
146
+ "size": 244
147
+ },
148
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/test.jsonl": {
149
+ "sha256": "ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819",
150
+ "size": 1204426
151
+ },
152
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/train.jsonl": {
153
+ "sha256": "d6d6a57c3aaaf6527647d0a21cc6027a58a6b40e2099eebb9a0689c0212ee372",
154
+ "size": 55474225
155
+ },
156
+ "snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/validation.jsonl": {
157
+ "sha256": "411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f",
158
+ "size": 292566
159
+ },
160
+ "snapshots/20260917-expanded-pilot-hf-backup/fleet/plan.json": {
161
+ "sha256": "657979189140a298f383cbcf722c425c64f7b07b7e0332c32ac1b076445028e2",
162
+ "size": 2790
163
+ },
164
+ "snapshots/20260917-expanded-pilot-hf-backup/fleet/reference-validation-predictions.json": {
165
+ "sha256": "e671e1508185765552b0f933ba03f356be62143c531d8ef534457d34b1645c9b",
166
+ "size": 333136
167
+ },
168
+ "source/.gitignore": {
169
+ "sha256": "c08a0eb968c43d53c1e54ef3f10e21ab4671c595cf25a2caa240b847acd06ddd",
170
+ "size": 73
171
+ },
172
+ "source/AGENTS.md": {
173
+ "sha256": "d27c6c6c5720db9c57efb1ee50a756c3d0486351fa2dcb33fd06a64fe7df7ac7",
174
+ "size": 1382
175
+ },
176
+ "source/EXPANDED_DATA.md": {
177
+ "sha256": "d22f87f799bb9e27c08f7cd16d74e48703492588a82106044307b0da614a9a9c",
178
+ "size": 8582
179
+ },
180
+ "source/FLEET_RUN.md": {
181
+ "sha256": "ed654924597f0ad466f6f5f8e0aeb571c0eb86dd8e5db3afd661bfed69fb667e",
182
+ "size": 14607
183
+ },
184
+ "source/FLEET_SCOUT.md": {
185
+ "sha256": "789b020758a2713c74003f7d8d7af2d099c348d786c58c54136738a7ddcf7285",
186
+ "size": 8383
187
+ },
188
+ "source/HANDOVER.md": {
189
+ "sha256": "dcb0d18d8ce426dad3699853fc4891b92904a6d3c21e2dfbbefaca95d1935349",
190
+ "size": 21479
191
+ },
192
+ "source/HF_MODEL_CARD.md": {
193
+ "sha256": "1aa69fde0c3aa8c665885433104e352fdce6695ce1634ce720f1221b19c42784",
194
+ "size": 7433
195
+ },
196
+ "source/HUGGINGFACE.md": {
197
+ "sha256": "521ae9303dfc69b65ba570811814c758434565e0b8057da97a135dcb5d7aab05",
198
+ "size": 5610
199
+ },
200
+ "source/JEV_HARNESS.md": {
201
+ "sha256": "c0f4a0d9055689b6a8be6fb66aadb43ed448609002618dd4d7e3857808c5e88b",
202
+ "size": 4683
203
+ },
204
+ "source/NEXT_STEPS.md": {
205
+ "sha256": "8fc19584153ca56b3e86ca77af7dd01e6d5a8d05a998973359dc003e5ff56a4f",
206
+ "size": 12181
207
+ },
208
+ "source/PLAN.md": {
209
+ "sha256": "b7453ea000250fd11a61fbb7414047819f5898816913fdb5d14fe18a08218f1f",
210
+ "size": 17060
211
+ },
212
+ "source/PLAYGROUND.md": {
213
+ "sha256": "11ced3671ace90629e9c6c38d4c8cd48cd129686bd482cc961fed06aa9795c27",
214
+ "size": 6288
215
+ },
216
+ "source/README.md": {
217
+ "sha256": "f5e5afde7c7073bd5f6e482e9f8eb61f8cd76268293a13626d6bc812a6bdc812",
218
+ "size": 3740
219
+ },
220
+ "source/RESEARCH_BRIEF.md": {
221
+ "sha256": "9ed40f7da30fbc0d1492d30e6bd5a2f315167f077293cbc3f7345de3a20ef164",
222
+ "size": 24894
223
+ },
224
+ "source/RESEARCH_NOTES.md": {
225
+ "sha256": "e11d9f1e09481e741cd333d7fe84e566bd468db40af77d92f5c361f4c428306c",
226
+ "size": 13713
227
+ },
228
+ "source/RESULTS.md": {
229
+ "sha256": "c8ebd5b813ae7b51311deb74c79c7ba8cb4b35b0959f791052fc7d24587ab694",
230
+ "size": 32971
231
+ },
232
+ "source/data_transition.py": {
233
+ "sha256": "93aaa89b4de3aa78c34f03a5643e31f91f738c832395334966368df9b902621c",
234
+ "size": 3274
235
+ },
236
+ "source/decision_model.py": {
237
+ "sha256": "a3d8aeb02a1ac765c6cc30ff175acad0664560f01ab5403e22cade924d17371e",
238
+ "size": 9258
239
+ },
240
+ "source/examples/jev_request.json": {
241
+ "sha256": "74f07501aa665284ab0611b6a3ed1fdc046821be10dddaa0a503efc52d7d7eb5",
242
+ "size": 703
243
+ },
244
+ "source/experiment.py": {
245
+ "sha256": "c779c3936aa1c2c51052f035df7bc0895a2de79c9ffc6c50fb0ee848832e17c7",
246
+ "size": 43729
247
+ },
248
+ "source/jev_harness.py": {
249
+ "sha256": "4d4e979cb7ae352bcdacaaa6d64045e6b5e550b1a9721d4bad545045bee6c67f",
250
+ "size": 16898
251
+ },
252
+ "source/playground.py": {
253
+ "sha256": "b10c400421dd8558a7fef8ddde632676cdfe7f63edf94184f299f9ed569c010a",
254
+ "size": 18520
255
+ },
256
+ "source/results/20260916-fleet-setup/Qwen3-4B-Instruct-2507-cdbee75f-files.json": {
257
+ "sha256": "3ceddf4e5228246ddb9a6ad381fb31793a61b050a8fb10934a345225e6068fe6",
258
+ "size": 1971
259
+ },
260
+ "source/results/20260916-fleet-setup/Qwen3.5-2B-15852e8c-files.json": {
261
+ "sha256": "a78fb2654833d89f5b274b562a4595f2dff001065db2cf64f83bcd1a206c7576",
262
+ "size": 1950
263
+ },
264
+ "source/results/20260916-fleet-setup/cpu-tests.json": {
265
+ "sha256": "2110698e40353ee830b3f7bf792c5f2c8ea0aff54c8fa7d9b6b49fde99540b71",
266
+ "size": 328
267
+ },
268
+ "source/results/20260916-fleet-setup/crossfit-tests.json": {
269
+ "sha256": "a804cfb5eaf7a963085cb399da2f0789f8a50c56bd6c38f44b7906c4fffe2d3f",
270
+ "size": 532
271
+ },
272
+ "source/results/20260916-fleet-setup/evidence-index.json": {
273
+ "sha256": "db4d25099b45af53cffa54551cb0c56f82b1a0228915455873183f31a228148e",
274
+ "size": 12480
275
+ },
276
+ "source/results/20260916-fleet-setup/fleet-current-status.json": {
277
+ "sha256": "06b07b3ed301bdfaacfe4ded988033e525e4ea7580b0f6bdbe6bdaceb4a1bab9",
278
+ "size": 11377
279
+ },
280
+ "source/results/20260916-fleet-setup/fleet-launch-verification.json": {
281
+ "sha256": "9f6ee33c3004645c851033f87b8fb748478d37a2b59dff75b35fff9e1f45a413",
282
+ "size": 553
283
+ },
284
+ "source/results/20260916-fleet-setup/fleet-plan.json": {
285
+ "sha256": "bff9ed43f7fd7710ec3915144424daf1f0ffb1160ae86ee3a654cbba7974cd4b",
286
+ "size": 1795
287
+ },
288
+ "source/results/20260916-fleet-setup/fleet-state.json": {
289
+ "sha256": "a84107d6c7ee5c96995cd3fd314d5ce217c1ef29f6296f8771b3bdfdaa6fc018",
290
+ "size": 664
291
+ },
292
+ "source/results/20260916-fleet-setup/gx10-current-best_validation_predictions.json": {
293
+ "sha256": "9f1f8ccbc26a29ba0a0615ff1cbf278776c01c003b63176bae41df5a5bde92e6",
294
+ "size": 332042
295
+ },
296
+ "source/results/20260916-fleet-setup/gx10-current-best_validation_selection.json": {
297
+ "sha256": "9cbc9717355d4d8e17d1d072cae9c243c37e2497e691ca0685fe5b6194c3d74b",
298
+ "size": 19696
299
+ },
300
+ "source/results/20260916-fleet-setup/gx10-current-correctness_initial.json": {
301
+ "sha256": "5a3a0b5b7cd5b9ac2840fe9a6de37d60fc069f1958e315a8a0c973799a960198",
302
+ "size": 352
303
+ },
304
+ "source/results/20260916-fleet-setup/gx10-current-manifest.json": {
305
+ "sha256": "8c81e2582258ec2c554b284e355e8059f317d91d24ec775e32dc86dc696c5845",
306
+ "size": 13020
307
+ },
308
+ "source/results/20260916-fleet-setup/gx10-current-startup-verification.json": {
309
+ "sha256": "875e7c410ad0c98328737c2f34b5da1d9dedf9d5ba44cfed0b9bdd43226f73af",
310
+ "size": 24964
311
+ },
312
+ "source/results/20260916-fleet-setup/gx10-current-validation_step_000178_predictions.json": {
313
+ "sha256": "9f1f8ccbc26a29ba0a0615ff1cbf278776c01c003b63176bae41df5a5bde92e6",
314
+ "size": 332042
315
+ },
316
+ "source/results/20260916-fleet-setup/gx10-python-stack-proof.json": {
317
+ "sha256": "9ee3222204613414ecbcba0a2faecf740c03e1d771fbdcc0af5ff5d98a5bb62a",
318
+ "size": 364
319
+ },
320
+ "source/results/20260916-fleet-setup/gx10-resume-state-verification.json": {
321
+ "sha256": "5ce557a67fd757854667e36473754c6b14928615c9a6564bc036fe3bd0e1465c",
322
+ "size": 635
323
+ },
324
+ "source/results/20260916-fleet-setup/gx10-transition-checkpoint.json": {
325
+ "sha256": "f14a3b2da291655c3296e1a5fad439c83de1cf7e143318e74781b4fb1767f6a6",
326
+ "size": 581
327
+ },
328
+ "source/results/20260916-fleet-setup/selection-diagnostic.json": {
329
+ "sha256": "783190fe970f0930b449632065b2689d5b2e91a6d7a31415cf4710c9907ba55f",
330
+ "size": 4482
331
+ },
332
+ "source/results/20260916-fleet-setup/selection_migration.json": {
333
+ "sha256": "37bc6f9606d3e204bcdf45e0cd50bea2e5713f8d639c98fbd39f92c665c527bc",
334
+ "size": 1389
335
+ },
336
+ "source/results/20260916-fleet-setup/spark-a-best-validation-selection.json": {
337
+ "sha256": "e6cd9400edd4d510fe0342f726aead056c075c0829a767400e1ffd335dc86b39",
338
+ "size": 19697
339
+ },
340
+ "source/results/20260916-fleet-setup/spark-a-campaign-plan.json": {
341
+ "sha256": "f04bbd72d5a7cc59980714f1ef674ec7c9af309d8a956a7e5a03b766c647af74",
342
+ "size": 777
343
+ },
344
+ "source/results/20260916-fleet-setup/spark-a-campaign-state-snapshot.json": {
345
+ "sha256": "417d9f1552cfa82ddc8ad49224b986ef5d58f89d942a73b338a6aa6646cb4ab9",
346
+ "size": 1421
347
+ },
348
+ "source/results/20260916-fleet-setup/spark-a-campaign-updates-verified.json": {
349
+ "sha256": "52eae7f877102c4c46a9cf3275a354784f0c95f13c9772cc578db12c101a7698",
350
+ "size": 4674
351
+ },
352
+ "source/results/20260916-fleet-setup/spark-a-candidate-launch.json": {
353
+ "sha256": "e35162501a6d207f7467b44cc27b3d3fb1133db4f212a0b84942271909bdda91",
354
+ "size": 815
355
+ },
356
+ "source/results/20260916-fleet-setup/spark-a-inherited-validation-selection.json": {
357
+ "sha256": "2659b2b3fe8b7c8d9e614b30c25a99ad7100cd007892fba9b4ea6629fde0d887",
358
+ "size": 19687
359
+ },
360
+ "source/results/20260916-fleet-setup/spark-a-initial-validation-selection.json": {
361
+ "sha256": "e6cd9400edd4d510fe0342f726aead056c075c0829a767400e1ffd335dc86b39",
362
+ "size": 19697
363
+ },
364
+ "source/results/20260916-fleet-setup/spark-a-inputs-verified.json": {
365
+ "sha256": "7e6e8642a752993aa7181dcfb96e016ca4dbdf454109ff99d7174efd5a3c6f5b",
366
+ "size": 2202
367
+ },
368
+ "source/results/20260916-fleet-setup/spark-a-launcher.json": {
369
+ "sha256": "422197489b6de521d73d746a8bf21cfabc9ac82da82dfa892cfd68ce3526fd34",
370
+ "size": 140
371
+ },
372
+ "source/results/20260916-fleet-setup/spark-a-model-copy-verified.json": {
373
+ "sha256": "ffed13b2c1945c091bf2ea586556f94ba5e1ff67e7aeb92e1cbd49c488ae2ef9",
374
+ "size": 2188
375
+ },
376
+ "source/results/20260916-fleet-setup/spark-a-pilot-updates-verified.json": {
377
+ "sha256": "0ba2ad5967dbc5fc2e7a01a4bbf7a4ec6f1a7fe50bb815d9cefc63f314d1b154",
378
+ "size": 2897
379
+ },
380
+ "source/results/20260916-fleet-setup/spark-a-pilot/best_validation_predictions.json": {
381
+ "sha256": "e671e1508185765552b0f933ba03f356be62143c531d8ef534457d34b1645c9b",
382
+ "size": 333136
383
+ },
384
+ "source/results/20260916-fleet-setup/spark-a-pilot/correctness_final.json": {
385
+ "sha256": "cb2649db1735dbcbc90f83d9f95520a7dd553471b7d0c88316205207ad66acb8",
386
+ "size": 355
387
+ },
388
+ "source/results/20260916-fleet-setup/spark-a-pilot/correctness_initial.json": {
389
+ "sha256": "5d6334a0c66ed567fa9709df4ea5b6040c88bb5258e864332e88be2bb9586634",
390
+ "size": 355
391
+ },
392
+ "source/results/20260916-fleet-setup/spark-a-pilot/data_filter.json": {
393
+ "sha256": "3a514e34a8a6f7e35776bafe5b19015d6c5cec0eaabea5558aa896471aba3b56",
394
+ "size": 2360
395
+ },
396
+ "source/results/20260916-fleet-setup/spark-a-pilot/initial_validation_predictions.json": {
397
+ "sha256": "e671e1508185765552b0f933ba03f356be62143c531d8ef534457d34b1645c9b",
398
+ "size": 333136
399
+ },
400
+ "source/results/20260916-fleet-setup/spark-a-pilot/manifest.json": {
401
+ "sha256": "18672b6c49227a08ba1303a4958a9fa08ae78b5fe59e30c99902e5d61ccaddf0",
402
+ "size": 12238
403
+ },
404
+ "source/results/20260916-fleet-setup/spark-a-pilot/summary.json": {
405
+ "sha256": "9bda8ed53ed42f7a8b4dea08acbb6c4c58635bf64dafa925b1a3575e9a4f89b4",
406
+ "size": 11499
407
+ },
408
+ "source/results/20260916-fleet-setup/spark-a-pilot/training.jsonl": {
409
+ "sha256": "fe3be4437a159b98ddc84fb2da04f42de44aa10eebd8e19029d92f4a9585bc8f",
410
+ "size": 2082
411
+ },
412
+ "source/results/20260916-fleet-setup/spark-a-pilot/validation.jsonl": {
413
+ "sha256": "ef002be6c6dd5a27fa35f555d0a3fb3544f885ed806e026edeb2491fbb9ccd84",
414
+ "size": 6191
415
+ },
416
+ "source/results/20260916-fleet-setup/spark-a-pilot/validation_step_000008_predictions.json": {
417
+ "sha256": "d520b46127014161384c46d856fcfa167d4bb667a369235678f2e96ab0e54029",
418
+ "size": 333027
419
+ },
420
+ "source/results/20260916-fleet-setup/spark-a-python-stack-verified.json": {
421
+ "sha256": "8391cccb0d72f8a9af7e631075435a83ba3582c8bdd30b6ead0a3e50d90c46d3",
422
+ "size": 1965
423
+ },
424
+ "source/results/20260916-fleet-setup/spark-a-resume-state-verified.json": {
425
+ "sha256": "18ad5ae7e27f1d7c20b4de4cc0537fd1d77cf665f00ec91c87df9872c328d3c6",
426
+ "size": 1008
427
+ },
428
+ "source/results/20260916-fleet-setup/spark-a-resume-verified.json": {
429
+ "sha256": "7bbd348d2aed3478ec8967913068ce50c85acc6853d2dd33c56e1772c26369f0",
430
+ "size": 625
431
+ },
432
+ "source/results/20260916-fleet-setup/spark-a-service-stop.json": {
433
+ "sha256": "b2091baa22a19691fd9e5f4ce17c4bf61bcc07f2bdecd67b42fcd8e22b9af907",
434
+ "size": 1923
435
+ },
436
+ "source/results/20260916-fleet-setup/spark-a-setup-status.json": {
437
+ "sha256": "9ebe7f313f592080899534e55fda097aece13e559bf6fde67a0489f191cbe50d",
438
+ "size": 102
439
+ },
440
+ "source/results/20260916-fleet-setup/spark-a-training-correctness-initial.json": {
441
+ "sha256": "cb2649db1735dbcbc90f83d9f95520a7dd553471b7d0c88316205207ad66acb8",
442
+ "size": 355
443
+ },
444
+ "source/results/20260916-fleet-setup/spark-a-training-manifest.json": {
445
+ "sha256": "839a92247fa8a861e853e46a18a97cebb9b4514b327562dc59066b8363fdf06a",
446
+ "size": 12984
447
+ },
448
+ "source/results/20260916-fleet-setup/spark-a-verification/data_filter.json": {
449
+ "sha256": "3a514e34a8a6f7e35776bafe5b19015d6c5cec0eaabea5558aa896471aba3b56",
450
+ "size": 2360
451
+ },
452
+ "source/results/20260916-fleet-setup/spark-a-verification/http_255_choices_response.json": {
453
+ "sha256": "36ce43c4fc4885a73daa7f57c27143f311d0a075621e8d2f1121a4cf64b815f4",
454
+ "size": 12050
455
+ },
456
+ "source/results/20260916-fleet-setup/spark-a-verification/http_long_context_response.json": {
457
+ "sha256": "762a8383eaec4cc0ca22738d8cc9abde026781c37468ec2b5bf3058fd0e487d5",
458
+ "size": 216
459
+ },
460
+ "source/results/20260916-fleet-setup/spark-a-verification/http_response.json": {
461
+ "sha256": "f43f195c9ac3b62714e4dc11f99b71ac7e56aec89f979a481298b8039bd7a24f",
462
+ "size": 859
463
+ },
464
+ "source/results/20260916-fleet-setup/spark-a-verification/manifest.json": {
465
+ "sha256": "6a598e188f2d829ac3b3f7762fd67047691ba2246bddf959e3a073b8640986f7",
466
+ "size": 264
467
+ },
468
+ "source/results/20260916-fleet-setup/spark-a-verification/reload_predictions.json": {
469
+ "sha256": "f56623fbe5d0c26e6433bea78c77ecd61f4459f1ae3b0b62b808dae4a836b2fd",
470
+ "size": 10480
471
+ },
472
+ "source/results/20260916-fleet-setup/spark-a-verification/stress.json": {
473
+ "sha256": "5471c772bf1f56ad0c2df20c329eccf77aa0161eeb5953295c4959c9e298cf06",
474
+ "size": 1169
475
+ },
476
+ "source/results/20260916-fleet-setup/spark-a-verification/verification.json": {
477
+ "sha256": "3835333780af23e0c3398aa22b4b0da207a399e26ffa868978a457c4370a7f02",
478
+ "size": 2300
479
+ },
480
+ "source/results/20260916-fleet-setup/spark-a-warmstart-verified.json": {
481
+ "sha256": "516d935684a1a747ea6d74389568507a2fe49743a4d5b11dca89524f5992eaaf",
482
+ "size": 634
483
+ },
484
+ "source/results/20260916-fleet-setup/spark-b-best-validation-selection.json": {
485
+ "sha256": "9d16f6144a3ff81ffbd1a6cf78078d4bd01fa683f8bbfd4028627591686958f6",
486
+ "size": 19690
487
+ },
488
+ "source/results/20260916-fleet-setup/spark-b-correctness-initial.json": {
489
+ "sha256": "39d23512e99743437b88b5b10110e501853976b5292581481b5d2a5b7f2c5cdd",
490
+ "size": 391
491
+ },
492
+ "source/results/20260916-fleet-setup/spark-b-current-status.json": {
493
+ "sha256": "00af691a392d04dc2b8359a4232f836ec8f96501153025acc91ff370bcc9f357",
494
+ "size": 2861
495
+ },
496
+ "source/results/20260916-fleet-setup/spark-b-data-filter.json": {
497
+ "sha256": "e60da06955fe9e8a70a3de1ddaeb261e53af8f3a5e1d13ee62932d3773a1be24",
498
+ "size": 1572
499
+ },
500
+ "source/results/20260916-fleet-setup/spark-b-evidence-index.json": {
501
+ "sha256": "8b6e4a1172667c614749a3135d060183eefbf701a4f0c011a57b452d0d9b4490",
502
+ "size": 6279
503
+ },
504
+ "source/results/20260916-fleet-setup/spark-b-http-255-choices-response.json": {
505
+ "sha256": "fe9c590782a9584a7ccca84efcce728918e6a0d53edb11a86de4ca06a9d0ec9c",
506
+ "size": 11977
507
+ },
508
+ "source/results/20260916-fleet-setup/spark-b-http-long-context-response.json": {
509
+ "sha256": "77537e1b07b9746e7c75c503de97d570bbedfdc10c4e3be602d9aca7dfa6fd98",
510
+ "size": 202
511
+ },
512
+ "source/results/20260916-fleet-setup/spark-b-http-response.json": {
513
+ "sha256": "2baaeafd62463d40bd7318f0a24bfa423897f1139de2c60ccd16d072fd4955c6",
514
+ "size": 844
515
+ },
516
+ "source/results/20260916-fleet-setup/spark-b-inherited-validation-selection.json": {
517
+ "sha256": "9d16f6144a3ff81ffbd1a6cf78078d4bd01fa683f8bbfd4028627591686958f6",
518
+ "size": 19690
519
+ },
520
+ "source/results/20260916-fleet-setup/spark-b-initial-validation-selection.json": {
521
+ "sha256": "9d16f6144a3ff81ffbd1a6cf78078d4bd01fa683f8bbfd4028627591686958f6",
522
+ "size": 19690
523
+ },
524
+ "source/results/20260916-fleet-setup/spark-b-model-copy-verified.json": {
525
+ "sha256": "d6c5d97275d80f93e0a0713f3801eed6b173046d9d5c95989fa91504ad5c5b3f",
526
+ "size": 2155
527
+ },
528
+ "source/results/20260916-fleet-setup/spark-b-plan.json": {
529
+ "sha256": "509505b3fcb2af1d50b16bfdecd5833a7691da217700923d080aced6d679016f",
530
+ "size": 777
531
+ },
532
+ "source/results/20260916-fleet-setup/spark-b-python-stack-verified.json": {
533
+ "sha256": "c1eecf2c6d3c1a664180abdc99161a2e0372af5b0dbfab4fc354ebb49a5abfb6",
534
+ "size": 1965
535
+ },
536
+ "source/results/20260916-fleet-setup/spark-b-reload-predictions.json": {
537
+ "sha256": "4cc860ca3e6a65a904d17609d74102e39283480fd4f1e5f69cee4d8c7dc135bb",
538
+ "size": 10455
539
+ },
540
+ "source/results/20260916-fleet-setup/spark-b-service-stop.json": {
541
+ "sha256": "8706294be28526fc6fb1ece61964a61085a32636e5cd399387a2bd6e3dd7012f",
542
+ "size": 1201
543
+ },
544
+ "source/results/20260916-fleet-setup/spark-b-setup-launch.json": {
545
+ "sha256": "a065853c9ad9d6f63475fcc4996b289c7bb914bf50129e06b38d820145499347",
546
+ "size": 915
547
+ },
548
+ "source/results/20260916-fleet-setup/spark-b-setup-stages.log": {
549
+ "sha256": "13216b31b4287dc33421e7a8264133b5615a15e463b45fffa740eacfd4dd843d",
550
+ "size": 260
551
+ },
552
+ "source/results/20260916-fleet-setup/spark-b-setup-status.json": {
553
+ "sha256": "3f11175e52779f7ffb1e1848f6a57e859d1a318b10c8424d2695464d060a1a82",
554
+ "size": 102
555
+ },
556
+ "source/results/20260916-fleet-setup/spark-b-startup-verification.json": {
557
+ "sha256": "9eeef85649a5765a9a465fd082d5ef71687afbe1c7c910b6b1395507b3faeb94",
558
+ "size": 9581
559
+ },
560
+ "source/results/20260916-fleet-setup/spark-b-stress.json": {
561
+ "sha256": "617709fb02a0098b6ccbfce17d5817cbacf680178cd0a2d50e6c02557b2160d8",
562
+ "size": 1162
563
+ },
564
+ "source/results/20260916-fleet-setup/spark-b-training-manifest.json": {
565
+ "sha256": "e6f39dc5c797a0c1f5aa6ca176509d4cb1ef176a98af0bb6ee0946a5acbbc09e",
566
+ "size": 10911
567
+ },
568
+ "source/results/20260916-fleet-setup/spark-b-verification-manifest.json": {
569
+ "sha256": "891e5a059daa63397bcd403bf4f2e4297c5aba983272d25b316ac6a68c3b9457",
570
+ "size": 264
571
+ },
572
+ "source/results/20260916-fleet-setup/spark-b-verification.json": {
573
+ "sha256": "0ee15bcf7d01d312b91ec935fefa9944338847cf8a1f67cabba79e1a141c0af3",
574
+ "size": 2279
575
+ },
576
+ "source/results/20260916T154714Z/base_token_yes_minus_no_predictions.json": {
577
+ "sha256": "5c6dc5ab1c21d68cff293fbf73cdea66447d9a7ad2c3aaaac976608c7a696ecd",
578
+ "size": 31072
579
+ },
580
+ "source/results/20260916T154714Z/calibrated_predictions.json": {
581
+ "sha256": "cfd85bd7ae72ac090f326a8fe534b242230dad9b8772ddb50d3bddda50e25e70",
582
+ "size": 33872
583
+ },
584
+ "source/results/20260916T154714Z/calibration.jsonl": {
585
+ "sha256": "09e51eeffa7cdc2faad78f3dc6fd7b994580aae7e0ac2ecf8f1ceafdd7f984a0",
586
+ "size": 16193
587
+ },
588
+ "source/results/20260916T154714Z/calibration_predictions.json": {
589
+ "sha256": "f0d1fb5ba5d9778ccc4db37830372ec2257ce76f6a082eb63b00ee6009f768be",
590
+ "size": 23027
591
+ },
592
+ "source/results/20260916T154714Z/exit_code": {
593
+ "sha256": "4355a46b19d348dc2f57c046f8ef63d4538ebb936000f3c9ee954a27460dd865",
594
+ "size": 2
595
+ },
596
+ "source/results/20260916T154714Z/initial_scalar_predictions.json": {
597
+ "sha256": "6aa8afae1afb92d4ebf273d80741f3b165e694a20280a1da2e9144425c2d8294",
598
+ "size": 28395
599
+ },
600
+ "source/results/20260916T154714Z/manifest.json": {
601
+ "sha256": "4b4a55380d16561b0703d2b6c1d94b05a7f4d129b193656c1003dc80548e6414",
602
+ "size": 3462
603
+ },
604
+ "source/results/20260916T154714Z/parity_diagnosis.json": {
605
+ "sha256": "060e44827c102ed8c4f5a317dea9822a94fa8d87d8d795fb4a596e9954947caa",
606
+ "size": 6482
607
+ },
608
+ "source/results/20260916T154714Z/run.log": {
609
+ "sha256": "1659885bb673aed6f5ad0249111c75edab218bb46aefa7e2d3c38f2bc381448c",
610
+ "size": 11626
611
+ },
612
+ "source/results/20260916T154714Z/test.jsonl": {
613
+ "sha256": "4c0c365566099ff0941b2da473d6d3cca445ef1174c931d9d93f9bb475679599",
614
+ "size": 22779
615
+ },
616
+ "source/results/20260916T154714Z/train.jsonl": {
617
+ "sha256": "5ff456f912cf22ddefd8e0e24baf3956f0f010cd4c605e186a36659b4b931ea0",
618
+ "size": 61325
619
+ },
620
+ "source/results/20260916T154714Z/trained_predictions.json": {
621
+ "sha256": "8ebd0aad15bd04b4399cb467c4bad2bca813345179f737fbe510031d05a2e8a8",
622
+ "size": 34043
623
+ },
624
+ "source/results/20260916T154714Z/training.json": {
625
+ "sha256": "e3b0a701faa2eb67a3bbb70f4662e3f71d1587a396443afa709ec5e548577723",
626
+ "size": 10673
627
+ },
628
+ "source/results/20260916T155124Z/base_token_yes_minus_no_predictions.json": {
629
+ "sha256": "14bdc2ffad677e3fd3b6c58fae74efe3a7530df29ac9e181c2379a0cd5085ce1",
630
+ "size": 33837
631
+ },
632
+ "source/results/20260916T155124Z/benchmark.json": {
633
+ "sha256": "01027c6338ff2e36e3b7808f47c4f279bc9a2e52f24efc0bf7fd5744b5b572d8",
634
+ "size": 7692
635
+ },
636
+ "source/results/20260916T155124Z/calibrated_predictions.json": {
637
+ "sha256": "807e1f999a9ee64fa5f93851d1ee29da2d0fd41ed23e44e46f6391fb1f822a2c",
638
+ "size": 33865
639
+ },
640
+ "source/results/20260916T155124Z/calibration.jsonl": {
641
+ "sha256": "09e51eeffa7cdc2faad78f3dc6fd7b994580aae7e0ac2ecf8f1ceafdd7f984a0",
642
+ "size": 16193
643
+ },
644
+ "source/results/20260916T155124Z/calibration_predictions.json": {
645
+ "sha256": "1f3a0277c66f27c5ff08028ce6a7ccd90b9ccd5f7105b74fb8a3ae69c532422d",
646
+ "size": 23026
647
+ },
648
+ "source/results/20260916T155124Z/correctness.json": {
649
+ "sha256": "566f735523c8d1e46e4b11534b436d11edf1a9b1aef90e380aa10024b2b19abe",
650
+ "size": 329
651
+ },
652
+ "source/results/20260916T155124Z/exit_code": {
653
+ "sha256": "9a271f2a916b0b6ee6cecb2426f0b3206ef074578be55d9bc94f6f3fe3ab86aa",
654
+ "size": 2
655
+ },
656
+ "source/results/20260916T155124Z/initial_scalar_predictions.json": {
657
+ "sha256": "6aa8afae1afb92d4ebf273d80741f3b165e694a20280a1da2e9144425c2d8294",
658
+ "size": 28395
659
+ },
660
+ "source/results/20260916T155124Z/manifest.json": {
661
+ "sha256": "51b417c25f4b6f169b214bd6e4662a59b59365fc7e048feaa555941a05208a61",
662
+ "size": 3609
663
+ },
664
+ "source/results/20260916T155124Z/metrics.json": {
665
+ "sha256": "f5ecee19932286363c1e95bdb91c977dc907c9dc2a0f2bee85cfbb9c01472738",
666
+ "size": 6152
667
+ },
668
+ "source/results/20260916T155124Z/run.log": {
669
+ "sha256": "e0589922a98bdf412f749004cae387f8f60ef9c6f43f983de2cf433f6d41b6bc",
670
+ "size": 17478
671
+ },
672
+ "source/results/20260916T155124Z/test.jsonl": {
673
+ "sha256": "4c0c365566099ff0941b2da473d6d3cca445ef1174c931d9d93f9bb475679599",
674
+ "size": 22779
675
+ },
676
+ "source/results/20260916T155124Z/train.jsonl": {
677
+ "sha256": "5ff456f912cf22ddefd8e0e24baf3956f0f010cd4c605e186a36659b4b931ea0",
678
+ "size": 61325
679
+ },
680
+ "source/results/20260916T155124Z/trained_predictions.json": {
681
+ "sha256": "a0a020c3ec7134e82eab3fe22fe2e533d0dfe2393bf2e2f21c26020c09e5ba2a",
682
+ "size": 34016
683
+ },
684
+ "source/results/20260916T155124Z/training.json": {
685
+ "sha256": "6586a74826b7c0c7423a673cb7d9f6ce753dd097513802da18f975d769c9620a",
686
+ "size": 10643
687
+ },
688
+ "source/results/20260916T155314Z/benchmark.json": {
689
+ "sha256": "74fe266abe96c407487ccbe71ad4d30b78990b8c3382b68f64063ce76f85ee28",
690
+ "size": 7688
691
+ },
692
+ "source/results/20260916T155314Z/calibrated_predictions.json": {
693
+ "sha256": "aa4ffbf0fd363d2ba5a7f6d4ccff355dc636e88f8ef1b723eb2b0426e73aac46",
694
+ "size": 33876
695
+ },
696
+ "source/results/20260916T155314Z/calibration.jsonl": {
697
+ "sha256": "09e51eeffa7cdc2faad78f3dc6fd7b994580aae7e0ac2ecf8f1ceafdd7f984a0",
698
+ "size": 16193
699
+ },
700
+ "source/results/20260916T155314Z/calibration_predictions.json": {
701
+ "sha256": "eb2b47919f765cef2eb6d38ea1aff23e17ec2691f3b8052ee205b126130d7905",
702
+ "size": 22993
703
+ },
704
+ "source/results/20260916T155314Z/correctness.json": {
705
+ "sha256": "faf4077e01a6b0fede1988c98a586d71fc81c454b621a8159f9206d097aacce3",
706
+ "size": 335
707
+ },
708
+ "source/results/20260916T155314Z/exit_code": {
709
+ "sha256": "9a271f2a916b0b6ee6cecb2426f0b3206ef074578be55d9bc94f6f3fe3ab86aa",
710
+ "size": 2
711
+ },
712
+ "source/results/20260916T155314Z/manifest.json": {
713
+ "sha256": "51d594b45c7d67db6ea0be5a61d9fdf1d3d8b7e2b6805630e0a5bb255b734761",
714
+ "size": 3725
715
+ },
716
+ "source/results/20260916T155314Z/metrics.json": {
717
+ "sha256": "f0d51ea2136aff729163daccd56435636b751633a826f1e834cd4b332fb8e880",
718
+ "size": 4966
719
+ },
720
+ "source/results/20260916T155314Z/resume_verification.json": {
721
+ "sha256": "e9e80d9357d17c01f650f9b42cd358ed7193588736c98d91aaed21bc9ed30122",
722
+ "size": 773
723
+ },
724
+ "source/results/20260916T155314Z/resumed_initial_predictions.json": {
725
+ "sha256": "a0a020c3ec7134e82eab3fe22fe2e533d0dfe2393bf2e2f21c26020c09e5ba2a",
726
+ "size": 34016
727
+ },
728
+ "source/results/20260916T155314Z/run.log": {
729
+ "sha256": "05e8ef6eb53ff8fb01fb2571054501a4fd70b8d301e1dd35c81b248c02eb5769",
730
+ "size": 6914
731
+ },
732
+ "source/results/20260916T155314Z/test.jsonl": {
733
+ "sha256": "4c0c365566099ff0941b2da473d6d3cca445ef1174c931d9d93f9bb475679599",
734
+ "size": 22779
735
+ },
736
+ "source/results/20260916T155314Z/train.jsonl": {
737
+ "sha256": "5ff456f912cf22ddefd8e0e24baf3956f0f010cd4c605e186a36659b4b931ea0",
738
+ "size": 61325
739
+ },
740
+ "source/results/20260916T155314Z/trained_predictions.json": {
741
+ "sha256": "bdd004b8bb553cd446f159491f2be9c9e5e3f8f9b92061454ff20fb9850fcc92",
742
+ "size": 34004
743
+ },
744
+ "source/results/20260916T155314Z/training.json": {
745
+ "sha256": "6d56c059831aff521d59894702d7a66574d9e7ba8f965f745fa29e7531c5eb23",
746
+ "size": 180
747
+ },
748
+ "source/results/20260916T161253Z-precision/exit_code": {
749
+ "sha256": "9a271f2a916b0b6ee6cecb2426f0b3206ef074578be55d9bc94f6f3fe3ab86aa",
750
+ "size": 2
751
+ },
752
+ "source/results/20260916T161253Z-precision/manifest.json": {
753
+ "sha256": "7626809df6622a05507b11e65444e1d1b28e9a03ea81838169977991aabc6de5",
754
+ "size": 1169
755
+ },
756
+ "source/results/20260916T161253Z-precision/precision.json": {
757
+ "sha256": "5151de2f07119b2c6de007060899883f0cdb36ff728982df2b95de092d0097d8",
758
+ "size": 110342
759
+ },
760
+ "source/results/20260916T161253Z-precision/run.log": {
761
+ "sha256": "215e898ab1cf849c98ef56263e8d22a570d79263af2a39af925863adc394591f",
762
+ "size": 1655
763
+ },
764
+ "source/results/20260916T161355Z-precision/exit_code": {
765
+ "sha256": "9a271f2a916b0b6ee6cecb2426f0b3206ef074578be55d9bc94f6f3fe3ab86aa",
766
+ "size": 2
767
+ },
768
+ "source/results/20260916T161355Z-precision/manifest.json": {
769
+ "sha256": "93d8421244498c398f437adf4e6af0b4eb6b9b1a6f4d447564d1442faba56eae",
770
+ "size": 1103
771
+ },
772
+ "source/results/20260916T161355Z-precision/precision.json": {
773
+ "sha256": "d53276f38ba72bb01219d5e70183476cded0cd1b4bddb4cfb1ddddb9ec212000",
774
+ "size": 153559
775
+ },
776
+ "source/results/20260916T161355Z-precision/run.log": {
777
+ "sha256": "6696cda7c457d6c5c178d3adc4301b254dfe406b22dcfa8f0975c8b055755097",
778
+ "size": 2359
779
+ },
780
+ "source/results/20260916T182352Z-train/correctness_final.json": {
781
+ "sha256": "55e085c42e416d00fd4cdad9fb72d121f780fb9005226549ce7eaeddc9eea051",
782
+ "size": 408
783
+ },
784
+ "source/results/20260916T182352Z-train/correctness_initial.json": {
785
+ "sha256": "7414b27fe26d2d52ea023ba5cc4d8bc9a46d9bf60a8d54358ad2ad803c03c26a",
786
+ "size": 479
787
+ },
788
+ "source/results/20260916T182352Z-train/data_filter.json": {
789
+ "sha256": "e60da06955fe9e8a70a3de1ddaeb261e53af8f3a5e1d13ee62932d3773a1be24",
790
+ "size": 1572
791
+ },
792
+ "source/results/20260916T182352Z-train/initial_validation_predictions.json": {
793
+ "sha256": "4ee92802d7a8d642c8801dfd96f5bb18df80f745632fc1c8d99f9da9fe731c5b",
794
+ "size": 332151
795
+ },
796
+ "source/results/20260916T182352Z-train/manifest.json": {
797
+ "sha256": "e1efa2f4d350c678ba726e88dc85d8e7f3c8dc4e698742897892e578e5d94c58",
798
+ "size": 9460
799
+ },
800
+ "source/results/20260916T182352Z-train/resume_verification.json": {
801
+ "sha256": "501cb2750d0d5ba914dd066bd1312b39ec1e476ab8c074ac1b2758680f8a1928",
802
+ "size": 986
803
+ },
804
+ "source/results/20260916T182352Z-train/summary.json": {
805
+ "sha256": "05cf399210326916c92e404ed815ca8644bd3db09fa027cbfe4686fac47a902f",
806
+ "size": 11569
807
+ },
808
+ "source/results/20260916T182352Z-train/training.jsonl": {
809
+ "sha256": "231253f8237df8a40c95e6550fc36975b12b3b657e80da72c98a7712d365e48d",
810
+ "size": 10396
811
+ },
812
+ "source/results/20260916T182352Z-train/validation.jsonl": {
813
+ "sha256": "fdc0ad4f5d854f2ea33e45e644117e16d831d01576b9f011ccbca75745bc6c9b",
814
+ "size": 6202
815
+ },
816
+ "source/results/20260916T182352Z-train/validation_step_000040_predictions.json": {
817
+ "sha256": "d44337f6e32bd8f1e9db01f533ce031dafec5a6162fe43d19dde40e0a3c39c75",
818
+ "size": 332208
819
+ },
820
+ "source/results/20260916T183240Z-train/correctness_final.json": {
821
+ "sha256": "f941e3b1c39b5636b59ac513a9a0673ef683d708c52e505cb95b8271163efaf8",
822
+ "size": 391
823
+ },
824
+ "source/results/20260916T183240Z-train/correctness_initial.json": {
825
+ "sha256": "39d23512e99743437b88b5b10110e501853976b5292581481b5d2a5b7f2c5cdd",
826
+ "size": 391
827
+ },
828
+ "source/results/20260916T183240Z-train/manifest.json": {
829
+ "sha256": "ea6f9690fed03a43d2698161c3aae1685aca289cffd7f430e4ade5c32d16bb2d",
830
+ "size": 9818
831
+ },
832
+ "source/results/20260916T183240Z-train/summary.json": {
833
+ "sha256": "873e71eda53e3d000d805faa86d5a436611e97d74f331f89b9ed08d055f116a4",
834
+ "size": 11522
835
+ },
836
+ "source/results/20260916T183240Z-train/training.jsonl": {
837
+ "sha256": "7ea9596dd9503513d55715f4d0dee5f3aa8e22bd291288c4fde32511f8691677",
838
+ "size": 261
839
+ },
840
+ "source/results/20260916T183240Z-train/validation.jsonl": {
841
+ "sha256": "5f3e0dc2ed0e96ebff2673c637ff167c9f7d14691e1b98770f28b55467cc818a",
842
+ "size": 6190
843
+ },
844
+ "source/results/20260916T183751Z-verify2b/data_filter.json": {
845
+ "sha256": "e60da06955fe9e8a70a3de1ddaeb261e53af8f3a5e1d13ee62932d3773a1be24",
846
+ "size": 1572
847
+ },
848
+ "source/results/20260916T183751Z-verify2b/http_response.json": {
849
+ "sha256": "b8715eaf3ec74bd24569243798586f16a6ef7dffe330e98a54f22f094fe58735",
850
+ "size": 844
851
+ },
852
+ "source/results/20260916T183751Z-verify2b/manifest.json": {
853
+ "sha256": "e054c24da609c97b50fd3ecce33637bcc64f51c42ad723149bbbc2da786eeb70",
854
+ "size": 265
855
+ },
856
+ "source/results/20260916T183751Z-verify2b/reload_predictions.json": {
857
+ "sha256": "4e1ffbed45af3820bcf8c10d7638c925178e45284e9fc3cd86359ee230b80a67",
858
+ "size": 10453
859
+ },
860
+ "source/results/20260916T183751Z-verify2b/stress.json": {
861
+ "sha256": "efb4c0bc18cfca61511c7026a5b12e39fa423c00f0bfe63dd77343d03a52cbe5",
862
+ "size": 1163
863
+ },
864
+ "source/results/20260916T183751Z-verify2b/verification.json": {
865
+ "sha256": "49d95feaf1c72c076027b0aac71af26c80ecb6c7446755f88bd9171b79e6230c",
866
+ "size": 1920
867
+ },
868
+ "source/results/20260916T183823Z-train/correctness_final.json": {
869
+ "sha256": "5d6334a0c66ed567fa9709df4ea5b6040c88bb5258e864332e88be2bb9586634",
870
+ "size": 355
871
+ },
872
+ "source/results/20260916T183823Z-train/correctness_initial.json": {
873
+ "sha256": "6098996918d3fba844ad759459abb8063c2b7fdf9c65a891d7821a1d50dd9866",
874
+ "size": 423
875
+ },
876
+ "source/results/20260916T183823Z-train/data_filter.json": {
877
+ "sha256": "3a514e34a8a6f7e35776bafe5b19015d6c5cec0eaabea5558aa896471aba3b56",
878
+ "size": 2360
879
+ },
880
+ "source/results/20260916T183823Z-train/initial_validation_predictions.json": {
881
+ "sha256": "1189bd4f4d964b97e6fbeb8a10f8eaeb2c26e314e50b0a319b7ece322b325810",
882
+ "size": 324764
883
+ },
884
+ "source/results/20260916T183823Z-train/initial_validation_temperature_diagnostic.json": {
885
+ "sha256": "c466c9cb9b02206c5d1dab80db4fa579805f1ef890f549e6c27758240c35deda",
886
+ "size": 340
887
+ },
888
+ "source/results/20260916T183823Z-train/manifest.json": {
889
+ "sha256": "2cd93249a3c129203ddf3bb75827f810b30a61863e4dbf1bd5d2a1d7094419b8",
890
+ "size": 11460
891
+ },
892
+ "source/results/20260916T183823Z-train/summary.json": {
893
+ "sha256": "75044924b3f3b5a9b777627bccb1d80c7565b76a4520cc9c04228b0c7be93560",
894
+ "size": 11209
895
+ },
896
+ "source/results/20260916T183823Z-train/training.jsonl": {
897
+ "sha256": "2d425555511db013cae9ecafb28e2e1c7364c4dfcecd02b187f9c8c561a11e9e",
898
+ "size": 10422
899
+ },
900
+ "source/results/20260916T183823Z-train/validation.jsonl": {
901
+ "sha256": "964474c2ec7d6ee9c0f66055f51322944cc75f73a34adb5293849a3be6c1586c",
902
+ "size": 6173
903
+ },
904
+ "source/results/20260916T183823Z-train/validation_step_000040_predictions.json": {
905
+ "sha256": "e671e1508185765552b0f933ba03f356be62143c531d8ef534457d34b1645c9b",
906
+ "size": 333136
907
+ },
908
+ "source/results/20260916T185718Z-verify4b/data_filter.json": {
909
+ "sha256": "3a514e34a8a6f7e35776bafe5b19015d6c5cec0eaabea5558aa896471aba3b56",
910
+ "size": 2360
911
+ },
912
+ "source/results/20260916T185718Z-verify4b/http_255_choices_response.json": {
913
+ "sha256": "965b21465ccebb55e8c1b468da6e57653fe55dab5f2b48dec4ad207fdc322bfb",
914
+ "size": 12061
915
+ },
916
+ "source/results/20260916T185718Z-verify4b/http_long_context_response.json": {
917
+ "sha256": "fde2f6a2b43c93f77b292a5b89899254a7eee882598056a2a318d722c2b66f61",
918
+ "size": 216
919
+ },
920
+ "source/results/20260916T185718Z-verify4b/http_response.json": {
921
+ "sha256": "21ea4ea6c3e2c1c9e4383f5c5e340aaf6ee01b0903a9ce4d24672df63cf674d9",
922
+ "size": 861
923
+ },
924
+ "source/results/20260916T185718Z-verify4b/manifest.json": {
925
+ "sha256": "1fea5e941a3e38e11dc4e071c1dacb5de7c25345032207a95bb351e6cc211351",
926
+ "size": 265
927
+ },
928
+ "source/results/20260916T185718Z-verify4b/reload_predictions.json": {
929
+ "sha256": "2fee3e52111c3cd92babe0e36e5f2add009a4f69a3272a5fbacf7577026bc7f5",
930
+ "size": 10470
931
+ },
932
+ "source/results/20260916T185718Z-verify4b/stress.json": {
933
+ "sha256": "d93f33a15430685cc6357c9d2bdf08a10fa1e24d69d47fd58f66a4a9be7e46be",
934
+ "size": 1172
935
+ },
936
+ "source/results/20260916T185718Z-verify4b/verification.json": {
937
+ "sha256": "aede2c98c9773fd022e1ab807dfc66066a198dd01a9f35a6ca387e8e8341e48d",
938
+ "size": 2304
939
+ },
940
+ "source/results/20260916T185910Z-24h-launch/plan.json": {
941
+ "sha256": "979c1f0e0d4701c66115c209a25cf115d3210603a00d00ccd8b5ad027524b4fb",
942
+ "size": 649
943
+ },
944
+ "source/results/20260916T185910Z-24h-launch/resume_verification.json": {
945
+ "sha256": "772ab09829d7fb4403dcd7d3bb3685dee9e8559504b464191ff92f59a1a547d4",
946
+ "size": 933
947
+ },
948
+ "source/results/20260916T185910Z-24h-launch/state_snapshot.json": {
949
+ "sha256": "24cc3d6e9e17be546e68aea7654b4244d5b6bd883f5cf99ddb9e0c67cf87d7fa",
950
+ "size": 1328
951
+ },
952
+ "source/results/20260916T185910Z-24h-launch/training_correctness_initial.json": {
953
+ "sha256": "5d6334a0c66ed567fa9709df4ea5b6040c88bb5258e864332e88be2bb9586634",
954
+ "size": 355
955
+ },
956
+ "source/results/20260916T185910Z-24h-launch/training_manifest.json": {
957
+ "sha256": "b7c2e6e05d86f9c01a2aebd573e07ccd39da74bf25fd11e09d1150dc3326a619",
958
+ "size": 11618
959
+ },
960
+ "source/results/20260917-expanded-data/campaign-launch-state.json": {
961
+ "sha256": "51c8875e38951544da389d7eef7dd6a4277a66edc62280ca70dc7e740589dbc7",
962
+ "size": 1441
963
+ },
964
+ "source/results/20260917-expanded-data/campaign-plan.json": {
965
+ "sha256": "287680a47b1ed211396c4287d7420a58691d7134573bb482989a70979d7abd0b",
966
+ "size": 777
967
+ },
968
+ "source/results/20260917-expanded-data/campaign-step8-proof.json": {
969
+ "sha256": "0a81426bba7d4ff1d2f9a9c138810b08386f6a2b7818d393451e91565e7e175f",
970
+ "size": 2144
971
+ },
972
+ "source/results/20260917-expanded-data/data_filter.json": {
973
+ "sha256": "f07eef84b3081ad86bb5b48f810bbed569a76ee8e82cba9228beec908232d79e",
974
+ "size": 2462
975
+ },
976
+ "source/results/20260917-expanded-data/dataset-manifest.json": {
977
+ "sha256": "fde6ee7ce2eca20cb22cdbbe4db0ddbdb29a8ea9d597906d88e545939a5b602c",
978
+ "size": 12761
979
+ },
980
+ "source/results/20260917-expanded-data/expanded-pilot-backup-verification.json": {
981
+ "sha256": "c87bd718f4f87e3ad8abcb6ac700cab22c920bdf96e7ec3bbd5c0b0362e4c2b3",
982
+ "size": 1822
983
+ },
984
+ "source/results/20260917-expanded-data/fleet-plan.json": {
985
+ "sha256": "657979189140a298f383cbcf722c425c64f7b07b7e0332c32ac1b076445028e2",
986
+ "size": 2790
987
+ },
988
+ "source/results/20260917-expanded-data/fleet-registration.json": {
989
+ "sha256": "942085a629dd14c650cc3a71445a936da079e7fcb2413bb5d57768ce38b72081",
990
+ "size": 1544
991
+ },
992
+ "source/results/20260917-expanded-data/fleet-startup-status.json": {
993
+ "sha256": "ea8fea26861f88ebad35520e97558d7ac3ff3b804cb65175355fd1c4d69b8875",
994
+ "size": 3909
995
+ },
996
+ "source/results/20260917-expanded-data/gx10-original-stopped.json": {
997
+ "sha256": "d74f24127430776f1b2ce144c6f19a253dab03087a4d40881771d8dbe55f4d26",
998
+ "size": 14269
999
+ },
1000
+ "source/results/20260917-expanded-data/parent-snapshot.json": {
1001
+ "sha256": "fa43228a19a32fc2caf5480799a2746b4da619b42406c0f3898d36059107050f",
1002
+ "size": 11628
1003
+ },
1004
+ "source/results/20260917-expanded-data/pilot-correctness_final.json": {
1005
+ "sha256": "d5a9fc3e02e6e5d204d4c5154c21d30dcf92aa7621620f1ea7f64e824a4ecfef",
1006
+ "size": 355
1007
+ },
1008
+ "source/results/20260917-expanded-data/pilot-correctness_initial.json": {
1009
+ "sha256": "cd6b7a551708c15a09099a58d7863fc0eee17e8b5ff36fd3ed2fdb7147ffbf3a",
1010
+ "size": 356
1011
+ },
1012
+ "source/results/20260917-expanded-data/pilot-initial-prediction-parity.json": {
1013
+ "sha256": "cfd1ee30a1eefafad89f9df963cbc2a5b881c5f92cdfaca0eed5d23046c74d93",
1014
+ "size": 149
1015
+ },
1016
+ "source/results/20260917-expanded-data/pilot-launch.json": {
1017
+ "sha256": "27725f203df163ac92a29ddf936eb7f4c71f00205f04652c5d3149ee3ba39802",
1018
+ "size": 1319
1019
+ },
1020
+ "source/results/20260917-expanded-data/pilot-manifest.json": {
1021
+ "sha256": "80dc3efef131f59bc7bb5005bc6d1de46350c405604dd3a710aa1f2d34c3762b",
1022
+ "size": 14253
1023
+ },
1024
+ "source/results/20260917-expanded-data/pilot-source-data-proof.json": {
1025
+ "sha256": "c1b06da73c64320e3b06f25b3c0e17667cc5c989aa91b5cb5f832fa536e2a6b1",
1026
+ "size": 1205
1027
+ },
1028
+ "source/results/20260917-expanded-data/pilot-step0-proof.json": {
1029
+ "sha256": "7726a7a4aa15ba8c0d39f43588135f785daa7f77aa3d261dbe30f6074468247c",
1030
+ "size": 2580
1031
+ },
1032
+ "source/results/20260917-expanded-data/pilot-step8-optimizer-proof.json": {
1033
+ "sha256": "6eaddcfd827bb789e3ffc1aaf75069e6f52f811c5534977ad9b5a264d59c3ec6",
1034
+ "size": 591
1035
+ },
1036
+ "source/results/20260917-expanded-data/pilot-summary.json": {
1037
+ "sha256": "bc5e3ad91ecf3ef13ed3b82e408ebdd7110102a924f8b046e5ac3631f7fcdda7",
1038
+ "size": 11417
1039
+ },
1040
+ "source/results/20260917-expanded-data/pilot-training.jsonl": {
1041
+ "sha256": "c8685bc7d4674014ffdff7a36a71fbfdea18cc091269650fff748c10fd6e72f3",
1042
+ "size": 2086
1043
+ },
1044
+ "source/results/20260917-expanded-data/pilot-validation.jsonl": {
1045
+ "sha256": "3d617fed3f0d250c872da58d55b67210b02332c39675181e2691afebc74cca88",
1046
+ "size": 21176
1047
+ },
1048
+ "source/results/20260917-expanded-data/pilot-verification-repository.json": {
1049
+ "sha256": "2b4322b2cc16568f1015aa428639810116dfafaf2b9025236ca255bb5cdbc1da",
1050
+ "size": 2857
1051
+ },
1052
+ "source/results/20260917-expanded-data/postbuild-audit.json": {
1053
+ "sha256": "a72ee77ba11017679548b06a2b956f7d2e56ca07e95eba165bb48aefbb75da18",
1054
+ "size": 6956
1055
+ },
1056
+ "source/results/20260917-expanded-data/tests.json": {
1057
+ "sha256": "84745faf8106f084c6a3675ccd08c8c6e958b765b9e126f79362eb6fdb3f7221",
1058
+ "size": 273
1059
+ },
1060
+ "source/results/20260917-expanded-data/tokenization-proof.json": {
1061
+ "sha256": "c2c506ad72513d573eec723adb3960a00f9416fb974e9359f7b61b108f1ae2bb",
1062
+ "size": 2220
1063
+ },
1064
+ "source/results/20260917-fleet-progress/ensemble-reference.json": {
1065
+ "sha256": "3db8094bb4e01c2dfe74e880754b82c3ab08800529e357bdb4f1678beb21d40a",
1066
+ "size": 6229
1067
+ },
1068
+ "source/results/20260917-fleet-progress/fixed-ensemble-validation.json": {
1069
+ "sha256": "6c5dfa9d3528d4357cc89e90d10c7711b72eae50ddd67d67e15c4b902612c46d",
1070
+ "size": 11460
1071
+ },
1072
+ "source/results/20260917-fleet-progress/fleet-four-candidates-status.json": {
1073
+ "sha256": "2e2f2e6b8710562944b32c6d9b2619b415a6b79fd636dd85f1c1d1b48be186f4",
1074
+ "size": 3234
1075
+ },
1076
+ "source/results/20260917-fleet-progress/four-candidate-registration.json": {
1077
+ "sha256": "5d907228c8bc6ac48123359310bd5f56d567f2e72c30df3fd57d132c326f602d",
1078
+ "size": 3568
1079
+ },
1080
+ "source/results/20260917-fleet-progress/gx10-status.json": {
1081
+ "sha256": "3fb4d4c8d97d49d3f43460ff287bf8a619f863d4b93a9249b23b26f448b880a0",
1082
+ "size": 4242
1083
+ },
1084
+ "source/results/20260917-fleet-progress/hf-final-watcher-launch.json": {
1085
+ "sha256": "9654f1c752208de6c831fb7d6d9f1c0ec51c21f91351a148f9febd51aa579c41",
1086
+ "size": 2519
1087
+ },
1088
+ "source/results/20260917-fleet-progress/hf-final-watcher-relaunch.json": {
1089
+ "sha256": "a4e903e9eafeb4a911b0a532f8670b3b2b4f0cd9d324756009c078552a9c49d0",
1090
+ "size": 1732
1091
+ },
1092
+ "source/results/20260917-fleet-progress/hf-initial-artifacts-publication.json": {
1093
+ "sha256": "c8e938d34a13f17d3073c6ac4cb5de7b46f4cadf1e9f93f0e724766bf4f33e45",
1094
+ "size": 2261
1095
+ },
1096
+ "source/results/20260917-fleet-progress/hf-snapshot-publication.json": {
1097
+ "sha256": "686312f36485ada49c37bd2d1c129c39cdf86ff70316a87d65c0b0ef0cc363fc",
1098
+ "size": 1031
1099
+ },
1100
+ "source/results/20260917-fleet-progress/hf-write-auth-verified.json": {
1101
+ "sha256": "9e32210dd16078cf29f1339915f0712ba78925aadbe119d5a4c55b4f5512173f",
1102
+ "size": 316
1103
+ },
1104
+ "source/results/20260917-fleet-progress/refinement-parent.json": {
1105
+ "sha256": "f7c2765a5b9cf6fc794a30ec3b50bec986a046ed4deef080e3728b185c95a6ab",
1106
+ "size": 1417
1107
+ },
1108
+ "source/results/20260917-fleet-progress/spark-a-status-20260917T021317Z.json": {
1109
+ "sha256": "3c268f0e52bd5eb030f5d2799d51f6b33f52f750e7f277441c03ec83c24f93a4",
1110
+ "size": 11252
1111
+ },
1112
+ "source/results/20260917-fleet-progress/spark-b-2b-completed-audit.json": {
1113
+ "sha256": "53dea13fc44a987a071caa489ad2e832b0e434c516ada6de82423c5d2555aaef",
1114
+ "size": 482109
1115
+ },
1116
+ "source/results/20260917-fleet-progress/spark-b-2b-completed-summary.json": {
1117
+ "sha256": "16c665afdd0e3069ad2e18156da09c6dec8f58856b1ace9828feb559fd0e4620",
1118
+ "size": 8395
1119
+ },
1120
+ "source/results/20260917-fleet-progress/spark-b-evidence-index.json": {
1121
+ "sha256": "db281c7bfe8f023ed019038ab7c3e754fbdfda348a3eb5b2e9e4d90c73831b7c",
1122
+ "size": 6475
1123
+ },
1124
+ "source/results/20260917-fleet-progress/spark-b-readonly-summary.json": {
1125
+ "sha256": "16c665afdd0e3069ad2e18156da09c6dec8f58856b1ace9828feb559fd0e4620",
1126
+ "size": 8395
1127
+ },
1128
+ "source/results/20260917-fleet-progress/spark-b-refinement-best-validation-selection.json": {
1129
+ "sha256": "a458754f2b605558fecc8c6506349d9cdb1b7d6bd25fb4f701d481b1e0a778bb",
1130
+ "size": 19698
1131
+ },
1132
+ "source/results/20260917-fleet-progress/spark-b-refinement-campaign-launch.json": {
1133
+ "sha256": "203b65133d572f7b37bccfb6a4564ba5462efca651a797c3232d8c4420fede46",
1134
+ "size": 4381
1135
+ },
1136
+ "source/results/20260917-fleet-progress/spark-b-refinement-correctness-initial.json": {
1137
+ "sha256": "f9ad8b0308a09577abc390412f214cc6289b1108ced3907570c6d4cf58539158",
1138
+ "size": 354
1139
+ },
1140
+ "source/results/20260917-fleet-progress/spark-b-refinement-current-status.json": {
1141
+ "sha256": "a38095e1f9512022aad934a8db31d78985bfe0e336e4245c72203b6a5f68b7b2",
1142
+ "size": 4735
1143
+ },
1144
+ "source/results/20260917-fleet-progress/spark-b-refinement-data-filter.json": {
1145
+ "sha256": "3a514e34a8a6f7e35776bafe5b19015d6c5cec0eaabea5558aa896471aba3b56",
1146
+ "size": 2360
1147
+ },
1148
+ "source/results/20260917-fleet-progress/spark-b-refinement-http-255-choices-response.json": {
1149
+ "sha256": "b20caf2fb538c935bc5936c92c472082af58c54e4acd16dd2dc1447fe0b7d91f",
1150
+ "size": 11995
1151
+ },
1152
+ "source/results/20260917-fleet-progress/spark-b-refinement-http-long-context-response.json": {
1153
+ "sha256": "8d13e7a618fa1a77a1cafead1afc6eee038411b3e09ff49d3e22afcbf6c0aa23",
1154
+ "size": 215
1155
+ },
1156
+ "source/results/20260917-fleet-progress/spark-b-refinement-http-response.json": {
1157
+ "sha256": "cc339ccbd358dd410b52eaa9cb205971e8159eb3d2794ec2ee1a98f749943408",
1158
+ "size": 864
1159
+ },
1160
+ "source/results/20260917-fleet-progress/spark-b-refinement-inherited-validation-selection.json": {
1161
+ "sha256": "a458754f2b605558fecc8c6506349d9cdb1b7d6bd25fb4f701d481b1e0a778bb",
1162
+ "size": 19698
1163
+ },
1164
+ "source/results/20260917-fleet-progress/spark-b-refinement-initial-predictions-verified.json": {
1165
+ "sha256": "7817371003ac0883e10d6847fcd094cbc0f0ba9831ceedbb149e79aec9a23ccd",
1166
+ "size": 707
1167
+ },
1168
+ "source/results/20260917-fleet-progress/spark-b-refinement-initial-validation-selection.json": {
1169
+ "sha256": "9675dfe39457a171238d81083d2c1e61dcb5aefa2429e2f2c9567ab79047e6c3",
1170
+ "size": 19695
1171
+ },
1172
+ "source/results/20260917-fleet-progress/spark-b-refinement-inputs-verified.json": {
1173
+ "sha256": "8c5be28c56f988321f89dce076a1f6a834547795b2672b25d2d11ec60ad08bb4",
1174
+ "size": 3472
1175
+ },
1176
+ "source/results/20260917-fleet-progress/spark-b-refinement-pilot-correctness-final.json": {
1177
+ "sha256": "f9ad8b0308a09577abc390412f214cc6289b1108ced3907570c6d4cf58539158",
1178
+ "size": 354
1179
+ },
1180
+ "source/results/20260917-fleet-progress/spark-b-refinement-pilot-correctness-initial.json": {
1181
+ "sha256": "fd1b0d769f7a3f1cddacf00c19b0e3cec76c97c1d1e83a434ec2950bf178c677",
1182
+ "size": 355
1183
+ },
1184
+ "source/results/20260917-fleet-progress/spark-b-refinement-pilot-launch.json": {
1185
+ "sha256": "d85c147ec85752b94eddc95b82ebdc6b93024d0915bdc0723ee65e941fde207d",
1186
+ "size": 1419
1187
+ },
1188
+ "source/results/20260917-fleet-progress/spark-b-refinement-pilot-manifest.json": {
1189
+ "sha256": "e1ac6aaf86db2fff6c007a1018503877976ba04de7415125594142158b6f11a5",
1190
+ "size": 12394
1191
+ },
1192
+ "source/results/20260917-fleet-progress/spark-b-refinement-pilot-summary.json": {
1193
+ "sha256": "929d4851d5b829a4cd69d337ab858f4f78480d6a654968d83141583c846fcec4",
1194
+ "size": 11392
1195
+ },
1196
+ "source/results/20260917-fleet-progress/spark-b-refinement-pilot-training.jsonl": {
1197
+ "sha256": "6865669d51de01930927a807d94d34b78499004fbf6dfd78cba55ab26554ecb7",
1198
+ "size": 2114
1199
+ },
1200
+ "source/results/20260917-fleet-progress/spark-b-refinement-pilot-validation.jsonl": {
1201
+ "sha256": "704b478c575080d30e1e555303b96b9307212d91c071673137f602a5b1db9fb6",
1202
+ "size": 21145
1203
+ },
1204
+ "source/results/20260917-fleet-progress/spark-b-refinement-pilot-verified.json": {
1205
+ "sha256": "f39eebaabb44ff03e02353c93acdf7dafe512747b3778d4b4c553a7ceb7b78d4",
1206
+ "size": 15161
1207
+ },
1208
+ "source/results/20260917-fleet-progress/spark-b-refinement-plan.json": {
1209
+ "sha256": "3e146d2daa7fa8e12689bebcfc8f3c8e27942328e15c2cf978a1950db21bb086",
1210
+ "size": 777
1211
+ },
1212
+ "source/results/20260917-fleet-progress/spark-b-refinement-reload-predictions.json": {
1213
+ "sha256": "3c61f093082b06bf5274036d6e5f428465664ec9ae8125afc9ff99c781fc498a",
1214
+ "size": 10396
1215
+ },
1216
+ "source/results/20260917-fleet-progress/spark-b-refinement-resume-verified.json": {
1217
+ "sha256": "0a2f072c250d91db2f419bd95c0c9eaee113229d3598f655d61ff05524773fcf",
1218
+ "size": 3785
1219
+ },
1220
+ "source/results/20260917-fleet-progress/spark-b-refinement-setup-status.json": {
1221
+ "sha256": "c60e76f3a6a71b8893d1223db7091be667b3a95e02d9c919011ce4873b4b66e9",
1222
+ "size": 199
1223
+ },
1224
+ "source/results/20260917-fleet-progress/spark-b-refinement-startup-verified.json": {
1225
+ "sha256": "97d75c66c1c1d76843cb2cb3a326780c452305501211da3b2ab3abef5a537b1b",
1226
+ "size": 52576
1227
+ },
1228
+ "source/results/20260917-fleet-progress/spark-b-refinement-stress.json": {
1229
+ "sha256": "edc899aa2db8fae51a01818f0655a78b359b83dc521b3f7f399e458ffa5eeec7",
1230
+ "size": 1166
1231
+ },
1232
+ "source/results/20260917-fleet-progress/spark-b-refinement-training-manifest.json": {
1233
+ "sha256": "0fe2c1bde4a47879d3ba040b97aa26c3025101dec10192e86188fe2cd98c33cf",
1234
+ "size": 13002
1235
+ },
1236
+ "source/results/20260917-fleet-progress/spark-b-refinement-verification-launch.json": {
1237
+ "sha256": "ff87b649b1d2a9bb13d27e0bc8456d3fd5cd4076719e6833121ff69ca0775706",
1238
+ "size": 1977
1239
+ },
1240
+ "source/results/20260917-fleet-progress/spark-b-refinement-verification-manifest.json": {
1241
+ "sha256": "997bee30fd9e1ea8420fb6671687b2e37c5aed76e6b1594521749bf73f35a01b",
1242
+ "size": 264
1243
+ },
1244
+ "source/results/20260917-fleet-progress/spark-b-refinement-verification.json": {
1245
+ "sha256": "5d1a8d155bb53e27d2ffbe087d8b2e31686503fa277588c889d04865be3869c0",
1246
+ "size": 2298
1247
+ },
1248
+ "source/results/20260917-fleet-progress/spark-b-refinement-warmstart-verified.json": {
1249
+ "sha256": "72d8df546f492770738205561ecdc6c8c063b02c60f551ab752f3aafdebd78a7",
1250
+ "size": 998
1251
+ },
1252
+ "source/results/20260917-fleet-progress/validation-trends.json": {
1253
+ "sha256": "ddecc44c77276d0a6ca4d2418ad6f08a2e8caf00b327422e93f3407343610ac9",
1254
+ "size": 4517
1255
+ },
1256
+ "source/results/20260917-playground-layout/1366x768-live-context.png": {
1257
+ "sha256": "5d4fa0c503119afa25f7b921e25d20a0a2ece929676c98e3e919737d0beb4ac2",
1258
+ "size": 88828
1259
+ },
1260
+ "source/results/20260917-playground-layout/320x568-live-context.png": {
1261
+ "sha256": "7e653a88d59187ea1c97927dfca0522f33ce4f84c06259efa9a646f43da6dd2f",
1262
+ "size": 45666
1263
+ },
1264
+ "source/results/20260917-playground-layout/390x360-live-choices.png": {
1265
+ "sha256": "2ad9367df087f6c90db0591a68bee040f9edec168af1cdad458dc8af7973b863",
1266
+ "size": 28895
1267
+ },
1268
+ "source/results/20260917-playground-layout/README.md": {
1269
+ "sha256": "eb43a78b91aef3d09f45d09ca726255bdadb99e63a8400cc8de2cc9644d3a0c9",
1270
+ "size": 1140
1271
+ },
1272
+ "source/results/20260917-playground-layout/baseline-overflow.json": {
1273
+ "sha256": "8180a1442584716e7dce6b7648a1ce9f76cc95e4f18abdbc6643b737ed495bfb",
1274
+ "size": 678
1275
+ },
1276
+ "source/results/20260917-playground-layout/fixture-1366x768-results.png": {
1277
+ "sha256": "39546d353d1baa4c46df77e63dd957acbfaa933ade53b47737d22a3fa8b4a736",
1278
+ "size": 93487
1279
+ },
1280
+ "source/results/20260917-playground-layout/fixture-320x568-results.png": {
1281
+ "sha256": "fb39de0b46b0892fe60ce9a9790ddd9081d060e91606895b7c52bb383d1348c7",
1282
+ "size": 40152
1283
+ },
1284
+ "source/results/20260917-playground-layout/fixture-844x390-choices.png": {
1285
+ "sha256": "e3bdfb431ebafac7a80e5d05ecbc2b9f000670450abcf3d8b5a5fdccd7a54f18",
1286
+ "size": 38029
1287
+ },
1288
+ "source/results/20260917-playground-layout/fixture-layout-verification.json": {
1289
+ "sha256": "ea0e2ba3e66b6aff3a15d11bec57318ea38c12316cf1f7288061e98fa2030567",
1290
+ "size": 5250
1291
+ },
1292
+ "source/results/20260917-playground-layout/live-layout.json": {
1293
+ "sha256": "6eb0b12867dd89ab898c1c6215e9995a4247531de1e55e227d99b9e742d3ed83",
1294
+ "size": 2154
1295
+ },
1296
+ "source/results/20260917-playground-layout/served-assets.json": {
1297
+ "sha256": "c831a0c6228d0ab92236368cf90cbd9651f266c8750d411d2f99b63f69c54533",
1298
+ "size": 754
1299
+ },
1300
+ "source/results/20260917-playground/browser-verification.json": {
1301
+ "sha256": "2806c5e75e560df7c0266e89aa92ac216bbca419be2094dee09e205d797ee460",
1302
+ "size": 3596
1303
+ },
1304
+ "source/results/20260917-playground/concurrent-training.json": {
1305
+ "sha256": "e581deac6f588ddf0a6effc595857a11bcbbf51979a5fdf812565e7611d608bc",
1306
+ "size": 11763
1307
+ },
1308
+ "source/results/20260917-playground/desktop.png": {
1309
+ "sha256": "0cbb5a6c9a9576ddceb95148024060350c5294779ed55063fb844f5978a3bf53",
1310
+ "size": 138954
1311
+ },
1312
+ "source/results/20260917-playground/frontend-fixture-check.json": {
1313
+ "sha256": "d368e15625343d2e94e3268a19e21fbd5f4a1cbe4994a4ef15b0542a83cfa1a6",
1314
+ "size": 1179
1315
+ },
1316
+ "source/results/20260917-playground/launch.json": {
1317
+ "sha256": "6af85b4328f8a253b3cc467290cad15eb0883268276eddeede0aa64d99395a54",
1318
+ "size": 1991
1319
+ },
1320
+ "source/results/20260917-playground/mobile.png": {
1321
+ "sha256": "e4711a2fa70b2d952e5f1b9e825dee5526c7eb6f837f99bf7f676aecd24c9af5",
1322
+ "size": 127118
1323
+ },
1324
+ "source/results/20260917-playground/models.json": {
1325
+ "sha256": "7a4e33a07801ab5fe918bb5a94c032900bd798513f6ab99608c9f06bb8c0572c",
1326
+ "size": 1251
1327
+ },
1328
+ "source/results/20260917-playground/port-7466.json": {
1329
+ "sha256": "8f10e4fd4a52386292235c814705ca11316836d4ed3f1c09311657ab211905f6",
1330
+ "size": 3827
1331
+ },
1332
+ "source/results/20260917-playground/snapshots.json": {
1333
+ "sha256": "a8187b1e1258b3c1ad50767c3e6b24d8e33b9533284f50347c8719c93ab1aea4",
1334
+ "size": 1386
1335
+ },
1336
+ "source/results/checkpoint-sha256.txt": {
1337
+ "sha256": "18fc16e95c38a28c1ec832f956c7ffc03fe9d0fdeb90348a84b73f2d9199937a",
1338
+ "size": 254
1339
+ },
1340
+ "source/results/public-decisions-v1-manifest.json": {
1341
+ "sha256": "adf5a8ca2bab60cf7429a82b1b7a2d1ae7d3cc812de04dd195a478dfd7c6f628",
1342
+ "size": 5266
1343
+ },
1344
+ "source/scripts/campaign_status.py": {
1345
+ "sha256": "1500b5e24f06231c7aafd7277aefd840582e35997e265f79db93614828d34411",
1346
+ "size": 2301
1347
+ },
1348
+ "source/scripts/diagnose_parity.py": {
1349
+ "sha256": "08b5d66d316ebda98a2226251a4f952701f86a1d5726ce7a4d7e8fb22755da5a",
1350
+ "size": 3511
1351
+ },
1352
+ "source/scripts/download_candidate.py": {
1353
+ "sha256": "d06a2c01be0cf6577f927fb37e3bc1eab014949fd934e4f4d9825674adba608e",
1354
+ "size": 1141
1355
+ },
1356
+ "source/scripts/download_model.py": {
1357
+ "sha256": "72ad9a5de44d09e2ee4ed8afb7c3c0ff6fb48987410a3f7f0368353572bf1f4d",
1358
+ "size": 1013
1359
+ },
1360
+ "source/scripts/fleet_campaign.py": {
1361
+ "sha256": "e69fdff96f92c6943b7895be11df444f016d0b744a1b9441995a5f8bb7af9d54",
1362
+ "size": 33706
1363
+ },
1364
+ "source/scripts/fleet_status.py": {
1365
+ "sha256": "2519ced157ef4ac4fa449eebabae5740d7527778d578b4ac6720583010fa5217",
1366
+ "size": 10289
1367
+ },
1368
+ "source/scripts/investigate_precision.py": {
1369
+ "sha256": "609b744a926d8a45b87ba8d225e5312ee0c71a7096b21bd3589c5846e1dc847c",
1370
+ "size": 10519
1371
+ },
1372
+ "source/scripts/launch_24h.py": {
1373
+ "sha256": "39c26dc10535d3adac09209b0743edf2ed384563e732512ad62f8c6f37161a83",
1374
+ "size": 19127
1375
+ },
1376
+ "source/scripts/prepare_expanded_data.py": {
1377
+ "sha256": "5c05478b29c84218784690f3c7c3ec994615fec1c57826591607189f191c56b4",
1378
+ "size": 18386
1379
+ },
1380
+ "source/scripts/prepare_expansion_backup.py": {
1381
+ "sha256": "0b546fd6b96e34316fddcb06314fa072d1ddb9c06d7e6979ccd790421b93f419",
1382
+ "size": 21154
1383
+ },
1384
+ "source/scripts/prepare_public_data.py": {
1385
+ "sha256": "32ea84aa719818e1141b958b6ef27a85f7ddb86bfcd7c1585fc487b25253d977",
1386
+ "size": 11886
1387
+ },
1388
+ "source/scripts/publish_hf_final.py": {
1389
+ "sha256": "278efc5878d7ebc5d1171f3a735d6c7d78275d93650e5a8a16c3f55b8353a577",
1390
+ "size": 20356
1391
+ },
1392
+ "source/scripts/publish_hf_snapshot.py": {
1393
+ "sha256": "b5fe16a00fcbc5ab97121428c6ce750275ee193438e3ffe24aac5e3bb325c018",
1394
+ "size": 15165
1395
+ },
1396
+ "source/scripts/run_experiment.sh": {
1397
+ "sha256": "661a6309fc54a2a8aff918f14a553c72dcb21730bd6a3cfc55d6ccd4700d11d3",
1398
+ "size": 1213
1399
+ },
1400
+ "source/scripts/run_precision.sh": {
1401
+ "sha256": "766d82b30cf3e83951f662685b4472ee053c7fbc171c4ff140823bf4d1c2782f",
1402
+ "size": 998
1403
+ },
1404
+ "source/scripts/run_smoke.sh": {
1405
+ "sha256": "39d59f2120f362729d1c2e384391b82be1e580dcc1aca0eaa7ab231115225574",
1406
+ "size": 1054
1407
+ },
1408
+ "source/scripts/start_spark_candidate.sh": {
1409
+ "sha256": "c2ca18b008a144de7cb264c9fcca634d68c4a8a317db3638e8a70b3dbcff064b",
1410
+ "size": 6545
1411
+ },
1412
+ "source/scripts/verify_artifact.py": {
1413
+ "sha256": "9833350e9d72c0065b15206bb71c5a8b5a6b3185219db90074e369ede563a985",
1414
+ "size": 9971
1415
+ },
1416
+ "source/scripts/verify_expanded_startup.py": {
1417
+ "sha256": "deabba820a3580e578d2d955d3ac3fd2e3982af999523c1d9a8b6dc6e2e1b755",
1418
+ "size": 17887
1419
+ },
1420
+ "source/scripts/verify_playground.cjs": {
1421
+ "sha256": "a55aa0e945baeb9e536c2b9cce7a9aa2364244b587e90477190add96f78601dc",
1422
+ "size": 6961
1423
+ },
1424
+ "source/scripts/verify_playground_layout.cjs": {
1425
+ "sha256": "dea3fa07c568c141cd58fead8c547db7196f4a48fe4ea4cea0312d2ba7bd8bb4",
1426
+ "size": 13341
1427
+ },
1428
+ "source/selection.py": {
1429
+ "sha256": "be0a7a8496b5b830aa572ceba93606320f442063fd38180503fd6980dc1c578f",
1430
+ "size": 5335
1431
+ },
1432
+ "source/smoke_data.py": {
1433
+ "sha256": "06b3cbac1c8c4a86b8aecbee4459073cc3e46d4ddcd576392f3cb4805924f815",
1434
+ "size": 1807
1435
+ },
1436
+ "source/smoke_train.py": {
1437
+ "sha256": "8cdeb2b397177fc9c26638aaa871501ddab8e3871aa1573ecd98f66590f5c228",
1438
+ "size": 19630
1439
+ },
1440
+ "source/tests/test_campaign.py": {
1441
+ "sha256": "90b132f655db9e9fd7b71c4916d47064c51e65cc1cc0de1a69f0b0cdc8d1cad7",
1442
+ "size": 8567
1443
+ },
1444
+ "source/tests/test_data_transition.py": {
1445
+ "sha256": "695fc2113cac41b22ba910682656a445c08a881bd80cc0bd72b9aeb58250cd05",
1446
+ "size": 3964
1447
+ },
1448
+ "source/tests/test_expanded_data.py": {
1449
+ "sha256": "f272faecf6aaccf94fb460b9ff5105916ca3a97319606f1dd9d16003465e5f16",
1450
+ "size": 7270
1451
+ },
1452
+ "source/tests/test_expanded_startup.py": {
1453
+ "sha256": "39f48547760823ea2818c1cae38d0bbafe430f9620f484f745b06712909fce22",
1454
+ "size": 6654
1455
+ },
1456
+ "source/tests/test_expansion_backup.py": {
1457
+ "sha256": "3845483f6469dc30122ae31a2bcbf6b1ad7ce69663308266f831142cd5d4fd1e",
1458
+ "size": 11522
1459
+ },
1460
+ "source/tests/test_fleet_campaign.py": {
1461
+ "sha256": "c5a64a4014c35b490c96d704f3762f3dcc0569a11ec6a1fe14507736c2c8cfe9",
1462
+ "size": 20309
1463
+ },
1464
+ "source/tests/test_fleet_status.py": {
1465
+ "sha256": "681eabe15b1de8380302a15f87cba411037da44c73a5e8b817455a7bb13dd061",
1466
+ "size": 3469
1467
+ },
1468
+ "source/tests/test_playground.py": {
1469
+ "sha256": "f63b58eef3b0f9c42aa3d445ceb4fa5462e936deb7b11e2e0fc962af6ed32c96",
1470
+ "size": 10663
1471
+ },
1472
+ "source/tests/test_publish_hf_final.py": {
1473
+ "sha256": "dd18fb43526d23e36166787a1d3eae5a12aedd6ffc54951459c97c37b1f2c3ee",
1474
+ "size": 15397
1475
+ },
1476
+ "source/tests/test_publish_hf_snapshot.py": {
1477
+ "sha256": "f48905828b51060f9a265505245eb10f814b47b8610ce5b01d67b2005c828edd",
1478
+ "size": 9332
1479
+ },
1480
+ "source/tests/test_scorer.py": {
1481
+ "sha256": "2c6f5d5e9ff634049cbe9c88b5f126710298b65ce603d7ced2072bc06e3974d8",
1482
+ "size": 5091
1483
+ },
1484
+ "source/tests/test_selection.py": {
1485
+ "sha256": "09fc10fd394ea870307175698a81b516a0bafc536757be918f5cacb86d124770",
1486
+ "size": 4090
1487
+ },
1488
+ "source/tests/test_training_harness.py": {
1489
+ "sha256": "8b8445144c62fa7d7b947e3e858e5f5d732ca31cf2fae4372349381b564e76bb",
1490
+ "size": 23833
1491
+ },
1492
+ "source/training_model.py": {
1493
+ "sha256": "d5b0aefeeb5290816bc0b669aa0a8cbbe27f6a12b9cb23c141ac9b9ae9ee4e65",
1494
+ "size": 7862
1495
+ },
1496
+ "source/web/playground.css": {
1497
+ "sha256": "4cd80320c0218d4e51a1c0193048eacc63dbe62f8b24d0bceef9179e62f34211",
1498
+ "size": 20388
1499
+ },
1500
+ "source/web/playground.html": {
1501
+ "sha256": "dedfed05a7548af47cd7d51d76b730c99e6c4fc89bcc4b8641b4b1f50dfc8537",
1502
+ "size": 9539
1503
+ },
1504
+ "source/web/playground.js": {
1505
+ "sha256": "bec1dc7195f34af2cf3dd44fe66b2b4d01480a378a7ce0424f69a854393ab009",
1506
+ "size": 22258
1507
+ },
1508
+ "sources/24b8ccf60d388f9cbb184e03a6ae260a1f5a8b86/manifest.json": {
1509
+ "sha256": "20dda62c84fc09dbed64be6698ee64659324ee8b8df69b6b2e3f4817616268be",
1510
+ "size": 52672
1511
+ },
1512
+ "sources/24b8ccf60d388f9cbb184e03a6ae260a1f5a8b86/source.tar.gz": {
1513
+ "sha256": "de8f0338165c9300aa22a49a759381e873f3faf0fdc997192e8226862413e66b",
1514
+ "size": 2005098
1515
+ },
1516
+ "sources/4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1/manifest.json": {
1517
+ "sha256": "44f77f5d586ff77ce6cb7ddffe901d1f93b1afb4aeef322024d7de209faf0771",
1518
+ "size": 23766
1519
+ },
1520
+ "sources/4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1/source.tar.gz": {
1521
+ "sha256": "6f76ba9b5ee699249f5258d52ef05f372aea3f425dc951c2681b1b4a1f2be918",
1522
+ "size": 700511
1523
+ },
1524
+ "sources/fe7e48f806a7e947d689b9e966d38bc0dbedcc2e/manifest.json": {
1525
+ "sha256": "802684c831e937fc025ecb3b226ecf7371e7d3eddc13a1e312457dd3bbbf4f51",
1526
+ "size": 57137
1527
+ },
1528
+ "sources/fe7e48f806a7e947d689b9e966d38bc0dbedcc2e/source.tar.gz": {
1529
+ "sha256": "39eebc9c3e32a2f85460fdac90aeefb2f2e0668425ecd0a64ad65f1a5d468270",
1530
+ "size": 2040983
1531
+ }
1532
+ },
1533
+ "format": "opensysone-snapshot-publication-v1",
1534
+ "repo_id": "andyshu/opensysone",
1535
+ "snapshot_id": "20260917-expanded-pilot-hf-backup",
1536
+ "snapshot_path": "snapshots/20260917-expanded-pilot-hf-backup",
1537
+ "source_commit": "fe7e48f806a7e947d689b9e966d38bc0dbedcc2e",
1538
+ "training_source_commits": [
1539
+ "24b8ccf60d388f9cbb184e03a6ae260a1f5a8b86",
1540
+ "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1"
1541
+ ]
1542
+ }
snapshots/20260917-expanded-pilot-hf-backup/SHA256SUMS ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 5b9eccc4c4e2bf1306e2def3e663b66e9f0cd1f3791dff04313039eb07426be5 artifacts/expanded-gx10-4b-pilot/best.pt
2
+ ced9908b7a6a9d6f682c7afac42d12316467b7a8aa83dcfd4b09b6af8aee6d6e artifacts/expanded-gx10-4b-pilot/best_validation_selection.json
3
+ cf4cd775165cc2bb02113d17b94facd0db7b94fcff87b27389b3280fbb712469 artifacts/expanded-gx10-4b-pilot/checkpoint.pt
4
+ d5a9fc3e02e6e5d204d4c5154c21d30dcf92aa7621620f1ea7f64e824a4ecfef artifacts/expanded-gx10-4b-pilot/correctness_final.json
5
+ cd6b7a551708c15a09099a58d7863fc0eee17e8b5ff36fd3ed2fdb7147ffbf3a artifacts/expanded-gx10-4b-pilot/correctness_initial.json
6
+ f07eef84b3081ad86bb5b48f810bbed569a76ee8e82cba9228beec908232d79e artifacts/expanded-gx10-4b-pilot/data_filter.json
7
+ 9a271f2a916b0b6ee6cecb2426f0b3206ef074578be55d9bc94f6f3fe3ab86aa artifacts/expanded-gx10-4b-pilot/exit_code
8
+ 80dc3efef131f59bc7bb5005bc6d1de46350c405604dd3a710aa1f2d34c3762b artifacts/expanded-gx10-4b-pilot/manifest.json
9
+ bc5e3ad91ecf3ef13ed3b82e408ebdd7110102a924f8b046e5ac3631f7fcdda7 artifacts/expanded-gx10-4b-pilot/summary.json
10
+ 1e0a2f838251c8eb411b59f4fa061db67ec308ce249d166901760b097c64bd6c artifacts/expanded-gx10-4b-pilot/validation_step_000000_predictions.json
11
+ 5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024 artifacts/warm-start-parent/best.pt
12
+ ced9908b7a6a9d6f682c7afac42d12316467b7a8aa83dcfd4b09b6af8aee6d6e artifacts/warm-start-parent/best_validation_selection.json
13
+ f9ad8b0308a09577abc390412f214cc6289b1108ced3907570c6d4cf58539158 artifacts/warm-start-parent/correctness_initial.json
14
+ 3a514e34a8a6f7e35776bafe5b19015d6c5cec0eaabea5558aa896471aba3b56 artifacts/warm-start-parent/data_filter.json
15
+ 0fe2c1bde4a47879d3ba040b97aa26c3025101dec10192e86188fe2cd98c33cf artifacts/warm-start-parent/manifest.json
16
+ fa43228a19a32fc2caf5480799a2746b4da619b42406c0f3898d36059107050f artifacts/warm-start-parent/parent-snapshot.json
17
+ adf5a8ca2bab60cf7429a82b1b7a2d1ae7d3cc812de04dd195a478dfd7c6f628 artifacts/warm-start-parent/source-data-manifest.json
18
+ 1e0a2f838251c8eb411b59f4fa061db67ec308ce249d166901760b097c64bd6c artifacts/warm-start-parent/validation_step_001500_predictions.json
19
+ d23a648a7d73cb8d6f139033985b1b6bb67be3bb510ebbb89f4432138c2ebd52 backup-manifest.json
20
+ 0b546fd6b96e34316fddcb06314fa072d1ddb9c06d7e6979ccd790421b93f419 backup-tools/prepare_expansion_backup.py
21
+ 3845483f6469dc30122ae31a2bcbf6b1ad7ce69663308266f831142cd5d4fd1e backup-tools/test_expansion_backup.py
22
+ adf5a8ca2bab60cf7429a82b1b7a2d1ae7d3cc812de04dd195a478dfd7c6f628 data/public-decisions-v1-20260916-manifest.json
23
+ 87f4430d84f2a6157593f9af44a356ede87bad18cc51fa9ae3db2fee96830691 data/public-decisions-v2-20260917/ATTRIBUTION.md
24
+ 58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4 data/public-decisions-v2-20260917/calibration.jsonl
25
+ 93ec461769c925a7f10de76ff7b04118ab731ed5f49ebb5517b2d448fa1fdafc data/public-decisions-v2-20260917/diagnostics/new_sources.jsonl
26
+ 0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3 data/public-decisions-v2-20260917/holdout.jsonl
27
+ fde6ee7ce2eca20cb22cdbbe4db0ddbdb29a8ea9d597906d88e545939a5b602c data/public-decisions-v2-20260917/manifest.json
28
+ f07eef84b3081ad86bb5b48f810bbed569a76ee8e82cba9228beec908232d79e data/public-decisions-v2-20260917/proof/data_filter.json
29
+ a72ee77ba11017679548b06a2b956f7d2e56ca07e95eba165bb48aefbb75da18 data/public-decisions-v2-20260917/proof/postbuild-audit.json
30
+ 5c05478b29c84218784690f3c7c3ec994615fec1c57826591607189f191c56b4 data/public-decisions-v2-20260917/proof/prepare_expanded_data.py
31
+ c2c506ad72513d573eec723adb3960a00f9416fb974e9359f7b61b108f1ae2bb data/public-decisions-v2-20260917/proof/tokenization-proof.json
32
+ f30b0dab7c917f74676fd1cffc988ffb175ddb9a843f2029ce715d7e44bc7f96 data/public-decisions-v2-20260917/proof/tokenize_only.py
33
+ 172917e887dcc013fe0f0bd6aa8c810aea1e2be67d3bc7ff63e9b9d92075cc34 data/public-decisions-v2-20260917/raw/commonsenseqa/README.md
34
+ cfe6e26e7e936a447a12f7eee50f2bffb0355c3b97459496d8c7296f65c5b353 data/public-decisions-v2-20260917/raw/hellaswag/README.md
35
+ f7bcb808a64959151698f2bca621572c4809459733d04b95d53f3c62588f1593 data/public-decisions-v2-20260917/raw/piqa/piqa/README.md
36
+ ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819 data/public-decisions-v2-20260917/test.jsonl
37
+ d6d6a57c3aaaf6527647d0a21cc6027a58a6b40e2099eebb9a0689c0212ee372 data/public-decisions-v2-20260917/train.jsonl
38
+ 411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f data/public-decisions-v2-20260917/validation.jsonl
39
+ 657979189140a298f383cbcf722c425c64f7b07b7e0332c32ac1b076445028e2 fleet/plan.json
40
+ e671e1508185765552b0f933ba03f356be62143c531d8ef534457d34b1645c9b fleet/reference-validation-predictions.json
snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/best.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5b9eccc4c4e2bf1306e2def3e663b66e9f0cd1f3791dff04313039eb07426be5
3
+ size 66203019
snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/best_validation_selection.json ADDED
@@ -0,0 +1,601 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metric": "crossfit_temperature_nll_v1",
3
+ "score": 0.1701497127614862,
4
+ "raw_macro_nll": 0.190872636672039,
5
+ "accuracy": 0.947265625,
6
+ "folds": [
7
+ {
8
+ "fold": 0,
9
+ "temperature_index": 51,
10
+ "temperature": 1.4893610777109154,
11
+ "training_macro_nll": 0.1622086936723931,
12
+ "heldout_ids": [
13
+ "arc:validation:ARC-Challenge:122",
14
+ "arc:validation:ARC-Challenge:126",
15
+ "arc:validation:ARC-Challenge:175",
16
+ "arc:validation:ARC-Challenge:187",
17
+ "arc:validation:ARC-Challenge:189",
18
+ "arc:validation:ARC-Challenge:192",
19
+ "arc:validation:ARC-Challenge:208",
20
+ "arc:validation:ARC-Challenge:238",
21
+ "arc:validation:ARC-Challenge:290",
22
+ "arc:validation:ARC-Challenge:293",
23
+ "arc:validation:ARC-Challenge:35",
24
+ "arc:validation:ARC-Challenge:43",
25
+ "arc:validation:ARC-Challenge:58",
26
+ "arc:validation:ARC-Challenge:7",
27
+ "arc:validation:ARC-Easy:160",
28
+ "arc:validation:ARC-Easy:201",
29
+ "arc:validation:ARC-Easy:229",
30
+ "arc:validation:ARC-Easy:238",
31
+ "arc:validation:ARC-Easy:253",
32
+ "arc:validation:ARC-Easy:259",
33
+ "arc:validation:ARC-Easy:272",
34
+ "arc:validation:ARC-Easy:298",
35
+ "arc:validation:ARC-Easy:329",
36
+ "arc:validation:ARC-Easy:331",
37
+ "arc:validation:ARC-Easy:337",
38
+ "arc:validation:ARC-Easy:362",
39
+ "arc:validation:ARC-Easy:443",
40
+ "arc:validation:ARC-Easy:500",
41
+ "arc:validation:ARC-Easy:545",
42
+ "arc:validation:ARC-Easy:68",
43
+ "arc:validation:ARC-Easy:7",
44
+ "arc:validation:ARC-Easy:74",
45
+ "banking:train:1053",
46
+ "banking:train:1594",
47
+ "banking:train:1747",
48
+ "banking:train:202",
49
+ "banking:train:2377",
50
+ "banking:train:2528",
51
+ "banking:train:2546",
52
+ "banking:train:2715",
53
+ "banking:train:3220",
54
+ "banking:train:3499",
55
+ "banking:train:3806",
56
+ "banking:train:3838",
57
+ "banking:train:3993",
58
+ "banking:train:4220",
59
+ "banking:train:5",
60
+ "banking:train:5192",
61
+ "banking:train:5394",
62
+ "banking:train:5846",
63
+ "banking:train:5863",
64
+ "banking:train:6032",
65
+ "banking:train:6508",
66
+ "banking:train:7579",
67
+ "banking:train:7600",
68
+ "banking:train:7987",
69
+ "banking:train:8155",
70
+ "banking:train:850",
71
+ "banking:train:8626",
72
+ "banking:train:8929",
73
+ "banking:train:9355",
74
+ "banking:train:9812",
75
+ "banking:train:987",
76
+ "banking:train:9972",
77
+ "boolq:validation:103",
78
+ "boolq:validation:1042",
79
+ "boolq:validation:1059",
80
+ "boolq:validation:1541",
81
+ "boolq:validation:1592",
82
+ "boolq:validation:1927",
83
+ "boolq:validation:2014",
84
+ "boolq:validation:2023",
85
+ "boolq:validation:2098",
86
+ "boolq:validation:2139",
87
+ "boolq:validation:2439",
88
+ "boolq:validation:2485",
89
+ "boolq:validation:2607",
90
+ "boolq:validation:2611",
91
+ "boolq:validation:2720",
92
+ "boolq:validation:2751",
93
+ "boolq:validation:282",
94
+ "boolq:validation:2843",
95
+ "boolq:validation:2953",
96
+ "boolq:validation:301",
97
+ "boolq:validation:3089",
98
+ "boolq:validation:313",
99
+ "boolq:validation:3201",
100
+ "boolq:validation:322",
101
+ "boolq:validation:3231",
102
+ "boolq:validation:39",
103
+ "boolq:validation:461",
104
+ "boolq:validation:558",
105
+ "boolq:validation:589",
106
+ "boolq:validation:731",
107
+ "boolq:validation:883",
108
+ "boolq:validation:915",
109
+ "snli:validation:1156",
110
+ "snli:validation:1330",
111
+ "snli:validation:1559",
112
+ "snli:validation:1735",
113
+ "snli:validation:1876",
114
+ "snli:validation:2251",
115
+ "snli:validation:2257",
116
+ "snli:validation:2377",
117
+ "snli:validation:2881",
118
+ "snli:validation:291",
119
+ "snli:validation:300",
120
+ "snli:validation:3049",
121
+ "snli:validation:3496",
122
+ "snli:validation:3871",
123
+ "snli:validation:4414",
124
+ "snli:validation:5140",
125
+ "snli:validation:5440",
126
+ "snli:validation:5806",
127
+ "snli:validation:6442",
128
+ "snli:validation:6772",
129
+ "snli:validation:6949",
130
+ "snli:validation:7228",
131
+ "snli:validation:7465",
132
+ "snli:validation:7510",
133
+ "snli:validation:7726",
134
+ "snli:validation:8095",
135
+ "snli:validation:8143",
136
+ "snli:validation:8623",
137
+ "snli:validation:8962",
138
+ "snli:validation:9451",
139
+ "snli:validation:9574",
140
+ "snli:validation:9655"
141
+ ],
142
+ "heldout_group_count": 128,
143
+ "training_group_count": 384
144
+ },
145
+ {
146
+ "fold": 1,
147
+ "temperature_index": 51,
148
+ "temperature": 1.4893610777109154,
149
+ "training_macro_nll": 0.15901522343009294,
150
+ "heldout_ids": [
151
+ "arc:validation:ARC-Challenge:101",
152
+ "arc:validation:ARC-Challenge:129",
153
+ "arc:validation:ARC-Challenge:148",
154
+ "arc:validation:ARC-Challenge:156",
155
+ "arc:validation:ARC-Challenge:167",
156
+ "arc:validation:ARC-Challenge:249",
157
+ "arc:validation:ARC-Challenge:263",
158
+ "arc:validation:ARC-Challenge:266",
159
+ "arc:validation:ARC-Challenge:267",
160
+ "arc:validation:ARC-Challenge:28",
161
+ "arc:validation:ARC-Challenge:281",
162
+ "arc:validation:ARC-Challenge:57",
163
+ "arc:validation:ARC-Challenge:63",
164
+ "arc:validation:ARC-Challenge:64",
165
+ "arc:validation:ARC-Challenge:86",
166
+ "arc:validation:ARC-Challenge:99",
167
+ "arc:validation:ARC-Easy:106",
168
+ "arc:validation:ARC-Easy:151",
169
+ "arc:validation:ARC-Easy:157",
170
+ "arc:validation:ARC-Easy:166",
171
+ "arc:validation:ARC-Easy:202",
172
+ "arc:validation:ARC-Easy:219",
173
+ "arc:validation:ARC-Easy:224",
174
+ "arc:validation:ARC-Easy:322",
175
+ "arc:validation:ARC-Easy:344",
176
+ "arc:validation:ARC-Easy:382",
177
+ "arc:validation:ARC-Easy:409",
178
+ "arc:validation:ARC-Easy:455",
179
+ "arc:validation:ARC-Easy:50",
180
+ "arc:validation:ARC-Easy:538",
181
+ "arc:validation:ARC-Easy:544",
182
+ "arc:validation:ARC-Easy:552",
183
+ "banking:train:1050",
184
+ "banking:train:1131",
185
+ "banking:train:1190",
186
+ "banking:train:124",
187
+ "banking:train:137",
188
+ "banking:train:1493",
189
+ "banking:train:1498",
190
+ "banking:train:1624",
191
+ "banking:train:2614",
192
+ "banking:train:2739",
193
+ "banking:train:3055",
194
+ "banking:train:306",
195
+ "banking:train:3312",
196
+ "banking:train:3724",
197
+ "banking:train:4070",
198
+ "banking:train:476",
199
+ "banking:train:4882",
200
+ "banking:train:5075",
201
+ "banking:train:5332",
202
+ "banking:train:5445",
203
+ "banking:train:6096",
204
+ "banking:train:746",
205
+ "banking:train:7828",
206
+ "banking:train:8176",
207
+ "banking:train:8315",
208
+ "banking:train:8653",
209
+ "banking:train:9019",
210
+ "banking:train:9370",
211
+ "banking:train:9454",
212
+ "banking:train:9595",
213
+ "banking:train:9862",
214
+ "banking:train:9963",
215
+ "boolq:validation:1189",
216
+ "boolq:validation:1320",
217
+ "boolq:validation:1326",
218
+ "boolq:validation:1381",
219
+ "boolq:validation:1410",
220
+ "boolq:validation:149",
221
+ "boolq:validation:1513",
222
+ "boolq:validation:1732",
223
+ "boolq:validation:1739",
224
+ "boolq:validation:1741",
225
+ "boolq:validation:1814",
226
+ "boolq:validation:1848",
227
+ "boolq:validation:1940",
228
+ "boolq:validation:2077",
229
+ "boolq:validation:2091",
230
+ "boolq:validation:2148",
231
+ "boolq:validation:2215",
232
+ "boolq:validation:2251",
233
+ "boolq:validation:2283",
234
+ "boolq:validation:2321",
235
+ "boolq:validation:2398",
236
+ "boolq:validation:253",
237
+ "boolq:validation:2575",
238
+ "boolq:validation:2907",
239
+ "boolq:validation:2909",
240
+ "boolq:validation:2978",
241
+ "boolq:validation:359",
242
+ "boolq:validation:394",
243
+ "boolq:validation:436",
244
+ "boolq:validation:579",
245
+ "boolq:validation:807",
246
+ "boolq:validation:949",
247
+ "snli:validation:2389",
248
+ "snli:validation:2485",
249
+ "snli:validation:255",
250
+ "snli:validation:3169",
251
+ "snli:validation:333",
252
+ "snli:validation:3700",
253
+ "snli:validation:3709",
254
+ "snli:validation:4093",
255
+ "snli:validation:420",
256
+ "snli:validation:4492",
257
+ "snli:validation:4516",
258
+ "snli:validation:4648",
259
+ "snli:validation:4888",
260
+ "snli:validation:5185",
261
+ "snli:validation:5332",
262
+ "snli:validation:5947",
263
+ "snli:validation:6124",
264
+ "snli:validation:660",
265
+ "snli:validation:6619",
266
+ "snli:validation:6898",
267
+ "snli:validation:7387",
268
+ "snli:validation:7492",
269
+ "snli:validation:7544",
270
+ "snli:validation:7996",
271
+ "snli:validation:8074",
272
+ "snli:validation:8554",
273
+ "snli:validation:8971",
274
+ "snli:validation:9163",
275
+ "snli:validation:9277",
276
+ "snli:validation:9490",
277
+ "snli:validation:9595",
278
+ "snli:validation:9688"
279
+ ],
280
+ "heldout_group_count": 128,
281
+ "training_group_count": 384
282
+ },
283
+ {
284
+ "fold": 2,
285
+ "temperature_index": 51,
286
+ "temperature": 1.4893610777109154,
287
+ "training_macro_nll": 0.15824280619091902,
288
+ "heldout_ids": [
289
+ "arc:validation:ARC-Challenge:120",
290
+ "arc:validation:ARC-Challenge:130",
291
+ "arc:validation:ARC-Challenge:194",
292
+ "arc:validation:ARC-Challenge:218",
293
+ "arc:validation:ARC-Challenge:235",
294
+ "arc:validation:ARC-Challenge:272",
295
+ "arc:validation:ARC-Challenge:277",
296
+ "arc:validation:ARC-Challenge:48",
297
+ "arc:validation:ARC-Challenge:88",
298
+ "arc:validation:ARC-Easy:115",
299
+ "arc:validation:ARC-Easy:12",
300
+ "arc:validation:ARC-Easy:131",
301
+ "arc:validation:ARC-Easy:145",
302
+ "arc:validation:ARC-Easy:179",
303
+ "arc:validation:ARC-Easy:18",
304
+ "arc:validation:ARC-Easy:186",
305
+ "arc:validation:ARC-Easy:191",
306
+ "arc:validation:ARC-Easy:215",
307
+ "arc:validation:ARC-Easy:283",
308
+ "arc:validation:ARC-Easy:314",
309
+ "arc:validation:ARC-Easy:315",
310
+ "arc:validation:ARC-Easy:378",
311
+ "arc:validation:ARC-Easy:384",
312
+ "arc:validation:ARC-Easy:403",
313
+ "arc:validation:ARC-Easy:432",
314
+ "arc:validation:ARC-Easy:439",
315
+ "arc:validation:ARC-Easy:448",
316
+ "arc:validation:ARC-Easy:461",
317
+ "arc:validation:ARC-Easy:469",
318
+ "arc:validation:ARC-Easy:470",
319
+ "arc:validation:ARC-Easy:504",
320
+ "arc:validation:ARC-Easy:561",
321
+ "banking:train:1142",
322
+ "banking:train:148",
323
+ "banking:train:1633",
324
+ "banking:train:1927",
325
+ "banking:train:2186",
326
+ "banking:train:2251",
327
+ "banking:train:23",
328
+ "banking:train:3048",
329
+ "banking:train:3665",
330
+ "banking:train:3952",
331
+ "banking:train:3999",
332
+ "banking:train:4364",
333
+ "banking:train:4619",
334
+ "banking:train:5049",
335
+ "banking:train:5335",
336
+ "banking:train:549",
337
+ "banking:train:5957",
338
+ "banking:train:5985",
339
+ "banking:train:6132",
340
+ "banking:train:6240",
341
+ "banking:train:6488",
342
+ "banking:train:7065",
343
+ "banking:train:7749",
344
+ "banking:train:8470",
345
+ "banking:train:8761",
346
+ "banking:train:895",
347
+ "banking:train:9256",
348
+ "banking:train:9406",
349
+ "banking:train:9450",
350
+ "banking:train:9467",
351
+ "banking:train:9588",
352
+ "banking:train:9716",
353
+ "boolq:validation:1130",
354
+ "boolq:validation:1167",
355
+ "boolq:validation:124",
356
+ "boolq:validation:1255",
357
+ "boolq:validation:1256",
358
+ "boolq:validation:1373",
359
+ "boolq:validation:15",
360
+ "boolq:validation:155",
361
+ "boolq:validation:1568",
362
+ "boolq:validation:16",
363
+ "boolq:validation:1654",
364
+ "boolq:validation:1702",
365
+ "boolq:validation:1792",
366
+ "boolq:validation:1818",
367
+ "boolq:validation:210",
368
+ "boolq:validation:2247",
369
+ "boolq:validation:2259",
370
+ "boolq:validation:2362",
371
+ "boolq:validation:2443",
372
+ "boolq:validation:2524",
373
+ "boolq:validation:2553",
374
+ "boolq:validation:2623",
375
+ "boolq:validation:2660",
376
+ "boolq:validation:2669",
377
+ "boolq:validation:2679",
378
+ "boolq:validation:2832",
379
+ "boolq:validation:357",
380
+ "boolq:validation:453",
381
+ "boolq:validation:49",
382
+ "boolq:validation:808",
383
+ "boolq:validation:90",
384
+ "boolq:validation:910",
385
+ "snli:validation:1612",
386
+ "snli:validation:168",
387
+ "snli:validation:1693",
388
+ "snli:validation:2134",
389
+ "snli:validation:2248",
390
+ "snli:validation:2422",
391
+ "snli:validation:2587",
392
+ "snli:validation:3007",
393
+ "snli:validation:3097",
394
+ "snli:validation:3772",
395
+ "snli:validation:4012",
396
+ "snli:validation:4168",
397
+ "snli:validation:4537",
398
+ "snli:validation:4681",
399
+ "snli:validation:5050",
400
+ "snli:validation:5632",
401
+ "snli:validation:5761",
402
+ "snli:validation:6055",
403
+ "snli:validation:6283",
404
+ "snli:validation:6715",
405
+ "snli:validation:7117",
406
+ "snli:validation:7192",
407
+ "snli:validation:7507",
408
+ "snli:validation:7594",
409
+ "snli:validation:7924",
410
+ "snli:validation:7954",
411
+ "snli:validation:8035",
412
+ "snli:validation:8110",
413
+ "snli:validation:8995",
414
+ "snli:validation:949",
415
+ "snli:validation:9523",
416
+ "snli:validation:9589"
417
+ ],
418
+ "heldout_group_count": 128,
419
+ "training_group_count": 384
420
+ },
421
+ {
422
+ "fold": 3,
423
+ "temperature_index": 53,
424
+ "temperature": 1.6557699634695275,
425
+ "training_macro_nll": 0.19176642571330008,
426
+ "heldout_ids": [
427
+ "arc:validation:ARC-Challenge:117",
428
+ "arc:validation:ARC-Challenge:135",
429
+ "arc:validation:ARC-Challenge:139",
430
+ "arc:validation:ARC-Challenge:154",
431
+ "arc:validation:ARC-Challenge:174",
432
+ "arc:validation:ARC-Challenge:199",
433
+ "arc:validation:ARC-Challenge:287",
434
+ "arc:validation:ARC-Challenge:295",
435
+ "arc:validation:ARC-Challenge:4",
436
+ "arc:validation:ARC-Challenge:50",
437
+ "arc:validation:ARC-Challenge:51",
438
+ "arc:validation:ARC-Challenge:61",
439
+ "arc:validation:ARC-Easy:107",
440
+ "arc:validation:ARC-Easy:121",
441
+ "arc:validation:ARC-Easy:124",
442
+ "arc:validation:ARC-Easy:171",
443
+ "arc:validation:ARC-Easy:235",
444
+ "arc:validation:ARC-Easy:241",
445
+ "arc:validation:ARC-Easy:279",
446
+ "arc:validation:ARC-Easy:289",
447
+ "arc:validation:ARC-Easy:37",
448
+ "arc:validation:ARC-Easy:389",
449
+ "arc:validation:ARC-Easy:394",
450
+ "arc:validation:ARC-Easy:399",
451
+ "arc:validation:ARC-Easy:423",
452
+ "arc:validation:ARC-Easy:492",
453
+ "arc:validation:ARC-Easy:499",
454
+ "arc:validation:ARC-Easy:523",
455
+ "arc:validation:ARC-Easy:533",
456
+ "arc:validation:ARC-Easy:569",
457
+ "arc:validation:ARC-Easy:69",
458
+ "arc:validation:ARC-Easy:93",
459
+ "banking:train:2049",
460
+ "banking:train:2350",
461
+ "banking:train:2497",
462
+ "banking:train:2506",
463
+ "banking:train:2720",
464
+ "banking:train:2788",
465
+ "banking:train:2967",
466
+ "banking:train:3502",
467
+ "banking:train:3726",
468
+ "banking:train:4014",
469
+ "banking:train:4735",
470
+ "banking:train:4803",
471
+ "banking:train:4917",
472
+ "banking:train:5477",
473
+ "banking:train:5851",
474
+ "banking:train:6551",
475
+ "banking:train:6839",
476
+ "banking:train:6926",
477
+ "banking:train:7809",
478
+ "banking:train:8092",
479
+ "banking:train:8186",
480
+ "banking:train:8215",
481
+ "banking:train:8667",
482
+ "banking:train:8752",
483
+ "banking:train:8763",
484
+ "banking:train:8851",
485
+ "banking:train:8917",
486
+ "banking:train:9145",
487
+ "banking:train:9279",
488
+ "banking:train:9676",
489
+ "banking:train:9837",
490
+ "banking:train:9967",
491
+ "boolq:validation:106",
492
+ "boolq:validation:1065",
493
+ "boolq:validation:1097",
494
+ "boolq:validation:1336",
495
+ "boolq:validation:1350",
496
+ "boolq:validation:1383",
497
+ "boolq:validation:1458",
498
+ "boolq:validation:1641",
499
+ "boolq:validation:1873",
500
+ "boolq:validation:1930",
501
+ "boolq:validation:1989",
502
+ "boolq:validation:2103",
503
+ "boolq:validation:234",
504
+ "boolq:validation:2355",
505
+ "boolq:validation:2384",
506
+ "boolq:validation:2440",
507
+ "boolq:validation:2444",
508
+ "boolq:validation:2478",
509
+ "boolq:validation:2872",
510
+ "boolq:validation:2879",
511
+ "boolq:validation:2924",
512
+ "boolq:validation:2971",
513
+ "boolq:validation:3164",
514
+ "boolq:validation:4",
515
+ "boolq:validation:415",
516
+ "boolq:validation:419",
517
+ "boolq:validation:428",
518
+ "boolq:validation:512",
519
+ "boolq:validation:563",
520
+ "boolq:validation:643",
521
+ "boolq:validation:696",
522
+ "boolq:validation:874",
523
+ "snli:validation:1102",
524
+ "snli:validation:1240",
525
+ "snli:validation:1261",
526
+ "snli:validation:1738",
527
+ "snli:validation:1978",
528
+ "snli:validation:2563",
529
+ "snli:validation:2659",
530
+ "snli:validation:2767",
531
+ "snli:validation:2836",
532
+ "snli:validation:3232",
533
+ "snli:validation:3655",
534
+ "snli:validation:3913",
535
+ "snli:validation:405",
536
+ "snli:validation:4051",
537
+ "snli:validation:4300",
538
+ "snli:validation:4846",
539
+ "snli:validation:498",
540
+ "snli:validation:5224",
541
+ "snli:validation:5575",
542
+ "snli:validation:6598",
543
+ "snli:validation:666",
544
+ "snli:validation:6925",
545
+ "snli:validation:6964",
546
+ "snli:validation:7045",
547
+ "snli:validation:7351",
548
+ "snli:validation:7588",
549
+ "snli:validation:7723",
550
+ "snli:validation:7750",
551
+ "snli:validation:7909",
552
+ "snli:validation:8974",
553
+ "snli:validation:9385",
554
+ "snli:validation:9646"
555
+ ],
556
+ "heldout_group_count": 128,
557
+ "training_group_count": 384
558
+ }
559
+ ],
560
+ "per_family": {
561
+ "arc": {
562
+ "count": 128,
563
+ "score": 0.13049231104529602,
564
+ "raw_macro_nll": 0.1125077638524943,
565
+ "accuracy": 0.96875
566
+ },
567
+ "banking": {
568
+ "count": 128,
569
+ "score": 0.05035731679828836,
570
+ "raw_macro_nll": 0.05960886883339138,
571
+ "accuracy": 0.9765625
572
+ },
573
+ "boolq": {
574
+ "count": 128,
575
+ "score": 0.28226935283016463,
576
+ "raw_macro_nll": 0.3498694938007437,
577
+ "accuracy": 0.9296875
578
+ },
579
+ "snli": {
580
+ "count": 128,
581
+ "score": 0.21747987037219577,
582
+ "raw_macro_nll": 0.24150442020152654,
583
+ "accuracy": 0.9140625
584
+ }
585
+ },
586
+ "policy": {
587
+ "id": "crossfit_temperature_nll_v1",
588
+ "folds": 4,
589
+ "seed": 431,
590
+ "grouping": "source group, nested within task family; groups shuffled within sorted families and assigned round-robin",
591
+ "temperature_grid": {
592
+ "count": 101,
593
+ "log10_min": -1.0,
594
+ "log10_max": 1.3,
595
+ "arithmetic": "Python float64"
596
+ },
597
+ "temperature_fit": "minimum macro-family NLL on the other three folds; smallest temperature wins ties",
598
+ "score": "macro-family mean of all held-out decision NLLs",
599
+ "deployment_temperature": "fit afresh on reserved calibration only after model selection"
600
+ }
601
+ }
snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/checkpoint.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cf4cd775165cc2bb02113d17b94facd0db7b94fcff87b27389b3280fbb712469
3
+ size 198820477
snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/correctness_final.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "branch_chunks_1_probability_max_abs": 0.0,
3
+ "branch_chunks_2_probability_max_abs": 3.5762786865234375e-07,
4
+ "branch_chunks_4_probability_max_abs": 2.384185791015625e-07,
5
+ "question_isolation_probability_max_abs": 0.0,
6
+ "candidate_permutation_probability_max_abs": 0.0,
7
+ "repeat_probability_max_abs": 0.0,
8
+ "tolerance_probability_abs": 0.0001
9
+ }
snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/correctness_initial.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "branch_chunks_1_probability_max_abs": 0.0,
3
+ "branch_chunks_2_probability_max_abs": 2.6496127247810364e-07,
4
+ "branch_chunks_4_probability_max_abs": 1.7369166016578674e-07,
5
+ "question_isolation_probability_max_abs": 0.0,
6
+ "candidate_permutation_probability_max_abs": 0.0,
7
+ "repeat_probability_max_abs": 0.0,
8
+ "tolerance_probability_abs": 0.0001
9
+ }
snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/data_filter.json ADDED
@@ -0,0 +1,100 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "calibration": {
3
+ "retained": 510,
4
+ "dropped_ids": [
5
+ "boolq:validation:200",
6
+ "boolq:validation:1836"
7
+ ],
8
+ "family_counts": {
9
+ "boolq": 126,
10
+ "snli": 128,
11
+ "banking": 128,
12
+ "arc": 128
13
+ },
14
+ "retained_id_sha256": "8a0d4add2dd95717d34915195dc7100878b8f443bf714656d54b336b629f6476",
15
+ "max_branch_tokens": 481
16
+ },
17
+ "holdout": {
18
+ "retained": 768,
19
+ "dropped_ids": [],
20
+ "family_counts": {
21
+ "social": 768
22
+ },
23
+ "retained_id_sha256": "805387bd9156d12d4d40b33e5926f209451f323814ad876a3f04a8abece3ae0b",
24
+ "max_branch_tokens": 123
25
+ },
26
+ "test": {
27
+ "retained": 2042,
28
+ "dropped_ids": [
29
+ "boolq:validation:1681",
30
+ "boolq:validation:1661",
31
+ "boolq:validation:2150",
32
+ "boolq:validation:2153",
33
+ "boolq:validation:3154",
34
+ "boolq:validation:561"
35
+ ],
36
+ "family_counts": {
37
+ "banking": 512,
38
+ "boolq": 506,
39
+ "arc": 512,
40
+ "snli": 512
41
+ },
42
+ "retained_id_sha256": "343960f63f954a0c05884459a13a3ef8560fd98c2caead3aa3c3b70022917ec5",
43
+ "max_branch_tokens": 440
44
+ },
45
+ "train": {
46
+ "retained": 80765,
47
+ "dropped_ids": [
48
+ "boolq:train:5085",
49
+ "boolq:train:1430",
50
+ "boolq:train:353",
51
+ "boolq:train:3547",
52
+ "boolq:train:5618",
53
+ "boolq:train:3872",
54
+ "boolq:train:6128",
55
+ "boolq:train:899",
56
+ "boolq:train:711",
57
+ "boolq:train:6969",
58
+ "boolq:train:7410",
59
+ "boolq:train:9405",
60
+ "boolq:train:3362",
61
+ "boolq:train:8317",
62
+ "boolq:train:4726",
63
+ "boolq:train:3163",
64
+ "boolq:train:7445",
65
+ "boolq:train:2181",
66
+ "boolq:train:8140",
67
+ "boolq:train:2141",
68
+ "boolq:train:204",
69
+ "boolq:train:1505",
70
+ "boolq:train:2352",
71
+ "boolq:train:9421",
72
+ "boolq:train:5517",
73
+ "boolq:train:6869",
74
+ "piqa:train:13223"
75
+ ],
76
+ "family_counts": {
77
+ "banking": 9608,
78
+ "snli": 20000,
79
+ "boolq": 7962,
80
+ "arc": 3345,
81
+ "hellaswag": 16000,
82
+ "piqa": 14360,
83
+ "commonsenseqa": 9490
84
+ },
85
+ "retained_id_sha256": "79e791c0e8b0f4312fdadcd62042a689d32d2bcf04c80419c8eebd94db99249c",
86
+ "max_branch_tokens": 509
87
+ },
88
+ "validation": {
89
+ "retained": 512,
90
+ "dropped_ids": [],
91
+ "family_counts": {
92
+ "boolq": 128,
93
+ "snli": 128,
94
+ "banking": 128,
95
+ "arc": 128
96
+ },
97
+ "retained_id_sha256": "b0ea1f550bee363d58849a00295771707fca92d9d8e630840d0f870a91a9f8a7",
98
+ "max_branch_tokens": 477
99
+ }
100
+ }
snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/exit_code ADDED
@@ -0,0 +1 @@
 
 
1
+ 0
snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/manifest.json ADDED
@@ -0,0 +1,370 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "config": {
3
+ "command": "train",
4
+ "model": "/home/andy/ai/models/opensysone/Qwen3-4B-Instruct-2507-cdbee75f",
5
+ "dataset": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917",
6
+ "output": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts",
7
+ "resume": null,
8
+ "warm_start": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/best.pt",
9
+ "allow_train_data_change": true,
10
+ "steps": 8,
11
+ "epochs": 3,
12
+ "rank": 8,
13
+ "alpha": 16.0,
14
+ "head_only": false,
15
+ "two_pass": true,
16
+ "lr": 2e-05,
17
+ "head_lr": 2e-05,
18
+ "seed": 433,
19
+ "effective_batch": 4,
20
+ "branch_batch_size": 1,
21
+ "max_tokens": 512,
22
+ "save_steps": 250,
23
+ "save_seconds": 900,
24
+ "eval_steps": 500,
25
+ "validation_per_family": 128,
26
+ "patience": 8,
27
+ "deadline": "2026-09-17T16:00:00Z",
28
+ "schedule_steps": 3500,
29
+ "selection_metric": "crossfit_temperature_nll_v1",
30
+ "adapters": true
31
+ },
32
+ "pid": 1627838,
33
+ "hostname": "gx10-9dd0",
34
+ "started_utc": "2026-09-17T07:07:59.676564+00:00",
35
+ "git_commit": "24b8ccf60d388f9cbb184e03a6ae260a1f5a8b86",
36
+ "git_status": "",
37
+ "source_sha256": {
38
+ "playground.py": "b10c400421dd8558a7fef8ddde632676cdfe7f63edf94184f299f9ed569c010a",
39
+ "selection.py": "be0a7a8496b5b830aa572ceba93606320f442063fd38180503fd6980dc1c578f",
40
+ "training_model.py": "d5b0aefeeb5290816bc0b669aa0a8cbbe27f6a12b9cb23c141ac9b9ae9ee4e65",
41
+ "decision_model.py": "a3d8aeb02a1ac765c6cc30ff175acad0664560f01ab5403e22cade924d17371e",
42
+ "data_transition.py": "93aaa89b4de3aa78c34f03a5643e31f91f738c832395334966368df9b902621c",
43
+ "smoke_data.py": "06b3cbac1c8c4a86b8aecbee4459073cc3e46d4ddcd576392f3cb4805924f815",
44
+ "jev_harness.py": "4d4e979cb7ae352bcdacaaa6d64045e6b5e550b1a9721d4bad545045bee6c67f",
45
+ "experiment.py": "c779c3936aa1c2c51052f035df7bc0895a2de79c9ffc6c50fb0ee848832e17c7",
46
+ "smoke_train.py": "8cdeb2b397177fc9c26638aaa871501ddab8e3871aa1573ecd98f66590f5c228",
47
+ "scripts/run_experiment.sh": "661a6309fc54a2a8aff918f14a553c72dcb21730bd6a3cfc55d6ccd4700d11d3",
48
+ "scripts/download_candidate.py": "d06a2c01be0cf6577f927fb37e3bc1eab014949fd934e4f4d9825674adba608e",
49
+ "scripts/prepare_public_data.py": "32ea84aa719818e1141b958b6ef27a85f7ddb86bfcd7c1585fc487b25253d977",
50
+ "scripts/verify_playground_layout.cjs": "dea3fa07c568c141cd58fead8c547db7196f4a48fe4ea4cea0312d2ba7bd8bb4",
51
+ "scripts/run_smoke.sh": "39d59f2120f362729d1c2e384391b82be1e580dcc1aca0eaa7ab231115225574",
52
+ "scripts/start_spark_candidate.sh": "c2ca18b008a144de7cb264c9fcca634d68c4a8a317db3638e8a70b3dbcff064b",
53
+ "scripts/launch_24h.py": "39c26dc10535d3adac09209b0743edf2ed384563e732512ad62f8c6f37161a83",
54
+ "scripts/verify_artifact.py": "9833350e9d72c0065b15206bb71c5a8b5a6b3185219db90074e369ede563a985",
55
+ "scripts/publish_hf_snapshot.py": "b5fe16a00fcbc5ab97121428c6ce750275ee193438e3ffe24aac5e3bb325c018",
56
+ "scripts/prepare_expanded_data.py": "5c05478b29c84218784690f3c7c3ec994615fec1c57826591607189f191c56b4",
57
+ "scripts/fleet_campaign.py": "e69fdff96f92c6943b7895be11df444f016d0b744a1b9441995a5f8bb7af9d54",
58
+ "scripts/run_precision.sh": "766d82b30cf3e83951f662685b4472ee053c7fbc171c4ff140823bf4d1c2782f",
59
+ "scripts/fleet_status.py": "2519ced157ef4ac4fa449eebabae5740d7527778d578b4ac6720583010fa5217",
60
+ "scripts/verify_playground.cjs": "a55aa0e945baeb9e536c2b9cce7a9aa2364244b587e90477190add96f78601dc",
61
+ "scripts/diagnose_parity.py": "08b5d66d316ebda98a2226251a4f952701f86a1d5726ce7a4d7e8fb22755da5a",
62
+ "scripts/publish_hf_final.py": "278efc5878d7ebc5d1171f3a735d6c7d78275d93650e5a8a16c3f55b8353a577",
63
+ "scripts/download_model.py": "72ad9a5de44d09e2ee4ed8afb7c3c0ff6fb48987410a3f7f0368353572bf1f4d",
64
+ "scripts/investigate_precision.py": "609b744a926d8a45b87ba8d225e5312ee0c71a7096b21bd3589c5846e1dc847c",
65
+ "scripts/campaign_status.py": "1500b5e24f06231c7aafd7277aefd840582e35997e265f79db93614828d34411"
66
+ },
67
+ "packages": {
68
+ "torch": "2.11.0+cu130",
69
+ "transformers": "5.15.0",
70
+ "pyarrow": "25.0.1",
71
+ "numpy": "2.5.2"
72
+ },
73
+ "cuda": "13.0",
74
+ "gpu": "NVIDIA GB10",
75
+ "capability": [
76
+ 12,
77
+ 1
78
+ ],
79
+ "cuda_cap_bytes": 17179869184,
80
+ "initial_mem_available_bytes": 107427065856,
81
+ "oom_score_adj": "0",
82
+ "parent_checkpoint_sha256": "5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024",
83
+ "initialization": {
84
+ "kind": "warm_start",
85
+ "parent_checkpoint": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/best.pt",
86
+ "parent_checkpoint_sha256": "5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024",
87
+ "restores_optimizer": false,
88
+ "restores_rng": false,
89
+ "parent_step": 1500,
90
+ "parent_source_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1",
91
+ "data_transition": {
92
+ "kind": "training_split_only",
93
+ "parent_dataset": "/home/andy/ai/opensysone/data/public-decisions-v1-20260916",
94
+ "dataset": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917",
95
+ "parent_data_signature": "c00527e505fba87aabbfa65ba5fb7a67cf628ac7d46d7598f270936a894e3884",
96
+ "data_signature": "76183c642668602f42b7f3e71a3fe03bd5bd76f064fce4ba92351d8703396207",
97
+ "parent_manifest_sha256": "adf5a8ca2bab60cf7429a82b1b7a2d1ae7d3cc812de04dd195a478dfd7c6f628",
98
+ "manifest_sha256": "fde6ee7ce2eca20cb22cdbbe4db0ddbdb29a8ea9d597906d88e545939a5b602c",
99
+ "protected_split_sha256": {
100
+ "validation": "411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f",
101
+ "calibration": "58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4",
102
+ "test": "ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819",
103
+ "holdout": "0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3"
104
+ }
105
+ }
106
+ },
107
+ "model_provenance": {
108
+ "model_id": "Qwen/Qwen3-4B-Instruct-2507",
109
+ "revision": "cdbee75f17c01a7cc42f958dc650907174af0554",
110
+ "license": "apache-2.0"
111
+ },
112
+ "data_signature": "76183c642668602f42b7f3e71a3fe03bd5bd76f064fce4ba92351d8703396207",
113
+ "total_parameters": 4038985729,
114
+ "trainable_parameters": 16517633,
115
+ "adapter_modules": [
116
+ "layers.0.self_attn.q_proj",
117
+ "layers.0.self_attn.k_proj",
118
+ "layers.0.self_attn.v_proj",
119
+ "layers.0.self_attn.o_proj",
120
+ "layers.0.mlp.gate_proj",
121
+ "layers.0.mlp.up_proj",
122
+ "layers.0.mlp.down_proj",
123
+ "layers.1.self_attn.q_proj",
124
+ "layers.1.self_attn.k_proj",
125
+ "layers.1.self_attn.v_proj",
126
+ "layers.1.self_attn.o_proj",
127
+ "layers.1.mlp.gate_proj",
128
+ "layers.1.mlp.up_proj",
129
+ "layers.1.mlp.down_proj",
130
+ "layers.2.self_attn.q_proj",
131
+ "layers.2.self_attn.k_proj",
132
+ "layers.2.self_attn.v_proj",
133
+ "layers.2.self_attn.o_proj",
134
+ "layers.2.mlp.gate_proj",
135
+ "layers.2.mlp.up_proj",
136
+ "layers.2.mlp.down_proj",
137
+ "layers.3.self_attn.q_proj",
138
+ "layers.3.self_attn.k_proj",
139
+ "layers.3.self_attn.v_proj",
140
+ "layers.3.self_attn.o_proj",
141
+ "layers.3.mlp.gate_proj",
142
+ "layers.3.mlp.up_proj",
143
+ "layers.3.mlp.down_proj",
144
+ "layers.4.self_attn.q_proj",
145
+ "layers.4.self_attn.k_proj",
146
+ "layers.4.self_attn.v_proj",
147
+ "layers.4.self_attn.o_proj",
148
+ "layers.4.mlp.gate_proj",
149
+ "layers.4.mlp.up_proj",
150
+ "layers.4.mlp.down_proj",
151
+ "layers.5.self_attn.q_proj",
152
+ "layers.5.self_attn.k_proj",
153
+ "layers.5.self_attn.v_proj",
154
+ "layers.5.self_attn.o_proj",
155
+ "layers.5.mlp.gate_proj",
156
+ "layers.5.mlp.up_proj",
157
+ "layers.5.mlp.down_proj",
158
+ "layers.6.self_attn.q_proj",
159
+ "layers.6.self_attn.k_proj",
160
+ "layers.6.self_attn.v_proj",
161
+ "layers.6.self_attn.o_proj",
162
+ "layers.6.mlp.gate_proj",
163
+ "layers.6.mlp.up_proj",
164
+ "layers.6.mlp.down_proj",
165
+ "layers.7.self_attn.q_proj",
166
+ "layers.7.self_attn.k_proj",
167
+ "layers.7.self_attn.v_proj",
168
+ "layers.7.self_attn.o_proj",
169
+ "layers.7.mlp.gate_proj",
170
+ "layers.7.mlp.up_proj",
171
+ "layers.7.mlp.down_proj",
172
+ "layers.8.self_attn.q_proj",
173
+ "layers.8.self_attn.k_proj",
174
+ "layers.8.self_attn.v_proj",
175
+ "layers.8.self_attn.o_proj",
176
+ "layers.8.mlp.gate_proj",
177
+ "layers.8.mlp.up_proj",
178
+ "layers.8.mlp.down_proj",
179
+ "layers.9.self_attn.q_proj",
180
+ "layers.9.self_attn.k_proj",
181
+ "layers.9.self_attn.v_proj",
182
+ "layers.9.self_attn.o_proj",
183
+ "layers.9.mlp.gate_proj",
184
+ "layers.9.mlp.up_proj",
185
+ "layers.9.mlp.down_proj",
186
+ "layers.10.self_attn.q_proj",
187
+ "layers.10.self_attn.k_proj",
188
+ "layers.10.self_attn.v_proj",
189
+ "layers.10.self_attn.o_proj",
190
+ "layers.10.mlp.gate_proj",
191
+ "layers.10.mlp.up_proj",
192
+ "layers.10.mlp.down_proj",
193
+ "layers.11.self_attn.q_proj",
194
+ "layers.11.self_attn.k_proj",
195
+ "layers.11.self_attn.v_proj",
196
+ "layers.11.self_attn.o_proj",
197
+ "layers.11.mlp.gate_proj",
198
+ "layers.11.mlp.up_proj",
199
+ "layers.11.mlp.down_proj",
200
+ "layers.12.self_attn.q_proj",
201
+ "layers.12.self_attn.k_proj",
202
+ "layers.12.self_attn.v_proj",
203
+ "layers.12.self_attn.o_proj",
204
+ "layers.12.mlp.gate_proj",
205
+ "layers.12.mlp.up_proj",
206
+ "layers.12.mlp.down_proj",
207
+ "layers.13.self_attn.q_proj",
208
+ "layers.13.self_attn.k_proj",
209
+ "layers.13.self_attn.v_proj",
210
+ "layers.13.self_attn.o_proj",
211
+ "layers.13.mlp.gate_proj",
212
+ "layers.13.mlp.up_proj",
213
+ "layers.13.mlp.down_proj",
214
+ "layers.14.self_attn.q_proj",
215
+ "layers.14.self_attn.k_proj",
216
+ "layers.14.self_attn.v_proj",
217
+ "layers.14.self_attn.o_proj",
218
+ "layers.14.mlp.gate_proj",
219
+ "layers.14.mlp.up_proj",
220
+ "layers.14.mlp.down_proj",
221
+ "layers.15.self_attn.q_proj",
222
+ "layers.15.self_attn.k_proj",
223
+ "layers.15.self_attn.v_proj",
224
+ "layers.15.self_attn.o_proj",
225
+ "layers.15.mlp.gate_proj",
226
+ "layers.15.mlp.up_proj",
227
+ "layers.15.mlp.down_proj",
228
+ "layers.16.self_attn.q_proj",
229
+ "layers.16.self_attn.k_proj",
230
+ "layers.16.self_attn.v_proj",
231
+ "layers.16.self_attn.o_proj",
232
+ "layers.16.mlp.gate_proj",
233
+ "layers.16.mlp.up_proj",
234
+ "layers.16.mlp.down_proj",
235
+ "layers.17.self_attn.q_proj",
236
+ "layers.17.self_attn.k_proj",
237
+ "layers.17.self_attn.v_proj",
238
+ "layers.17.self_attn.o_proj",
239
+ "layers.17.mlp.gate_proj",
240
+ "layers.17.mlp.up_proj",
241
+ "layers.17.mlp.down_proj",
242
+ "layers.18.self_attn.q_proj",
243
+ "layers.18.self_attn.k_proj",
244
+ "layers.18.self_attn.v_proj",
245
+ "layers.18.self_attn.o_proj",
246
+ "layers.18.mlp.gate_proj",
247
+ "layers.18.mlp.up_proj",
248
+ "layers.18.mlp.down_proj",
249
+ "layers.19.self_attn.q_proj",
250
+ "layers.19.self_attn.k_proj",
251
+ "layers.19.self_attn.v_proj",
252
+ "layers.19.self_attn.o_proj",
253
+ "layers.19.mlp.gate_proj",
254
+ "layers.19.mlp.up_proj",
255
+ "layers.19.mlp.down_proj",
256
+ "layers.20.self_attn.q_proj",
257
+ "layers.20.self_attn.k_proj",
258
+ "layers.20.self_attn.v_proj",
259
+ "layers.20.self_attn.o_proj",
260
+ "layers.20.mlp.gate_proj",
261
+ "layers.20.mlp.up_proj",
262
+ "layers.20.mlp.down_proj",
263
+ "layers.21.self_attn.q_proj",
264
+ "layers.21.self_attn.k_proj",
265
+ "layers.21.self_attn.v_proj",
266
+ "layers.21.self_attn.o_proj",
267
+ "layers.21.mlp.gate_proj",
268
+ "layers.21.mlp.up_proj",
269
+ "layers.21.mlp.down_proj",
270
+ "layers.22.self_attn.q_proj",
271
+ "layers.22.self_attn.k_proj",
272
+ "layers.22.self_attn.v_proj",
273
+ "layers.22.self_attn.o_proj",
274
+ "layers.22.mlp.gate_proj",
275
+ "layers.22.mlp.up_proj",
276
+ "layers.22.mlp.down_proj",
277
+ "layers.23.self_attn.q_proj",
278
+ "layers.23.self_attn.k_proj",
279
+ "layers.23.self_attn.v_proj",
280
+ "layers.23.self_attn.o_proj",
281
+ "layers.23.mlp.gate_proj",
282
+ "layers.23.mlp.up_proj",
283
+ "layers.23.mlp.down_proj",
284
+ "layers.24.self_attn.q_proj",
285
+ "layers.24.self_attn.k_proj",
286
+ "layers.24.self_attn.v_proj",
287
+ "layers.24.self_attn.o_proj",
288
+ "layers.24.mlp.gate_proj",
289
+ "layers.24.mlp.up_proj",
290
+ "layers.24.mlp.down_proj",
291
+ "layers.25.self_attn.q_proj",
292
+ "layers.25.self_attn.k_proj",
293
+ "layers.25.self_attn.v_proj",
294
+ "layers.25.self_attn.o_proj",
295
+ "layers.25.mlp.gate_proj",
296
+ "layers.25.mlp.up_proj",
297
+ "layers.25.mlp.down_proj",
298
+ "layers.26.self_attn.q_proj",
299
+ "layers.26.self_attn.k_proj",
300
+ "layers.26.self_attn.v_proj",
301
+ "layers.26.self_attn.o_proj",
302
+ "layers.26.mlp.gate_proj",
303
+ "layers.26.mlp.up_proj",
304
+ "layers.26.mlp.down_proj",
305
+ "layers.27.self_attn.q_proj",
306
+ "layers.27.self_attn.k_proj",
307
+ "layers.27.self_attn.v_proj",
308
+ "layers.27.self_attn.o_proj",
309
+ "layers.27.mlp.gate_proj",
310
+ "layers.27.mlp.up_proj",
311
+ "layers.27.mlp.down_proj",
312
+ "layers.28.self_attn.q_proj",
313
+ "layers.28.self_attn.k_proj",
314
+ "layers.28.self_attn.v_proj",
315
+ "layers.28.self_attn.o_proj",
316
+ "layers.28.mlp.gate_proj",
317
+ "layers.28.mlp.up_proj",
318
+ "layers.28.mlp.down_proj",
319
+ "layers.29.self_attn.q_proj",
320
+ "layers.29.self_attn.k_proj",
321
+ "layers.29.self_attn.v_proj",
322
+ "layers.29.self_attn.o_proj",
323
+ "layers.29.mlp.gate_proj",
324
+ "layers.29.mlp.up_proj",
325
+ "layers.29.mlp.down_proj",
326
+ "layers.30.self_attn.q_proj",
327
+ "layers.30.self_attn.k_proj",
328
+ "layers.30.self_attn.v_proj",
329
+ "layers.30.self_attn.o_proj",
330
+ "layers.30.mlp.gate_proj",
331
+ "layers.30.mlp.up_proj",
332
+ "layers.30.mlp.down_proj",
333
+ "layers.31.self_attn.q_proj",
334
+ "layers.31.self_attn.k_proj",
335
+ "layers.31.self_attn.v_proj",
336
+ "layers.31.self_attn.o_proj",
337
+ "layers.31.mlp.gate_proj",
338
+ "layers.31.mlp.up_proj",
339
+ "layers.31.mlp.down_proj",
340
+ "layers.32.self_attn.q_proj",
341
+ "layers.32.self_attn.k_proj",
342
+ "layers.32.self_attn.v_proj",
343
+ "layers.32.self_attn.o_proj",
344
+ "layers.32.mlp.gate_proj",
345
+ "layers.32.mlp.up_proj",
346
+ "layers.32.mlp.down_proj",
347
+ "layers.33.self_attn.q_proj",
348
+ "layers.33.self_attn.k_proj",
349
+ "layers.33.self_attn.v_proj",
350
+ "layers.33.self_attn.o_proj",
351
+ "layers.33.mlp.gate_proj",
352
+ "layers.33.mlp.up_proj",
353
+ "layers.33.mlp.down_proj",
354
+ "layers.34.self_attn.q_proj",
355
+ "layers.34.self_attn.k_proj",
356
+ "layers.34.self_attn.v_proj",
357
+ "layers.34.self_attn.o_proj",
358
+ "layers.34.mlp.gate_proj",
359
+ "layers.34.mlp.up_proj",
360
+ "layers.34.mlp.down_proj",
361
+ "layers.35.self_attn.q_proj",
362
+ "layers.35.self_attn.k_proj",
363
+ "layers.35.self_attn.v_proj",
364
+ "layers.35.self_attn.o_proj",
365
+ "layers.35.mlp.gate_proj",
366
+ "layers.35.mlp.up_proj",
367
+ "layers.35.mlp.down_proj"
368
+ ],
369
+ "load_and_tokenize_seconds": 4.631322219996946
370
+ }
snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/summary.json ADDED
@@ -0,0 +1,437 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "completed_step_target",
3
+ "completed_steps": 8,
4
+ "target_steps": 8,
5
+ "initial_validation": {
6
+ "overall": {
7
+ "n": 512,
8
+ "accuracy": 0.947265625,
9
+ "nll": 0.19087263293172443,
10
+ "brier_multiclass_sum": 0.09623258515322991,
11
+ "ece_top_label_10_equal_width_bins": 0.04369618045166135,
12
+ "reliability_bins": [
13
+ {
14
+ "count": 0,
15
+ "confidence_sum": 0.0,
16
+ "correct_sum": 0.0
17
+ },
18
+ {
19
+ "count": 0,
20
+ "confidence_sum": 0.0,
21
+ "correct_sum": 0.0
22
+ },
23
+ {
24
+ "count": 0,
25
+ "confidence_sum": 0.0,
26
+ "correct_sum": 0.0
27
+ },
28
+ {
29
+ "count": 1,
30
+ "confidence_sum": 0.35483407974243164,
31
+ "correct_sum": 0.0
32
+ },
33
+ {
34
+ "count": 2,
35
+ "confidence_sum": 0.9032753705978394,
36
+ "correct_sum": 0.0
37
+ },
38
+ {
39
+ "count": 11,
40
+ "confidence_sum": 6.043279945850372,
41
+ "correct_sum": 11.0
42
+ },
43
+ {
44
+ "count": 4,
45
+ "confidence_sum": 2.5671995282173157,
46
+ "correct_sum": 3.0
47
+ },
48
+ {
49
+ "count": 5,
50
+ "confidence_sum": 3.8174904584884644,
51
+ "correct_sum": 2.0
52
+ },
53
+ {
54
+ "count": 16,
55
+ "confidence_sum": 13.456127166748047,
56
+ "correct_sum": 12.0
57
+ },
58
+ {
59
+ "count": 473,
60
+ "confidence_sum": 469.4511967897415,
61
+ "correct_sum": 457.0
62
+ }
63
+ ],
64
+ "accuracy_vs_coverage": {
65
+ "0.25": {
66
+ "n": 128,
67
+ "accuracy": 1.0,
68
+ "min_confidence": 0.9999990463256836
69
+ },
70
+ "0.5": {
71
+ "n": 256,
72
+ "accuracy": 1.0,
73
+ "min_confidence": 0.9985522627830505
74
+ },
75
+ "0.75": {
76
+ "n": 384,
77
+ "accuracy": 0.9921875,
78
+ "min_confidence": 0.9913347363471985
79
+ },
80
+ "1.0": {
81
+ "n": 512,
82
+ "accuracy": 0.947265625,
83
+ "min_confidence": 0.35483407974243164
84
+ }
85
+ }
86
+ },
87
+ "per_family": {
88
+ "arc": {
89
+ "n": 128,
90
+ "accuracy": 0.96875,
91
+ "nll": 0.11250776012102981,
92
+ "brier_multiclass_sum": 0.054027314836344154,
93
+ "ece_top_label_10_equal_width_bins": 0.04739027423784137,
94
+ "reliability_bins": [
95
+ {
96
+ "count": 0,
97
+ "confidence_sum": 0.0,
98
+ "correct_sum": 0.0
99
+ },
100
+ {
101
+ "count": 0,
102
+ "confidence_sum": 0.0,
103
+ "correct_sum": 0.0
104
+ },
105
+ {
106
+ "count": 0,
107
+ "confidence_sum": 0.0,
108
+ "correct_sum": 0.0
109
+ },
110
+ {
111
+ "count": 0,
112
+ "confidence_sum": 0.0,
113
+ "correct_sum": 0.0
114
+ },
115
+ {
116
+ "count": 2,
117
+ "confidence_sum": 0.9032753705978394,
118
+ "correct_sum": 0.0
119
+ },
120
+ {
121
+ "count": 4,
122
+ "confidence_sum": 2.220052123069763,
123
+ "correct_sum": 4.0
124
+ },
125
+ {
126
+ "count": 2,
127
+ "confidence_sum": 1.3046249151229858,
128
+ "correct_sum": 2.0
129
+ },
130
+ {
131
+ "count": 3,
132
+ "confidence_sum": 2.2750282883644104,
133
+ "correct_sum": 1.0
134
+ },
135
+ {
136
+ "count": 4,
137
+ "confidence_sum": 3.4031789302825928,
138
+ "correct_sum": 4.0
139
+ },
140
+ {
141
+ "count": 113,
142
+ "confidence_sum": 112.18449258804321,
143
+ "correct_sum": 113.0
144
+ }
145
+ ],
146
+ "accuracy_vs_coverage": {
147
+ "0.25": {
148
+ "n": 32,
149
+ "accuracy": 1.0,
150
+ "min_confidence": 0.9999973773956299
151
+ },
152
+ "0.5": {
153
+ "n": 64,
154
+ "accuracy": 1.0,
155
+ "min_confidence": 0.9999173879623413
156
+ },
157
+ "0.75": {
158
+ "n": 96,
159
+ "accuracy": 1.0,
160
+ "min_confidence": 0.9919710159301758
161
+ },
162
+ "1.0": {
163
+ "n": 128,
164
+ "accuracy": 0.96875,
165
+ "min_confidence": 0.40406495332717896
166
+ }
167
+ }
168
+ },
169
+ "banking": {
170
+ "n": 128,
171
+ "accuracy": 0.9765625,
172
+ "nll": 0.059608867915258545,
173
+ "brier_multiclass_sum": 0.030860770989061745,
174
+ "ece_top_label_10_equal_width_bins": 0.01428686361759901,
175
+ "reliability_bins": [
176
+ {
177
+ "count": 0,
178
+ "confidence_sum": 0.0,
179
+ "correct_sum": 0.0
180
+ },
181
+ {
182
+ "count": 0,
183
+ "confidence_sum": 0.0,
184
+ "correct_sum": 0.0
185
+ },
186
+ {
187
+ "count": 0,
188
+ "confidence_sum": 0.0,
189
+ "correct_sum": 0.0
190
+ },
191
+ {
192
+ "count": 1,
193
+ "confidence_sum": 0.35483407974243164,
194
+ "correct_sum": 0.0
195
+ },
196
+ {
197
+ "count": 0,
198
+ "confidence_sum": 0.0,
199
+ "correct_sum": 0.0
200
+ },
201
+ {
202
+ "count": 0,
203
+ "confidence_sum": 0.0,
204
+ "correct_sum": 0.0
205
+ },
206
+ {
207
+ "count": 0,
208
+ "confidence_sum": 0.0,
209
+ "correct_sum": 0.0
210
+ },
211
+ {
212
+ "count": 0,
213
+ "confidence_sum": 0.0,
214
+ "correct_sum": 0.0
215
+ },
216
+ {
217
+ "count": 2,
218
+ "confidence_sum": 1.6012814044952393,
219
+ "correct_sum": 1.0
220
+ },
221
+ {
222
+ "count": 125,
223
+ "confidence_sum": 124.872603058815,
224
+ "correct_sum": 124.0
225
+ }
226
+ ],
227
+ "accuracy_vs_coverage": {
228
+ "0.25": {
229
+ "n": 32,
230
+ "accuracy": 1.0,
231
+ "min_confidence": 1.0
232
+ },
233
+ "0.5": {
234
+ "n": 64,
235
+ "accuracy": 1.0,
236
+ "min_confidence": 1.0
237
+ },
238
+ "0.75": {
239
+ "n": 96,
240
+ "accuracy": 1.0,
241
+ "min_confidence": 0.9999996423721313
242
+ },
243
+ "1.0": {
244
+ "n": 128,
245
+ "accuracy": 0.9765625,
246
+ "min_confidence": 0.35483407974243164
247
+ }
248
+ }
249
+ },
250
+ "boolq": {
251
+ "n": 128,
252
+ "accuracy": 0.9296875,
253
+ "nll": 0.3498694849111246,
254
+ "brier_multiclass_sum": 0.15866377323777012,
255
+ "ece_top_label_10_equal_width_bins": 0.08982685301452875,
256
+ "reliability_bins": [
257
+ {
258
+ "count": 0,
259
+ "confidence_sum": 0.0,
260
+ "correct_sum": 0.0
261
+ },
262
+ {
263
+ "count": 0,
264
+ "confidence_sum": 0.0,
265
+ "correct_sum": 0.0
266
+ },
267
+ {
268
+ "count": 0,
269
+ "confidence_sum": 0.0,
270
+ "correct_sum": 0.0
271
+ },
272
+ {
273
+ "count": 0,
274
+ "confidence_sum": 0.0,
275
+ "correct_sum": 0.0
276
+ },
277
+ {
278
+ "count": 0,
279
+ "confidence_sum": 0.0,
280
+ "correct_sum": 0.0
281
+ },
282
+ {
283
+ "count": 6,
284
+ "confidence_sum": 3.2728703022003174,
285
+ "correct_sum": 6.0
286
+ },
287
+ {
288
+ "count": 0,
289
+ "confidence_sum": 0.0,
290
+ "correct_sum": 0.0
291
+ },
292
+ {
293
+ "count": 0,
294
+ "confidence_sum": 0.0,
295
+ "correct_sum": 0.0
296
+ },
297
+ {
298
+ "count": 6,
299
+ "confidence_sum": 5.034650266170502,
300
+ "correct_sum": 6.0
301
+ },
302
+ {
303
+ "count": 116,
304
+ "confidence_sum": 114.8053577542305,
305
+ "correct_sum": 107.0
306
+ }
307
+ ],
308
+ "accuracy_vs_coverage": {
309
+ "0.25": {
310
+ "n": 32,
311
+ "accuracy": 1.0,
312
+ "min_confidence": 0.9983857870101929
313
+ },
314
+ "0.5": {
315
+ "n": 64,
316
+ "accuracy": 0.984375,
317
+ "min_confidence": 0.995145857334137
318
+ },
319
+ "0.75": {
320
+ "n": 96,
321
+ "accuracy": 0.9479166666666666,
322
+ "min_confidence": 0.9837860465049744
323
+ },
324
+ "1.0": {
325
+ "n": 128,
326
+ "accuracy": 0.9296875,
327
+ "min_confidence": 0.5146349668502808
328
+ }
329
+ }
330
+ },
331
+ "snli": {
332
+ "n": 128,
333
+ "accuracy": 0.9140625,
334
+ "nll": 0.2415044187794848,
335
+ "brier_multiclass_sum": 0.14137848154974367,
336
+ "ece_top_label_10_equal_width_bins": 0.06453468138352036,
337
+ "reliability_bins": [
338
+ {
339
+ "count": 0,
340
+ "confidence_sum": 0.0,
341
+ "correct_sum": 0.0
342
+ },
343
+ {
344
+ "count": 0,
345
+ "confidence_sum": 0.0,
346
+ "correct_sum": 0.0
347
+ },
348
+ {
349
+ "count": 0,
350
+ "confidence_sum": 0.0,
351
+ "correct_sum": 0.0
352
+ },
353
+ {
354
+ "count": 0,
355
+ "confidence_sum": 0.0,
356
+ "correct_sum": 0.0
357
+ },
358
+ {
359
+ "count": 0,
360
+ "confidence_sum": 0.0,
361
+ "correct_sum": 0.0
362
+ },
363
+ {
364
+ "count": 1,
365
+ "confidence_sum": 0.5503575205802917,
366
+ "correct_sum": 1.0
367
+ },
368
+ {
369
+ "count": 2,
370
+ "confidence_sum": 1.2625746130943298,
371
+ "correct_sum": 1.0
372
+ },
373
+ {
374
+ "count": 2,
375
+ "confidence_sum": 1.542462170124054,
376
+ "correct_sum": 1.0
377
+ },
378
+ {
379
+ "count": 4,
380
+ "confidence_sum": 3.417016565799713,
381
+ "correct_sum": 1.0
382
+ },
383
+ {
384
+ "count": 119,
385
+ "confidence_sum": 117.5887433886528,
386
+ "correct_sum": 113.0
387
+ }
388
+ ],
389
+ "accuracy_vs_coverage": {
390
+ "0.25": {
391
+ "n": 32,
392
+ "accuracy": 1.0,
393
+ "min_confidence": 0.9972984194755554
394
+ },
395
+ "0.5": {
396
+ "n": 64,
397
+ "accuracy": 1.0,
398
+ "min_confidence": 0.9949101805686951
399
+ },
400
+ "0.75": {
401
+ "n": 96,
402
+ "accuracy": 1.0,
403
+ "min_confidence": 0.9868677854537964
404
+ },
405
+ "1.0": {
406
+ "n": 128,
407
+ "accuracy": 0.9140625,
408
+ "min_confidence": 0.5503575205802917
409
+ }
410
+ }
411
+ }
412
+ }
413
+ },
414
+ "best_validation_macro_nll": 0.19087263293172443,
415
+ "selection_metric": "crossfit_temperature_nll_v1",
416
+ "best_validation_selection_score": 0.1701497127614862,
417
+ "training_seconds": 430.03222380098305,
418
+ "median_step_seconds": 8.901401424504002,
419
+ "processed_decisions": 32,
420
+ "actual_branch_tokens": 10584,
421
+ "padded_branch_tokens": 10584,
422
+ "peak_cuda_allocated_bytes": 16563131904,
423
+ "peak_cuda_reserved_bytes": 16649289728,
424
+ "final_mem_available_bytes": 86696333312,
425
+ "correctness": {
426
+ "branch_chunks_1_probability_max_abs": 0.0,
427
+ "branch_chunks_2_probability_max_abs": 3.5762786865234375e-07,
428
+ "branch_chunks_4_probability_max_abs": 2.384185791015625e-07,
429
+ "question_isolation_probability_max_abs": 0.0,
430
+ "candidate_permutation_probability_max_abs": 0.0,
431
+ "repeat_probability_max_abs": 0.0,
432
+ "tolerance_probability_abs": 0.0001
433
+ },
434
+ "final_correctness_status": "passed",
435
+ "checkpoint_sha256": "cf4cd775165cc2bb02113d17b94facd0db7b94fcff87b27389b3280fbb712469",
436
+ "best_sha256": "5b9eccc4c4e2bf1306e2def3e663b66e9f0cd1f3791dff04313039eb07426be5"
437
+ }
snapshots/20260917-expanded-pilot-hf-backup/artifacts/expanded-gx10-4b-pilot/validation_step_000000_predictions.json ADDED
The diff for this file is too large to render. See raw diff
 
snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/best.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024
3
+ size 66202635
snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/best_validation_selection.json ADDED
@@ -0,0 +1,601 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metric": "crossfit_temperature_nll_v1",
3
+ "score": 0.1701497127614862,
4
+ "raw_macro_nll": 0.190872636672039,
5
+ "accuracy": 0.947265625,
6
+ "folds": [
7
+ {
8
+ "fold": 0,
9
+ "temperature_index": 51,
10
+ "temperature": 1.4893610777109154,
11
+ "training_macro_nll": 0.1622086936723931,
12
+ "heldout_ids": [
13
+ "arc:validation:ARC-Challenge:122",
14
+ "arc:validation:ARC-Challenge:126",
15
+ "arc:validation:ARC-Challenge:175",
16
+ "arc:validation:ARC-Challenge:187",
17
+ "arc:validation:ARC-Challenge:189",
18
+ "arc:validation:ARC-Challenge:192",
19
+ "arc:validation:ARC-Challenge:208",
20
+ "arc:validation:ARC-Challenge:238",
21
+ "arc:validation:ARC-Challenge:290",
22
+ "arc:validation:ARC-Challenge:293",
23
+ "arc:validation:ARC-Challenge:35",
24
+ "arc:validation:ARC-Challenge:43",
25
+ "arc:validation:ARC-Challenge:58",
26
+ "arc:validation:ARC-Challenge:7",
27
+ "arc:validation:ARC-Easy:160",
28
+ "arc:validation:ARC-Easy:201",
29
+ "arc:validation:ARC-Easy:229",
30
+ "arc:validation:ARC-Easy:238",
31
+ "arc:validation:ARC-Easy:253",
32
+ "arc:validation:ARC-Easy:259",
33
+ "arc:validation:ARC-Easy:272",
34
+ "arc:validation:ARC-Easy:298",
35
+ "arc:validation:ARC-Easy:329",
36
+ "arc:validation:ARC-Easy:331",
37
+ "arc:validation:ARC-Easy:337",
38
+ "arc:validation:ARC-Easy:362",
39
+ "arc:validation:ARC-Easy:443",
40
+ "arc:validation:ARC-Easy:500",
41
+ "arc:validation:ARC-Easy:545",
42
+ "arc:validation:ARC-Easy:68",
43
+ "arc:validation:ARC-Easy:7",
44
+ "arc:validation:ARC-Easy:74",
45
+ "banking:train:1053",
46
+ "banking:train:1594",
47
+ "banking:train:1747",
48
+ "banking:train:202",
49
+ "banking:train:2377",
50
+ "banking:train:2528",
51
+ "banking:train:2546",
52
+ "banking:train:2715",
53
+ "banking:train:3220",
54
+ "banking:train:3499",
55
+ "banking:train:3806",
56
+ "banking:train:3838",
57
+ "banking:train:3993",
58
+ "banking:train:4220",
59
+ "banking:train:5",
60
+ "banking:train:5192",
61
+ "banking:train:5394",
62
+ "banking:train:5846",
63
+ "banking:train:5863",
64
+ "banking:train:6032",
65
+ "banking:train:6508",
66
+ "banking:train:7579",
67
+ "banking:train:7600",
68
+ "banking:train:7987",
69
+ "banking:train:8155",
70
+ "banking:train:850",
71
+ "banking:train:8626",
72
+ "banking:train:8929",
73
+ "banking:train:9355",
74
+ "banking:train:9812",
75
+ "banking:train:987",
76
+ "banking:train:9972",
77
+ "boolq:validation:103",
78
+ "boolq:validation:1042",
79
+ "boolq:validation:1059",
80
+ "boolq:validation:1541",
81
+ "boolq:validation:1592",
82
+ "boolq:validation:1927",
83
+ "boolq:validation:2014",
84
+ "boolq:validation:2023",
85
+ "boolq:validation:2098",
86
+ "boolq:validation:2139",
87
+ "boolq:validation:2439",
88
+ "boolq:validation:2485",
89
+ "boolq:validation:2607",
90
+ "boolq:validation:2611",
91
+ "boolq:validation:2720",
92
+ "boolq:validation:2751",
93
+ "boolq:validation:282",
94
+ "boolq:validation:2843",
95
+ "boolq:validation:2953",
96
+ "boolq:validation:301",
97
+ "boolq:validation:3089",
98
+ "boolq:validation:313",
99
+ "boolq:validation:3201",
100
+ "boolq:validation:322",
101
+ "boolq:validation:3231",
102
+ "boolq:validation:39",
103
+ "boolq:validation:461",
104
+ "boolq:validation:558",
105
+ "boolq:validation:589",
106
+ "boolq:validation:731",
107
+ "boolq:validation:883",
108
+ "boolq:validation:915",
109
+ "snli:validation:1156",
110
+ "snli:validation:1330",
111
+ "snli:validation:1559",
112
+ "snli:validation:1735",
113
+ "snli:validation:1876",
114
+ "snli:validation:2251",
115
+ "snli:validation:2257",
116
+ "snli:validation:2377",
117
+ "snli:validation:2881",
118
+ "snli:validation:291",
119
+ "snli:validation:300",
120
+ "snli:validation:3049",
121
+ "snli:validation:3496",
122
+ "snli:validation:3871",
123
+ "snli:validation:4414",
124
+ "snli:validation:5140",
125
+ "snli:validation:5440",
126
+ "snli:validation:5806",
127
+ "snli:validation:6442",
128
+ "snli:validation:6772",
129
+ "snli:validation:6949",
130
+ "snli:validation:7228",
131
+ "snli:validation:7465",
132
+ "snli:validation:7510",
133
+ "snli:validation:7726",
134
+ "snli:validation:8095",
135
+ "snli:validation:8143",
136
+ "snli:validation:8623",
137
+ "snli:validation:8962",
138
+ "snli:validation:9451",
139
+ "snli:validation:9574",
140
+ "snli:validation:9655"
141
+ ],
142
+ "heldout_group_count": 128,
143
+ "training_group_count": 384
144
+ },
145
+ {
146
+ "fold": 1,
147
+ "temperature_index": 51,
148
+ "temperature": 1.4893610777109154,
149
+ "training_macro_nll": 0.15901522343009294,
150
+ "heldout_ids": [
151
+ "arc:validation:ARC-Challenge:101",
152
+ "arc:validation:ARC-Challenge:129",
153
+ "arc:validation:ARC-Challenge:148",
154
+ "arc:validation:ARC-Challenge:156",
155
+ "arc:validation:ARC-Challenge:167",
156
+ "arc:validation:ARC-Challenge:249",
157
+ "arc:validation:ARC-Challenge:263",
158
+ "arc:validation:ARC-Challenge:266",
159
+ "arc:validation:ARC-Challenge:267",
160
+ "arc:validation:ARC-Challenge:28",
161
+ "arc:validation:ARC-Challenge:281",
162
+ "arc:validation:ARC-Challenge:57",
163
+ "arc:validation:ARC-Challenge:63",
164
+ "arc:validation:ARC-Challenge:64",
165
+ "arc:validation:ARC-Challenge:86",
166
+ "arc:validation:ARC-Challenge:99",
167
+ "arc:validation:ARC-Easy:106",
168
+ "arc:validation:ARC-Easy:151",
169
+ "arc:validation:ARC-Easy:157",
170
+ "arc:validation:ARC-Easy:166",
171
+ "arc:validation:ARC-Easy:202",
172
+ "arc:validation:ARC-Easy:219",
173
+ "arc:validation:ARC-Easy:224",
174
+ "arc:validation:ARC-Easy:322",
175
+ "arc:validation:ARC-Easy:344",
176
+ "arc:validation:ARC-Easy:382",
177
+ "arc:validation:ARC-Easy:409",
178
+ "arc:validation:ARC-Easy:455",
179
+ "arc:validation:ARC-Easy:50",
180
+ "arc:validation:ARC-Easy:538",
181
+ "arc:validation:ARC-Easy:544",
182
+ "arc:validation:ARC-Easy:552",
183
+ "banking:train:1050",
184
+ "banking:train:1131",
185
+ "banking:train:1190",
186
+ "banking:train:124",
187
+ "banking:train:137",
188
+ "banking:train:1493",
189
+ "banking:train:1498",
190
+ "banking:train:1624",
191
+ "banking:train:2614",
192
+ "banking:train:2739",
193
+ "banking:train:3055",
194
+ "banking:train:306",
195
+ "banking:train:3312",
196
+ "banking:train:3724",
197
+ "banking:train:4070",
198
+ "banking:train:476",
199
+ "banking:train:4882",
200
+ "banking:train:5075",
201
+ "banking:train:5332",
202
+ "banking:train:5445",
203
+ "banking:train:6096",
204
+ "banking:train:746",
205
+ "banking:train:7828",
206
+ "banking:train:8176",
207
+ "banking:train:8315",
208
+ "banking:train:8653",
209
+ "banking:train:9019",
210
+ "banking:train:9370",
211
+ "banking:train:9454",
212
+ "banking:train:9595",
213
+ "banking:train:9862",
214
+ "banking:train:9963",
215
+ "boolq:validation:1189",
216
+ "boolq:validation:1320",
217
+ "boolq:validation:1326",
218
+ "boolq:validation:1381",
219
+ "boolq:validation:1410",
220
+ "boolq:validation:149",
221
+ "boolq:validation:1513",
222
+ "boolq:validation:1732",
223
+ "boolq:validation:1739",
224
+ "boolq:validation:1741",
225
+ "boolq:validation:1814",
226
+ "boolq:validation:1848",
227
+ "boolq:validation:1940",
228
+ "boolq:validation:2077",
229
+ "boolq:validation:2091",
230
+ "boolq:validation:2148",
231
+ "boolq:validation:2215",
232
+ "boolq:validation:2251",
233
+ "boolq:validation:2283",
234
+ "boolq:validation:2321",
235
+ "boolq:validation:2398",
236
+ "boolq:validation:253",
237
+ "boolq:validation:2575",
238
+ "boolq:validation:2907",
239
+ "boolq:validation:2909",
240
+ "boolq:validation:2978",
241
+ "boolq:validation:359",
242
+ "boolq:validation:394",
243
+ "boolq:validation:436",
244
+ "boolq:validation:579",
245
+ "boolq:validation:807",
246
+ "boolq:validation:949",
247
+ "snli:validation:2389",
248
+ "snli:validation:2485",
249
+ "snli:validation:255",
250
+ "snli:validation:3169",
251
+ "snli:validation:333",
252
+ "snli:validation:3700",
253
+ "snli:validation:3709",
254
+ "snli:validation:4093",
255
+ "snli:validation:420",
256
+ "snli:validation:4492",
257
+ "snli:validation:4516",
258
+ "snli:validation:4648",
259
+ "snli:validation:4888",
260
+ "snli:validation:5185",
261
+ "snli:validation:5332",
262
+ "snli:validation:5947",
263
+ "snli:validation:6124",
264
+ "snli:validation:660",
265
+ "snli:validation:6619",
266
+ "snli:validation:6898",
267
+ "snli:validation:7387",
268
+ "snli:validation:7492",
269
+ "snli:validation:7544",
270
+ "snli:validation:7996",
271
+ "snli:validation:8074",
272
+ "snli:validation:8554",
273
+ "snli:validation:8971",
274
+ "snli:validation:9163",
275
+ "snli:validation:9277",
276
+ "snli:validation:9490",
277
+ "snli:validation:9595",
278
+ "snli:validation:9688"
279
+ ],
280
+ "heldout_group_count": 128,
281
+ "training_group_count": 384
282
+ },
283
+ {
284
+ "fold": 2,
285
+ "temperature_index": 51,
286
+ "temperature": 1.4893610777109154,
287
+ "training_macro_nll": 0.15824280619091902,
288
+ "heldout_ids": [
289
+ "arc:validation:ARC-Challenge:120",
290
+ "arc:validation:ARC-Challenge:130",
291
+ "arc:validation:ARC-Challenge:194",
292
+ "arc:validation:ARC-Challenge:218",
293
+ "arc:validation:ARC-Challenge:235",
294
+ "arc:validation:ARC-Challenge:272",
295
+ "arc:validation:ARC-Challenge:277",
296
+ "arc:validation:ARC-Challenge:48",
297
+ "arc:validation:ARC-Challenge:88",
298
+ "arc:validation:ARC-Easy:115",
299
+ "arc:validation:ARC-Easy:12",
300
+ "arc:validation:ARC-Easy:131",
301
+ "arc:validation:ARC-Easy:145",
302
+ "arc:validation:ARC-Easy:179",
303
+ "arc:validation:ARC-Easy:18",
304
+ "arc:validation:ARC-Easy:186",
305
+ "arc:validation:ARC-Easy:191",
306
+ "arc:validation:ARC-Easy:215",
307
+ "arc:validation:ARC-Easy:283",
308
+ "arc:validation:ARC-Easy:314",
309
+ "arc:validation:ARC-Easy:315",
310
+ "arc:validation:ARC-Easy:378",
311
+ "arc:validation:ARC-Easy:384",
312
+ "arc:validation:ARC-Easy:403",
313
+ "arc:validation:ARC-Easy:432",
314
+ "arc:validation:ARC-Easy:439",
315
+ "arc:validation:ARC-Easy:448",
316
+ "arc:validation:ARC-Easy:461",
317
+ "arc:validation:ARC-Easy:469",
318
+ "arc:validation:ARC-Easy:470",
319
+ "arc:validation:ARC-Easy:504",
320
+ "arc:validation:ARC-Easy:561",
321
+ "banking:train:1142",
322
+ "banking:train:148",
323
+ "banking:train:1633",
324
+ "banking:train:1927",
325
+ "banking:train:2186",
326
+ "banking:train:2251",
327
+ "banking:train:23",
328
+ "banking:train:3048",
329
+ "banking:train:3665",
330
+ "banking:train:3952",
331
+ "banking:train:3999",
332
+ "banking:train:4364",
333
+ "banking:train:4619",
334
+ "banking:train:5049",
335
+ "banking:train:5335",
336
+ "banking:train:549",
337
+ "banking:train:5957",
338
+ "banking:train:5985",
339
+ "banking:train:6132",
340
+ "banking:train:6240",
341
+ "banking:train:6488",
342
+ "banking:train:7065",
343
+ "banking:train:7749",
344
+ "banking:train:8470",
345
+ "banking:train:8761",
346
+ "banking:train:895",
347
+ "banking:train:9256",
348
+ "banking:train:9406",
349
+ "banking:train:9450",
350
+ "banking:train:9467",
351
+ "banking:train:9588",
352
+ "banking:train:9716",
353
+ "boolq:validation:1130",
354
+ "boolq:validation:1167",
355
+ "boolq:validation:124",
356
+ "boolq:validation:1255",
357
+ "boolq:validation:1256",
358
+ "boolq:validation:1373",
359
+ "boolq:validation:15",
360
+ "boolq:validation:155",
361
+ "boolq:validation:1568",
362
+ "boolq:validation:16",
363
+ "boolq:validation:1654",
364
+ "boolq:validation:1702",
365
+ "boolq:validation:1792",
366
+ "boolq:validation:1818",
367
+ "boolq:validation:210",
368
+ "boolq:validation:2247",
369
+ "boolq:validation:2259",
370
+ "boolq:validation:2362",
371
+ "boolq:validation:2443",
372
+ "boolq:validation:2524",
373
+ "boolq:validation:2553",
374
+ "boolq:validation:2623",
375
+ "boolq:validation:2660",
376
+ "boolq:validation:2669",
377
+ "boolq:validation:2679",
378
+ "boolq:validation:2832",
379
+ "boolq:validation:357",
380
+ "boolq:validation:453",
381
+ "boolq:validation:49",
382
+ "boolq:validation:808",
383
+ "boolq:validation:90",
384
+ "boolq:validation:910",
385
+ "snli:validation:1612",
386
+ "snli:validation:168",
387
+ "snli:validation:1693",
388
+ "snli:validation:2134",
389
+ "snli:validation:2248",
390
+ "snli:validation:2422",
391
+ "snli:validation:2587",
392
+ "snli:validation:3007",
393
+ "snli:validation:3097",
394
+ "snli:validation:3772",
395
+ "snli:validation:4012",
396
+ "snli:validation:4168",
397
+ "snli:validation:4537",
398
+ "snli:validation:4681",
399
+ "snli:validation:5050",
400
+ "snli:validation:5632",
401
+ "snli:validation:5761",
402
+ "snli:validation:6055",
403
+ "snli:validation:6283",
404
+ "snli:validation:6715",
405
+ "snli:validation:7117",
406
+ "snli:validation:7192",
407
+ "snli:validation:7507",
408
+ "snli:validation:7594",
409
+ "snli:validation:7924",
410
+ "snli:validation:7954",
411
+ "snli:validation:8035",
412
+ "snli:validation:8110",
413
+ "snli:validation:8995",
414
+ "snli:validation:949",
415
+ "snli:validation:9523",
416
+ "snli:validation:9589"
417
+ ],
418
+ "heldout_group_count": 128,
419
+ "training_group_count": 384
420
+ },
421
+ {
422
+ "fold": 3,
423
+ "temperature_index": 53,
424
+ "temperature": 1.6557699634695275,
425
+ "training_macro_nll": 0.19176642571330008,
426
+ "heldout_ids": [
427
+ "arc:validation:ARC-Challenge:117",
428
+ "arc:validation:ARC-Challenge:135",
429
+ "arc:validation:ARC-Challenge:139",
430
+ "arc:validation:ARC-Challenge:154",
431
+ "arc:validation:ARC-Challenge:174",
432
+ "arc:validation:ARC-Challenge:199",
433
+ "arc:validation:ARC-Challenge:287",
434
+ "arc:validation:ARC-Challenge:295",
435
+ "arc:validation:ARC-Challenge:4",
436
+ "arc:validation:ARC-Challenge:50",
437
+ "arc:validation:ARC-Challenge:51",
438
+ "arc:validation:ARC-Challenge:61",
439
+ "arc:validation:ARC-Easy:107",
440
+ "arc:validation:ARC-Easy:121",
441
+ "arc:validation:ARC-Easy:124",
442
+ "arc:validation:ARC-Easy:171",
443
+ "arc:validation:ARC-Easy:235",
444
+ "arc:validation:ARC-Easy:241",
445
+ "arc:validation:ARC-Easy:279",
446
+ "arc:validation:ARC-Easy:289",
447
+ "arc:validation:ARC-Easy:37",
448
+ "arc:validation:ARC-Easy:389",
449
+ "arc:validation:ARC-Easy:394",
450
+ "arc:validation:ARC-Easy:399",
451
+ "arc:validation:ARC-Easy:423",
452
+ "arc:validation:ARC-Easy:492",
453
+ "arc:validation:ARC-Easy:499",
454
+ "arc:validation:ARC-Easy:523",
455
+ "arc:validation:ARC-Easy:533",
456
+ "arc:validation:ARC-Easy:569",
457
+ "arc:validation:ARC-Easy:69",
458
+ "arc:validation:ARC-Easy:93",
459
+ "banking:train:2049",
460
+ "banking:train:2350",
461
+ "banking:train:2497",
462
+ "banking:train:2506",
463
+ "banking:train:2720",
464
+ "banking:train:2788",
465
+ "banking:train:2967",
466
+ "banking:train:3502",
467
+ "banking:train:3726",
468
+ "banking:train:4014",
469
+ "banking:train:4735",
470
+ "banking:train:4803",
471
+ "banking:train:4917",
472
+ "banking:train:5477",
473
+ "banking:train:5851",
474
+ "banking:train:6551",
475
+ "banking:train:6839",
476
+ "banking:train:6926",
477
+ "banking:train:7809",
478
+ "banking:train:8092",
479
+ "banking:train:8186",
480
+ "banking:train:8215",
481
+ "banking:train:8667",
482
+ "banking:train:8752",
483
+ "banking:train:8763",
484
+ "banking:train:8851",
485
+ "banking:train:8917",
486
+ "banking:train:9145",
487
+ "banking:train:9279",
488
+ "banking:train:9676",
489
+ "banking:train:9837",
490
+ "banking:train:9967",
491
+ "boolq:validation:106",
492
+ "boolq:validation:1065",
493
+ "boolq:validation:1097",
494
+ "boolq:validation:1336",
495
+ "boolq:validation:1350",
496
+ "boolq:validation:1383",
497
+ "boolq:validation:1458",
498
+ "boolq:validation:1641",
499
+ "boolq:validation:1873",
500
+ "boolq:validation:1930",
501
+ "boolq:validation:1989",
502
+ "boolq:validation:2103",
503
+ "boolq:validation:234",
504
+ "boolq:validation:2355",
505
+ "boolq:validation:2384",
506
+ "boolq:validation:2440",
507
+ "boolq:validation:2444",
508
+ "boolq:validation:2478",
509
+ "boolq:validation:2872",
510
+ "boolq:validation:2879",
511
+ "boolq:validation:2924",
512
+ "boolq:validation:2971",
513
+ "boolq:validation:3164",
514
+ "boolq:validation:4",
515
+ "boolq:validation:415",
516
+ "boolq:validation:419",
517
+ "boolq:validation:428",
518
+ "boolq:validation:512",
519
+ "boolq:validation:563",
520
+ "boolq:validation:643",
521
+ "boolq:validation:696",
522
+ "boolq:validation:874",
523
+ "snli:validation:1102",
524
+ "snli:validation:1240",
525
+ "snli:validation:1261",
526
+ "snli:validation:1738",
527
+ "snli:validation:1978",
528
+ "snli:validation:2563",
529
+ "snli:validation:2659",
530
+ "snli:validation:2767",
531
+ "snli:validation:2836",
532
+ "snli:validation:3232",
533
+ "snli:validation:3655",
534
+ "snli:validation:3913",
535
+ "snli:validation:405",
536
+ "snli:validation:4051",
537
+ "snli:validation:4300",
538
+ "snli:validation:4846",
539
+ "snli:validation:498",
540
+ "snli:validation:5224",
541
+ "snli:validation:5575",
542
+ "snli:validation:6598",
543
+ "snli:validation:666",
544
+ "snli:validation:6925",
545
+ "snli:validation:6964",
546
+ "snli:validation:7045",
547
+ "snli:validation:7351",
548
+ "snli:validation:7588",
549
+ "snli:validation:7723",
550
+ "snli:validation:7750",
551
+ "snli:validation:7909",
552
+ "snli:validation:8974",
553
+ "snli:validation:9385",
554
+ "snli:validation:9646"
555
+ ],
556
+ "heldout_group_count": 128,
557
+ "training_group_count": 384
558
+ }
559
+ ],
560
+ "per_family": {
561
+ "arc": {
562
+ "count": 128,
563
+ "score": 0.13049231104529602,
564
+ "raw_macro_nll": 0.1125077638524943,
565
+ "accuracy": 0.96875
566
+ },
567
+ "banking": {
568
+ "count": 128,
569
+ "score": 0.05035731679828836,
570
+ "raw_macro_nll": 0.05960886883339138,
571
+ "accuracy": 0.9765625
572
+ },
573
+ "boolq": {
574
+ "count": 128,
575
+ "score": 0.28226935283016463,
576
+ "raw_macro_nll": 0.3498694938007437,
577
+ "accuracy": 0.9296875
578
+ },
579
+ "snli": {
580
+ "count": 128,
581
+ "score": 0.21747987037219577,
582
+ "raw_macro_nll": 0.24150442020152654,
583
+ "accuracy": 0.9140625
584
+ }
585
+ },
586
+ "policy": {
587
+ "id": "crossfit_temperature_nll_v1",
588
+ "folds": 4,
589
+ "seed": 431,
590
+ "grouping": "source group, nested within task family; groups shuffled within sorted families and assigned round-robin",
591
+ "temperature_grid": {
592
+ "count": 101,
593
+ "log10_min": -1.0,
594
+ "log10_max": 1.3,
595
+ "arithmetic": "Python float64"
596
+ },
597
+ "temperature_fit": "minimum macro-family NLL on the other three folds; smallest temperature wins ties",
598
+ "score": "macro-family mean of all held-out decision NLLs",
599
+ "deployment_temperature": "fit afresh on reserved calibration only after model selection"
600
+ }
601
+ }
snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/correctness_initial.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "branch_chunks_1_probability_max_abs": 0.0,
3
+ "branch_chunks_2_probability_max_abs": 3.637978807091713e-09,
4
+ "branch_chunks_4_probability_max_abs": 2.153683453798294e-09,
5
+ "question_isolation_probability_max_abs": 0.0,
6
+ "candidate_permutation_probability_max_abs": 0.0,
7
+ "repeat_probability_max_abs": 0.0,
8
+ "tolerance_probability_abs": 0.0001
9
+ }
snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/data_filter.json ADDED
@@ -0,0 +1,96 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "test": {
3
+ "retained": 2042,
4
+ "dropped_ids": [
5
+ "boolq:validation:1681",
6
+ "boolq:validation:1661",
7
+ "boolq:validation:2150",
8
+ "boolq:validation:2153",
9
+ "boolq:validation:3154",
10
+ "boolq:validation:561"
11
+ ],
12
+ "family_counts": {
13
+ "banking": 512,
14
+ "boolq": 506,
15
+ "arc": 512,
16
+ "snli": 512
17
+ },
18
+ "retained_id_sha256": "343960f63f954a0c05884459a13a3ef8560fd98c2caead3aa3c3b70022917ec5",
19
+ "max_branch_tokens": 440
20
+ },
21
+ "validation": {
22
+ "retained": 512,
23
+ "dropped_ids": [],
24
+ "family_counts": {
25
+ "boolq": 128,
26
+ "snli": 128,
27
+ "banking": 128,
28
+ "arc": 128
29
+ },
30
+ "retained_id_sha256": "b0ea1f550bee363d58849a00295771707fca92d9d8e630840d0f870a91a9f8a7",
31
+ "max_branch_tokens": 477
32
+ },
33
+ "calibration": {
34
+ "retained": 510,
35
+ "dropped_ids": [
36
+ "boolq:validation:200",
37
+ "boolq:validation:1836"
38
+ ],
39
+ "family_counts": {
40
+ "boolq": 126,
41
+ "snli": 128,
42
+ "banking": 128,
43
+ "arc": 128
44
+ },
45
+ "retained_id_sha256": "8a0d4add2dd95717d34915195dc7100878b8f443bf714656d54b336b629f6476",
46
+ "max_branch_tokens": 481
47
+ },
48
+ "train": {
49
+ "retained": 40915,
50
+ "dropped_ids": [
51
+ "boolq:train:5085",
52
+ "boolq:train:1430",
53
+ "boolq:train:353",
54
+ "boolq:train:3547",
55
+ "boolq:train:5618",
56
+ "boolq:train:3872",
57
+ "boolq:train:6128",
58
+ "boolq:train:899",
59
+ "boolq:train:711",
60
+ "boolq:train:6969",
61
+ "boolq:train:7410",
62
+ "boolq:train:9405",
63
+ "boolq:train:3362",
64
+ "boolq:train:8317",
65
+ "boolq:train:4726",
66
+ "boolq:train:3163",
67
+ "boolq:train:7445",
68
+ "boolq:train:2181",
69
+ "boolq:train:8140",
70
+ "boolq:train:2141",
71
+ "boolq:train:204",
72
+ "boolq:train:1505",
73
+ "boolq:train:2352",
74
+ "boolq:train:9421",
75
+ "boolq:train:5517",
76
+ "boolq:train:6869"
77
+ ],
78
+ "family_counts": {
79
+ "banking": 9608,
80
+ "snli": 20000,
81
+ "boolq": 7962,
82
+ "arc": 3345
83
+ },
84
+ "retained_id_sha256": "7abdc36f620055b4debaf43c890a8bd249f1a216ce81e893ce3430a43e703e3c",
85
+ "max_branch_tokens": 509
86
+ },
87
+ "holdout": {
88
+ "retained": 768,
89
+ "dropped_ids": [],
90
+ "family_counts": {
91
+ "social": 768
92
+ },
93
+ "retained_id_sha256": "805387bd9156d12d4d40b33e5926f209451f323814ad876a3f04a8abece3ae0b",
94
+ "max_branch_tokens": 123
95
+ }
96
+ }
snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/manifest.json ADDED
@@ -0,0 +1,360 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "config": {
3
+ "command": "train",
4
+ "model": "/home/andy/ai/models/opensysone/Qwen3-4B-Instruct-2507-cdbee75f",
5
+ "dataset": "/home/andy/ai/opensysone/data/public-decisions-v1-20260916",
6
+ "output": "/home/andy/ai/opensysone/runs/20260917T023137Z-24h/training",
7
+ "resume": "/home/andy/ai/opensysone/runs/20260917T021605Z-train/artifacts/checkpoint.pt",
8
+ "warm_start": null,
9
+ "steps": null,
10
+ "epochs": 3,
11
+ "rank": 8,
12
+ "alpha": 16.0,
13
+ "head_only": false,
14
+ "two_pass": true,
15
+ "lr": 1e-05,
16
+ "head_lr": 1e-05,
17
+ "seed": 432,
18
+ "effective_batch": 4,
19
+ "branch_batch_size": 1,
20
+ "max_tokens": 512,
21
+ "save_steps": 250,
22
+ "save_seconds": 900,
23
+ "eval_steps": 500,
24
+ "validation_per_family": 128,
25
+ "patience": 8,
26
+ "deadline": "2026-09-17T16:00:00+00:00",
27
+ "schedule_steps": 5000,
28
+ "selection_metric": "crossfit_temperature_nll_v1",
29
+ "adapters": true
30
+ },
31
+ "pid": 484001,
32
+ "hostname": "spark-3e2a",
33
+ "started_utc": "2026-09-17T02:31:40.118906+00:00",
34
+ "git_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1",
35
+ "git_status": "",
36
+ "source_sha256": {
37
+ "jev_harness.py": "4d4e979cb7ae352bcdacaaa6d64045e6b5e550b1a9721d4bad545045bee6c67f",
38
+ "smoke_data.py": "06b3cbac1c8c4a86b8aecbee4459073cc3e46d4ddcd576392f3cb4805924f815",
39
+ "experiment.py": "c036cbe9fbf6f71c1bb141dcfbecfb06a2f4bd697b070d2274b1931480e92031",
40
+ "training_model.py": "d5b0aefeeb5290816bc0b669aa0a8cbbe27f6a12b9cb23c141ac9b9ae9ee4e65",
41
+ "smoke_train.py": "8cdeb2b397177fc9c26638aaa871501ddab8e3871aa1573ecd98f66590f5c228",
42
+ "decision_model.py": "a3d8aeb02a1ac765c6cc30ff175acad0664560f01ab5403e22cade924d17371e",
43
+ "selection.py": "be0a7a8496b5b830aa572ceba93606320f442063fd38180503fd6980dc1c578f",
44
+ "scripts/download_model.py": "72ad9a5de44d09e2ee4ed8afb7c3c0ff6fb48987410a3f7f0368353572bf1f4d",
45
+ "scripts/download_candidate.py": "d06a2c01be0cf6577f927fb37e3bc1eab014949fd934e4f4d9825674adba608e",
46
+ "scripts/run_precision.sh": "766d82b30cf3e83951f662685b4472ee053c7fbc171c4ff140823bf4d1c2782f",
47
+ "scripts/campaign_status.py": "1500b5e24f06231c7aafd7277aefd840582e35997e265f79db93614828d34411",
48
+ "scripts/verify_artifact.py": "9833350e9d72c0065b15206bb71c5a8b5a6b3185219db90074e369ede563a985",
49
+ "scripts/start_spark_candidate.sh": "c2ca18b008a144de7cb264c9fcca634d68c4a8a317db3638e8a70b3dbcff064b",
50
+ "scripts/run_experiment.sh": "661a6309fc54a2a8aff918f14a553c72dcb21730bd6a3cfc55d6ccd4700d11d3",
51
+ "scripts/fleet_campaign.py": "9e8d0511a579ace145b7ad079c11ce9b7465950569e86b22099caec2831d5f21",
52
+ "scripts/diagnose_parity.py": "08b5d66d316ebda98a2226251a4f952701f86a1d5726ce7a4d7e8fb22755da5a",
53
+ "scripts/prepare_public_data.py": "32ea84aa719818e1141b958b6ef27a85f7ddb86bfcd7c1585fc487b25253d977",
54
+ "scripts/run_smoke.sh": "39d59f2120f362729d1c2e384391b82be1e580dcc1aca0eaa7ab231115225574",
55
+ "scripts/launch_24h.py": "39c26dc10535d3adac09209b0743edf2ed384563e732512ad62f8c6f37161a83",
56
+ "scripts/investigate_precision.py": "609b744a926d8a45b87ba8d225e5312ee0c71a7096b21bd3589c5846e1dc847c"
57
+ },
58
+ "packages": {
59
+ "torch": "2.11.0+cu130",
60
+ "transformers": "5.15.0",
61
+ "pyarrow": "25.0.1",
62
+ "numpy": "2.5.2"
63
+ },
64
+ "cuda": "13.0",
65
+ "gpu": "NVIDIA GB10",
66
+ "capability": [
67
+ 12,
68
+ 1
69
+ ],
70
+ "cuda_cap_bytes": 17179869184,
71
+ "initial_mem_available_bytes": 125233938432,
72
+ "oom_score_adj": "0",
73
+ "parent_checkpoint_sha256": "ece94b096b7fa37eec9e180a5f652da86a9f4f42d49f7f124850f046915237a8",
74
+ "initialization": {
75
+ "kind": "resume",
76
+ "parent_checkpoint": "/home/andy/ai/opensysone/runs/20260917T021605Z-train/artifacts/checkpoint.pt",
77
+ "parent_checkpoint_sha256": "ece94b096b7fa37eec9e180a5f652da86a9f4f42d49f7f124850f046915237a8",
78
+ "restores_optimizer": true,
79
+ "restores_rng": true,
80
+ "parent_step": 8,
81
+ "parent_source_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1",
82
+ "parent_initialization": {
83
+ "kind": "warm_start",
84
+ "parent_checkpoint": "/home/andy/ai/opensysone/runs/20260917T021254Z-refinement-parent/best.pt",
85
+ "parent_checkpoint_sha256": "8956eb6c0cfbb02124aeefd99c3b418c55f55fdb9a64260350622d98dbba1aec",
86
+ "restores_optimizer": false,
87
+ "restores_rng": false,
88
+ "parent_step": 2500,
89
+ "parent_source_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1"
90
+ },
91
+ "selection_policy_change": {
92
+ "from": "crossfit_temperature_nll_v1",
93
+ "to": "crossfit_temperature_nll_v1",
94
+ "optimizer_and_rng_unchanged": true
95
+ }
96
+ },
97
+ "model_provenance": {
98
+ "model_id": "Qwen/Qwen3-4B-Instruct-2507",
99
+ "revision": "cdbee75f17c01a7cc42f958dc650907174af0554",
100
+ "license": "apache-2.0"
101
+ },
102
+ "data_signature": "c00527e505fba87aabbfa65ba5fb7a67cf628ac7d46d7598f270936a894e3884",
103
+ "total_parameters": 4038985729,
104
+ "trainable_parameters": 16517633,
105
+ "adapter_modules": [
106
+ "layers.0.self_attn.q_proj",
107
+ "layers.0.self_attn.k_proj",
108
+ "layers.0.self_attn.v_proj",
109
+ "layers.0.self_attn.o_proj",
110
+ "layers.0.mlp.gate_proj",
111
+ "layers.0.mlp.up_proj",
112
+ "layers.0.mlp.down_proj",
113
+ "layers.1.self_attn.q_proj",
114
+ "layers.1.self_attn.k_proj",
115
+ "layers.1.self_attn.v_proj",
116
+ "layers.1.self_attn.o_proj",
117
+ "layers.1.mlp.gate_proj",
118
+ "layers.1.mlp.up_proj",
119
+ "layers.1.mlp.down_proj",
120
+ "layers.2.self_attn.q_proj",
121
+ "layers.2.self_attn.k_proj",
122
+ "layers.2.self_attn.v_proj",
123
+ "layers.2.self_attn.o_proj",
124
+ "layers.2.mlp.gate_proj",
125
+ "layers.2.mlp.up_proj",
126
+ "layers.2.mlp.down_proj",
127
+ "layers.3.self_attn.q_proj",
128
+ "layers.3.self_attn.k_proj",
129
+ "layers.3.self_attn.v_proj",
130
+ "layers.3.self_attn.o_proj",
131
+ "layers.3.mlp.gate_proj",
132
+ "layers.3.mlp.up_proj",
133
+ "layers.3.mlp.down_proj",
134
+ "layers.4.self_attn.q_proj",
135
+ "layers.4.self_attn.k_proj",
136
+ "layers.4.self_attn.v_proj",
137
+ "layers.4.self_attn.o_proj",
138
+ "layers.4.mlp.gate_proj",
139
+ "layers.4.mlp.up_proj",
140
+ "layers.4.mlp.down_proj",
141
+ "layers.5.self_attn.q_proj",
142
+ "layers.5.self_attn.k_proj",
143
+ "layers.5.self_attn.v_proj",
144
+ "layers.5.self_attn.o_proj",
145
+ "layers.5.mlp.gate_proj",
146
+ "layers.5.mlp.up_proj",
147
+ "layers.5.mlp.down_proj",
148
+ "layers.6.self_attn.q_proj",
149
+ "layers.6.self_attn.k_proj",
150
+ "layers.6.self_attn.v_proj",
151
+ "layers.6.self_attn.o_proj",
152
+ "layers.6.mlp.gate_proj",
153
+ "layers.6.mlp.up_proj",
154
+ "layers.6.mlp.down_proj",
155
+ "layers.7.self_attn.q_proj",
156
+ "layers.7.self_attn.k_proj",
157
+ "layers.7.self_attn.v_proj",
158
+ "layers.7.self_attn.o_proj",
159
+ "layers.7.mlp.gate_proj",
160
+ "layers.7.mlp.up_proj",
161
+ "layers.7.mlp.down_proj",
162
+ "layers.8.self_attn.q_proj",
163
+ "layers.8.self_attn.k_proj",
164
+ "layers.8.self_attn.v_proj",
165
+ "layers.8.self_attn.o_proj",
166
+ "layers.8.mlp.gate_proj",
167
+ "layers.8.mlp.up_proj",
168
+ "layers.8.mlp.down_proj",
169
+ "layers.9.self_attn.q_proj",
170
+ "layers.9.self_attn.k_proj",
171
+ "layers.9.self_attn.v_proj",
172
+ "layers.9.self_attn.o_proj",
173
+ "layers.9.mlp.gate_proj",
174
+ "layers.9.mlp.up_proj",
175
+ "layers.9.mlp.down_proj",
176
+ "layers.10.self_attn.q_proj",
177
+ "layers.10.self_attn.k_proj",
178
+ "layers.10.self_attn.v_proj",
179
+ "layers.10.self_attn.o_proj",
180
+ "layers.10.mlp.gate_proj",
181
+ "layers.10.mlp.up_proj",
182
+ "layers.10.mlp.down_proj",
183
+ "layers.11.self_attn.q_proj",
184
+ "layers.11.self_attn.k_proj",
185
+ "layers.11.self_attn.v_proj",
186
+ "layers.11.self_attn.o_proj",
187
+ "layers.11.mlp.gate_proj",
188
+ "layers.11.mlp.up_proj",
189
+ "layers.11.mlp.down_proj",
190
+ "layers.12.self_attn.q_proj",
191
+ "layers.12.self_attn.k_proj",
192
+ "layers.12.self_attn.v_proj",
193
+ "layers.12.self_attn.o_proj",
194
+ "layers.12.mlp.gate_proj",
195
+ "layers.12.mlp.up_proj",
196
+ "layers.12.mlp.down_proj",
197
+ "layers.13.self_attn.q_proj",
198
+ "layers.13.self_attn.k_proj",
199
+ "layers.13.self_attn.v_proj",
200
+ "layers.13.self_attn.o_proj",
201
+ "layers.13.mlp.gate_proj",
202
+ "layers.13.mlp.up_proj",
203
+ "layers.13.mlp.down_proj",
204
+ "layers.14.self_attn.q_proj",
205
+ "layers.14.self_attn.k_proj",
206
+ "layers.14.self_attn.v_proj",
207
+ "layers.14.self_attn.o_proj",
208
+ "layers.14.mlp.gate_proj",
209
+ "layers.14.mlp.up_proj",
210
+ "layers.14.mlp.down_proj",
211
+ "layers.15.self_attn.q_proj",
212
+ "layers.15.self_attn.k_proj",
213
+ "layers.15.self_attn.v_proj",
214
+ "layers.15.self_attn.o_proj",
215
+ "layers.15.mlp.gate_proj",
216
+ "layers.15.mlp.up_proj",
217
+ "layers.15.mlp.down_proj",
218
+ "layers.16.self_attn.q_proj",
219
+ "layers.16.self_attn.k_proj",
220
+ "layers.16.self_attn.v_proj",
221
+ "layers.16.self_attn.o_proj",
222
+ "layers.16.mlp.gate_proj",
223
+ "layers.16.mlp.up_proj",
224
+ "layers.16.mlp.down_proj",
225
+ "layers.17.self_attn.q_proj",
226
+ "layers.17.self_attn.k_proj",
227
+ "layers.17.self_attn.v_proj",
228
+ "layers.17.self_attn.o_proj",
229
+ "layers.17.mlp.gate_proj",
230
+ "layers.17.mlp.up_proj",
231
+ "layers.17.mlp.down_proj",
232
+ "layers.18.self_attn.q_proj",
233
+ "layers.18.self_attn.k_proj",
234
+ "layers.18.self_attn.v_proj",
235
+ "layers.18.self_attn.o_proj",
236
+ "layers.18.mlp.gate_proj",
237
+ "layers.18.mlp.up_proj",
238
+ "layers.18.mlp.down_proj",
239
+ "layers.19.self_attn.q_proj",
240
+ "layers.19.self_attn.k_proj",
241
+ "layers.19.self_attn.v_proj",
242
+ "layers.19.self_attn.o_proj",
243
+ "layers.19.mlp.gate_proj",
244
+ "layers.19.mlp.up_proj",
245
+ "layers.19.mlp.down_proj",
246
+ "layers.20.self_attn.q_proj",
247
+ "layers.20.self_attn.k_proj",
248
+ "layers.20.self_attn.v_proj",
249
+ "layers.20.self_attn.o_proj",
250
+ "layers.20.mlp.gate_proj",
251
+ "layers.20.mlp.up_proj",
252
+ "layers.20.mlp.down_proj",
253
+ "layers.21.self_attn.q_proj",
254
+ "layers.21.self_attn.k_proj",
255
+ "layers.21.self_attn.v_proj",
256
+ "layers.21.self_attn.o_proj",
257
+ "layers.21.mlp.gate_proj",
258
+ "layers.21.mlp.up_proj",
259
+ "layers.21.mlp.down_proj",
260
+ "layers.22.self_attn.q_proj",
261
+ "layers.22.self_attn.k_proj",
262
+ "layers.22.self_attn.v_proj",
263
+ "layers.22.self_attn.o_proj",
264
+ "layers.22.mlp.gate_proj",
265
+ "layers.22.mlp.up_proj",
266
+ "layers.22.mlp.down_proj",
267
+ "layers.23.self_attn.q_proj",
268
+ "layers.23.self_attn.k_proj",
269
+ "layers.23.self_attn.v_proj",
270
+ "layers.23.self_attn.o_proj",
271
+ "layers.23.mlp.gate_proj",
272
+ "layers.23.mlp.up_proj",
273
+ "layers.23.mlp.down_proj",
274
+ "layers.24.self_attn.q_proj",
275
+ "layers.24.self_attn.k_proj",
276
+ "layers.24.self_attn.v_proj",
277
+ "layers.24.self_attn.o_proj",
278
+ "layers.24.mlp.gate_proj",
279
+ "layers.24.mlp.up_proj",
280
+ "layers.24.mlp.down_proj",
281
+ "layers.25.self_attn.q_proj",
282
+ "layers.25.self_attn.k_proj",
283
+ "layers.25.self_attn.v_proj",
284
+ "layers.25.self_attn.o_proj",
285
+ "layers.25.mlp.gate_proj",
286
+ "layers.25.mlp.up_proj",
287
+ "layers.25.mlp.down_proj",
288
+ "layers.26.self_attn.q_proj",
289
+ "layers.26.self_attn.k_proj",
290
+ "layers.26.self_attn.v_proj",
291
+ "layers.26.self_attn.o_proj",
292
+ "layers.26.mlp.gate_proj",
293
+ "layers.26.mlp.up_proj",
294
+ "layers.26.mlp.down_proj",
295
+ "layers.27.self_attn.q_proj",
296
+ "layers.27.self_attn.k_proj",
297
+ "layers.27.self_attn.v_proj",
298
+ "layers.27.self_attn.o_proj",
299
+ "layers.27.mlp.gate_proj",
300
+ "layers.27.mlp.up_proj",
301
+ "layers.27.mlp.down_proj",
302
+ "layers.28.self_attn.q_proj",
303
+ "layers.28.self_attn.k_proj",
304
+ "layers.28.self_attn.v_proj",
305
+ "layers.28.self_attn.o_proj",
306
+ "layers.28.mlp.gate_proj",
307
+ "layers.28.mlp.up_proj",
308
+ "layers.28.mlp.down_proj",
309
+ "layers.29.self_attn.q_proj",
310
+ "layers.29.self_attn.k_proj",
311
+ "layers.29.self_attn.v_proj",
312
+ "layers.29.self_attn.o_proj",
313
+ "layers.29.mlp.gate_proj",
314
+ "layers.29.mlp.up_proj",
315
+ "layers.29.mlp.down_proj",
316
+ "layers.30.self_attn.q_proj",
317
+ "layers.30.self_attn.k_proj",
318
+ "layers.30.self_attn.v_proj",
319
+ "layers.30.self_attn.o_proj",
320
+ "layers.30.mlp.gate_proj",
321
+ "layers.30.mlp.up_proj",
322
+ "layers.30.mlp.down_proj",
323
+ "layers.31.self_attn.q_proj",
324
+ "layers.31.self_attn.k_proj",
325
+ "layers.31.self_attn.v_proj",
326
+ "layers.31.self_attn.o_proj",
327
+ "layers.31.mlp.gate_proj",
328
+ "layers.31.mlp.up_proj",
329
+ "layers.31.mlp.down_proj",
330
+ "layers.32.self_attn.q_proj",
331
+ "layers.32.self_attn.k_proj",
332
+ "layers.32.self_attn.v_proj",
333
+ "layers.32.self_attn.o_proj",
334
+ "layers.32.mlp.gate_proj",
335
+ "layers.32.mlp.up_proj",
336
+ "layers.32.mlp.down_proj",
337
+ "layers.33.self_attn.q_proj",
338
+ "layers.33.self_attn.k_proj",
339
+ "layers.33.self_attn.v_proj",
340
+ "layers.33.self_attn.o_proj",
341
+ "layers.33.mlp.gate_proj",
342
+ "layers.33.mlp.up_proj",
343
+ "layers.33.mlp.down_proj",
344
+ "layers.34.self_attn.q_proj",
345
+ "layers.34.self_attn.k_proj",
346
+ "layers.34.self_attn.v_proj",
347
+ "layers.34.self_attn.o_proj",
348
+ "layers.34.mlp.gate_proj",
349
+ "layers.34.mlp.up_proj",
350
+ "layers.34.mlp.down_proj",
351
+ "layers.35.self_attn.q_proj",
352
+ "layers.35.self_attn.k_proj",
353
+ "layers.35.self_attn.v_proj",
354
+ "layers.35.self_attn.o_proj",
355
+ "layers.35.mlp.gate_proj",
356
+ "layers.35.mlp.up_proj",
357
+ "layers.35.mlp.down_proj"
358
+ ],
359
+ "load_and_tokenize_seconds": 3.123099140000704
360
+ }
snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/parent-snapshot.json ADDED
@@ -0,0 +1,273 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "created_utc": "2026-09-17T07:04:52.586355+00:00",
3
+ "purpose": "Frozen selected Spark B weights for a separate expanded-training-data warm start; no optimizer or RNG carry-over",
4
+ "directory": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent",
5
+ "usable_checkpoint": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/best.pt",
6
+ "source_host": "andy@192.168.8.204",
7
+ "source_directory": "/home/andy/ai/opensysone/runs/20260917T023137Z-24h/training",
8
+ "source_candidate": "spark-b-4b-refinement",
9
+ "source_campaign": "/home/andy/ai/opensysone/runs/20260917T023137Z-24h",
10
+ "source_step": 1500,
11
+ "source_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1",
12
+ "source_checkout_clean_at_snapshot": true,
13
+ "sha256": "5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024",
14
+ "model_provenance": {
15
+ "model_id": "Qwen/Qwen3-4B-Instruct-2507",
16
+ "revision": "cdbee75f17c01a7cc42f958dc650907174af0554",
17
+ "license": "apache-2.0"
18
+ },
19
+ "data_signature": "c00527e505fba87aabbfa65ba5fb7a67cf628ac7d46d7598f270936a894e3884",
20
+ "config": {
21
+ "command": "train",
22
+ "model": "/home/andy/ai/models/opensysone/Qwen3-4B-Instruct-2507-cdbee75f",
23
+ "dataset": "/home/andy/ai/opensysone/data/public-decisions-v1-20260916",
24
+ "output": "/home/andy/ai/opensysone/runs/20260917T023137Z-24h/training",
25
+ "resume": "/home/andy/ai/opensysone/runs/20260917T021605Z-train/artifacts/checkpoint.pt",
26
+ "warm_start": null,
27
+ "steps": null,
28
+ "epochs": 3,
29
+ "rank": 8,
30
+ "alpha": 16.0,
31
+ "head_only": false,
32
+ "two_pass": true,
33
+ "lr": 1e-05,
34
+ "head_lr": 1e-05,
35
+ "seed": 432,
36
+ "effective_batch": 4,
37
+ "branch_batch_size": 1,
38
+ "max_tokens": 512,
39
+ "save_steps": 250,
40
+ "save_seconds": 900,
41
+ "eval_steps": 500,
42
+ "validation_per_family": 128,
43
+ "patience": 8,
44
+ "deadline": "2026-09-17T16:00:00+00:00",
45
+ "schedule_steps": 5000,
46
+ "selection_metric": "crossfit_temperature_nll_v1",
47
+ "adapters": true
48
+ },
49
+ "prompt_version": "chat-verifier-v1",
50
+ "adapter_version": "additive-linear-v1",
51
+ "initialization": {
52
+ "kind": "resume",
53
+ "parent_checkpoint": "/home/andy/ai/opensysone/runs/20260917T021605Z-train/artifacts/checkpoint.pt",
54
+ "parent_checkpoint_sha256": "ece94b096b7fa37eec9e180a5f652da86a9f4f42d49f7f124850f046915237a8",
55
+ "restores_optimizer": true,
56
+ "restores_rng": true,
57
+ "parent_step": 8,
58
+ "parent_source_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1",
59
+ "parent_initialization": {
60
+ "kind": "warm_start",
61
+ "parent_checkpoint": "/home/andy/ai/opensysone/runs/20260917T021254Z-refinement-parent/best.pt",
62
+ "parent_checkpoint_sha256": "8956eb6c0cfbb02124aeefd99c3b418c55f55fdb9a64260350622d98dbba1aec",
63
+ "restores_optimizer": false,
64
+ "restores_rng": false,
65
+ "parent_step": 2500,
66
+ "parent_source_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1"
67
+ },
68
+ "selection_policy_change": {
69
+ "from": "crossfit_temperature_nll_v1",
70
+ "to": "crossfit_temperature_nll_v1",
71
+ "optimizer_and_rng_unchanged": true
72
+ }
73
+ },
74
+ "selection": {
75
+ "metric": "crossfit_temperature_nll_v1",
76
+ "score": 0.1701497127614862,
77
+ "raw_macro_nll": 0.190872636672039,
78
+ "accuracy": 0.947265625
79
+ },
80
+ "selected_metric_recomputed_exact": true,
81
+ "validation_identity_matches_frozen_reference": true,
82
+ "validation_decisions": 512,
83
+ "validation_reference_path": "/home/andy/ai/opensysone/runs/20260916T193741Z-24h/training/validation_step_002500_predictions.json",
84
+ "validation_reference_sha256": "16ff7745cdbfa53b87ccf2445796cf6a0c28d148a00e6c3926c72c36e75d058e",
85
+ "validation_identity_sha256": "9f7934043533a233543c382d1e6f0cfc28e719219825bab29885ddaa5a682750",
86
+ "per_family_counts": {
87
+ "arc": 128,
88
+ "banking": 128,
89
+ "boolq": 128,
90
+ "snli": 128
91
+ },
92
+ "source_data_manifest_sha256": "adf5a8ca2bab60cf7429a82b1b7a2d1ae7d3cc812de04dd195a478dfd7c6f628",
93
+ "source_split_sha256": {
94
+ "test": "ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819",
95
+ "validation": "411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f",
96
+ "calibration": "58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4",
97
+ "train": "314f2978aeeddec7b03a50412dae82595966ad4a0f8003f2e124b48e2497b550",
98
+ "holdout": "0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3"
99
+ },
100
+ "copied_files": {
101
+ "best.pt": {
102
+ "sha256": "5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024",
103
+ "bytes": 66202635
104
+ },
105
+ "validation_step_001500_predictions.json": {
106
+ "sha256": "1e0a2f838251c8eb411b59f4fa061db67ec308ce249d166901760b097c64bd6c",
107
+ "bytes": 331142
108
+ },
109
+ "best_validation_predictions.json": {
110
+ "sha256": "1e0a2f838251c8eb411b59f4fa061db67ec308ce249d166901760b097c64bd6c",
111
+ "bytes": 331142
112
+ },
113
+ "best_validation_selection.json": {
114
+ "sha256": "ced9908b7a6a9d6f682c7afac42d12316467b7a8aa83dcfd4b09b6af8aee6d6e",
115
+ "bytes": 19700
116
+ },
117
+ "manifest.json": {
118
+ "sha256": "0fe2c1bde4a47879d3ba040b97aa26c3025101dec10192e86188fe2cd98c33cf",
119
+ "bytes": 13002
120
+ },
121
+ "data_filter.json": {
122
+ "sha256": "3a514e34a8a6f7e35776bafe5b19015d6c5cec0eaabea5558aa896471aba3b56",
123
+ "bytes": 2360
124
+ },
125
+ "correctness_initial.json": {
126
+ "sha256": "f9ad8b0308a09577abc390412f214cc6289b1108ced3907570c6d4cf58539158",
127
+ "bytes": 354
128
+ },
129
+ "validation.jsonl": {
130
+ "sha256": "33a73b107bed65b65a7b3997ac6a1f097216e727c55fbf1ad02c9ccf0624bf20",
131
+ "bytes": 63612
132
+ }
133
+ },
134
+ "local_added_files": {
135
+ "source-data-manifest.json": {
136
+ "sha256": "adf5a8ca2bab60cf7429a82b1b7a2d1ae7d3cc812de04dd195a478dfd7c6f628",
137
+ "bytes": 5266
138
+ },
139
+ "config.json": {
140
+ "sha256": "09a10bf04d8ed6063cb4363f77469887a1eb7d78dbb2b85213f98458c8deb88f",
141
+ "bytes": 848
142
+ }
143
+ },
144
+ "remote_before": {
145
+ "source": "/home/andy/ai/opensysone/runs/20260917T023137Z-24h/training",
146
+ "selected_step": 1500,
147
+ "selection": {
148
+ "metric": "crossfit_temperature_nll_v1",
149
+ "score": 0.1701497127614862,
150
+ "raw_macro_nll": 0.190872636672039,
151
+ "accuracy": 0.947265625
152
+ },
153
+ "files": {
154
+ "best.pt": {
155
+ "sha256": "5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024",
156
+ "bytes": 66202635
157
+ },
158
+ "validation_step_001500_predictions.json": {
159
+ "sha256": "1e0a2f838251c8eb411b59f4fa061db67ec308ce249d166901760b097c64bd6c",
160
+ "bytes": 331142
161
+ },
162
+ "best_validation_predictions.json": {
163
+ "sha256": "1e0a2f838251c8eb411b59f4fa061db67ec308ce249d166901760b097c64bd6c",
164
+ "bytes": 331142
165
+ },
166
+ "best_validation_selection.json": {
167
+ "sha256": "ced9908b7a6a9d6f682c7afac42d12316467b7a8aa83dcfd4b09b6af8aee6d6e",
168
+ "bytes": 19700
169
+ },
170
+ "manifest.json": {
171
+ "sha256": "0fe2c1bde4a47879d3ba040b97aa26c3025101dec10192e86188fe2cd98c33cf",
172
+ "bytes": 13002
173
+ },
174
+ "data_filter.json": {
175
+ "sha256": "3a514e34a8a6f7e35776bafe5b19015d6c5cec0eaabea5558aa896471aba3b56",
176
+ "bytes": 2360
177
+ },
178
+ "correctness_initial.json": {
179
+ "sha256": "f9ad8b0308a09577abc390412f214cc6289b1108ced3907570c6d4cf58539158",
180
+ "bytes": 354
181
+ },
182
+ "validation.jsonl": {
183
+ "sha256": "33a73b107bed65b65a7b3997ac6a1f097216e727c55fbf1ad02c9ccf0624bf20",
184
+ "bytes": 63612
185
+ }
186
+ },
187
+ "checkout_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1",
188
+ "checkout_status": ""
189
+ },
190
+ "remote_after": {
191
+ "source": "/home/andy/ai/opensysone/runs/20260917T023137Z-24h/training",
192
+ "selected_step": 1500,
193
+ "selection": {
194
+ "metric": "crossfit_temperature_nll_v1",
195
+ "score": 0.1701497127614862,
196
+ "raw_macro_nll": 0.190872636672039,
197
+ "accuracy": 0.947265625
198
+ },
199
+ "files": {
200
+ "best.pt": {
201
+ "sha256": "5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024",
202
+ "bytes": 66202635
203
+ },
204
+ "validation_step_001500_predictions.json": {
205
+ "sha256": "1e0a2f838251c8eb411b59f4fa061db67ec308ce249d166901760b097c64bd6c",
206
+ "bytes": 331142
207
+ },
208
+ "best_validation_predictions.json": {
209
+ "sha256": "1e0a2f838251c8eb411b59f4fa061db67ec308ce249d166901760b097c64bd6c",
210
+ "bytes": 331142
211
+ },
212
+ "best_validation_selection.json": {
213
+ "sha256": "ced9908b7a6a9d6f682c7afac42d12316467b7a8aa83dcfd4b09b6af8aee6d6e",
214
+ "bytes": 19700
215
+ },
216
+ "manifest.json": {
217
+ "sha256": "0fe2c1bde4a47879d3ba040b97aa26c3025101dec10192e86188fe2cd98c33cf",
218
+ "bytes": 13002
219
+ },
220
+ "data_filter.json": {
221
+ "sha256": "3a514e34a8a6f7e35776bafe5b19015d6c5cec0eaabea5558aa896471aba3b56",
222
+ "bytes": 2360
223
+ },
224
+ "correctness_initial.json": {
225
+ "sha256": "f9ad8b0308a09577abc390412f214cc6289b1108ced3907570c6d4cf58539158",
226
+ "bytes": 354
227
+ },
228
+ "validation.jsonl": {
229
+ "sha256": "33a73b107bed65b65a7b3997ac6a1f097216e727c55fbf1ad02c9ccf0624bf20",
230
+ "bytes": 63612
231
+ }
232
+ },
233
+ "checkout_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1",
234
+ "checkout_status": ""
235
+ },
236
+ "stable_before_after_and_local_hashes": true,
237
+ "source_file_hashes_verified_against_git_commit": {
238
+ "jev_harness.py": "4d4e979cb7ae352bcdacaaa6d64045e6b5e550b1a9721d4bad545045bee6c67f",
239
+ "smoke_data.py": "06b3cbac1c8c4a86b8aecbee4459073cc3e46d4ddcd576392f3cb4805924f815",
240
+ "experiment.py": "c036cbe9fbf6f71c1bb141dcfbecfb06a2f4bd697b070d2274b1931480e92031",
241
+ "training_model.py": "d5b0aefeeb5290816bc0b669aa0a8cbbe27f6a12b9cb23c141ac9b9ae9ee4e65",
242
+ "smoke_train.py": "8cdeb2b397177fc9c26638aaa871501ddab8e3871aa1573ecd98f66590f5c228",
243
+ "decision_model.py": "a3d8aeb02a1ac765c6cc30ff175acad0664560f01ab5403e22cade924d17371e",
244
+ "selection.py": "be0a7a8496b5b830aa572ceba93606320f442063fd38180503fd6980dc1c578f",
245
+ "scripts/download_model.py": "72ad9a5de44d09e2ee4ed8afb7c3c0ff6fb48987410a3f7f0368353572bf1f4d",
246
+ "scripts/download_candidate.py": "d06a2c01be0cf6577f927fb37e3bc1eab014949fd934e4f4d9825674adba608e",
247
+ "scripts/run_precision.sh": "766d82b30cf3e83951f662685b4472ee053c7fbc171c4ff140823bf4d1c2782f",
248
+ "scripts/campaign_status.py": "1500b5e24f06231c7aafd7277aefd840582e35997e265f79db93614828d34411",
249
+ "scripts/verify_artifact.py": "9833350e9d72c0065b15206bb71c5a8b5a6b3185219db90074e369ede563a985",
250
+ "scripts/start_spark_candidate.sh": "c2ca18b008a144de7cb264c9fcca634d68c4a8a317db3638e8a70b3dbcff064b",
251
+ "scripts/run_experiment.sh": "661a6309fc54a2a8aff918f14a553c72dcb21730bd6a3cfc55d6ccd4700d11d3",
252
+ "scripts/fleet_campaign.py": "9e8d0511a579ace145b7ad079c11ce9b7465950569e86b22099caec2831d5f21",
253
+ "scripts/diagnose_parity.py": "08b5d66d316ebda98a2226251a4f952701f86a1d5726ce7a4d7e8fb22755da5a",
254
+ "scripts/prepare_public_data.py": "32ea84aa719818e1141b958b6ef27a85f7ddb86bfcd7c1585fc487b25253d977",
255
+ "scripts/run_smoke.sh": "39d59f2120f362729d1c2e384391b82be1e580dcc1aca0eaa7ab231115225574",
256
+ "scripts/launch_24h.py": "39c26dc10535d3adac09209b0743edf2ed384563e732512ad62f8c6f37161a83",
257
+ "scripts/investigate_precision.py": "609b744a926d8a45b87ba8d225e5312ee0c71a7096b21bd3589c5846e1dc847c"
258
+ },
259
+ "cpu_verification": {
260
+ "torch_version": "2.11.0+cu130",
261
+ "tensor_count": 506,
262
+ "trainable_parameters": 16517633,
263
+ "all_tensors_finite": true,
264
+ "all_tensors_cpu": true,
265
+ "cuda_initialized": false,
266
+ "oom_score_adj": "0",
267
+ "resumable_optimizer_present": false
268
+ },
269
+ "uncalibrated": true,
270
+ "reserved_predictions_accessed": false,
271
+ "active_training_modified": false,
272
+ "new_candidate_launched": false
273
+ }
snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/source-data-manifest.json ADDED
@@ -0,0 +1,136 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "version": "public-decisions-v1",
3
+ "sources": {
4
+ "snli": {
5
+ "repo": "stanfordnlp/snli",
6
+ "revision": "cdb5c3d5eed6ead6e5a341c8e56e669bb666725b",
7
+ "license": "CC-BY-SA-4.0",
8
+ "path": "plain_text"
9
+ },
10
+ "boolq": {
11
+ "repo": "google/boolq",
12
+ "revision": "35b264d03638db9f4ce671b711558bf7ff0f80d5",
13
+ "license": "CC-BY-SA-3.0",
14
+ "path": "data"
15
+ },
16
+ "arc": {
17
+ "repo": "allenai/ai2_arc",
18
+ "revision": "210d026faf9955653af8916fad021475a3f00453",
19
+ "license": "CC-BY-SA-4.0"
20
+ },
21
+ "banking": {
22
+ "repo": "PolyAI-LDN/task-specific-datasets",
23
+ "revision": "57ec275d8078af65b7731c2a98be812d844a6d6b",
24
+ "license": "CC-BY-4.0"
25
+ },
26
+ "social": {
27
+ "repo": "allenai/social_i_qa",
28
+ "revision": "8835ceb9141d7896d9d968634a9b21ae440e3ec5",
29
+ "url": "https://storage.googleapis.com/ai2-mosaic/public/socialiqa/socialiqa-train-dev.zip",
30
+ "license": "CC-BY-4.0",
31
+ "holdout_only": true
32
+ }
33
+ },
34
+ "raw_sha256": {
35
+ "raw/snli/plain_text/train-00000-of-00001.parquet": "ef9a7b25d97390a62aeda7abe26aec8640600f50b818eaeb9107097d60ac6620",
36
+ "raw/snli/plain_text/validation-00000-of-00001.parquet": "00f5ed8deaed007fef3022f0215b287efbab815b1bb31ac3f3ff4f4129d41ffe",
37
+ "raw/snli/plain_text/test-00000-of-00001.parquet": "4696deda851c4d2385f26b58f2e13f9ed9f08ea7b42a3f4c2b97a9d08448878c",
38
+ "raw/boolq/data/train-00000-of-00001.parquet": "4f028e992c0bd4df30b9f056f4946b64f5c23028034ff0ed5ea467d8538cc623",
39
+ "raw/boolq/data/validation-00000-of-00001.parquet": "52355d11524b4b874a9b9dcc278feb10f672d52c4f4eff9872e695ede59820f8",
40
+ "raw/arc/ARC-Challenge/train-00000-of-00001.parquet": "e488c1587ffdcfc8443f916c53488a95cd471c5790e0746c6bfe4cecf20962cb",
41
+ "raw/arc/ARC-Challenge/validation-00000-of-00001.parquet": "395a5c88d1580d69855fbaee9450270578df1ad5af6259771cd0a42c20e99f05",
42
+ "raw/arc/ARC-Challenge/test-00000-of-00001.parquet": "62f03257e737aed263f55c6abf87c7bb0028a44a6bdd2a26eb1279eb42c1d1e9",
43
+ "raw/arc/ARC-Easy/train-00000-of-00001.parquet": "b315db8a4be597dc7daa50a4e70d48dd7c990c32085629e6ccd8c926beaa80b5",
44
+ "raw/arc/ARC-Easy/validation-00000-of-00001.parquet": "ed890ff1e4cef7a7140d3a30dcea3ed2c9d467c6458f447ad9ef0176d8dcbb74",
45
+ "raw/arc/ARC-Easy/test-00000-of-00001.parquet": "4160597d618ae851c7eb04e281574f3f654776216ac6b6641588d64527b47177",
46
+ "raw/banking-categories.json": "53261da888122daf2d120d925458631d9619e15d82e56052e7a42e535ce32b63",
47
+ "raw/banking-train.csv": "b06e26ac675513959a63135f11b94ea7786ed02da65db93a5650d8838cbc664b",
48
+ "raw/banking-test.csv": "d12d6e3bc4c3103966ae786dc435913c0c563dfa328f5a3646d0e62cfeeb474d",
49
+ "raw/socialiqa-train-dev.zip": "ee073914a0fc33265cfbcfc50ec20df9b2e07809c3f420f599138d1f394ef5c3"
50
+ },
51
+ "source_script_sha256": "32ea84aa719818e1141b958b6ef27a85f7ddb86bfcd7c1585fc487b25253d977",
52
+ "split_sha256": {
53
+ "test": "ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819",
54
+ "validation": "411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f",
55
+ "calibration": "58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4",
56
+ "train": "314f2978aeeddec7b03a50412dae82595966ad4a0f8003f2e124b48e2497b550",
57
+ "holdout": "0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3"
58
+ },
59
+ "split_family_counts": {
60
+ "test": {
61
+ "banking": 512,
62
+ "boolq": 512,
63
+ "arc": 512,
64
+ "snli": 512
65
+ },
66
+ "validation": {
67
+ "boolq": 128,
68
+ "snli": 128,
69
+ "banking": 128,
70
+ "arc": 128
71
+ },
72
+ "calibration": {
73
+ "boolq": 128,
74
+ "snli": 128,
75
+ "banking": 128,
76
+ "arc": 128
77
+ },
78
+ "train": {
79
+ "banking": 9608,
80
+ "snli": 20000,
81
+ "boolq": 7988,
82
+ "arc": 3345
83
+ },
84
+ "holdout": {
85
+ "social": 768
86
+ }
87
+ },
88
+ "audit": {
89
+ "snli": {
90
+ "source_counts": {
91
+ "train": 549367,
92
+ "validation": 9842,
93
+ "test": 9824
94
+ },
95
+ "retained_train": 20000,
96
+ "group_definition": "normalized premise/passage/request/question/context"
97
+ },
98
+ "boolq": {
99
+ "source_counts": {
100
+ "train": 9427,
101
+ "validation": 3270
102
+ },
103
+ "retained_train": 7988,
104
+ "group_definition": "normalized premise/passage/request/question/context"
105
+ },
106
+ "arc": {
107
+ "source_counts": {
108
+ "train": 3370,
109
+ "validation": 869,
110
+ "test": 3548
111
+ },
112
+ "retained_train": 3345,
113
+ "group_definition": "normalized premise/passage/request/question/context"
114
+ },
115
+ "banking": {
116
+ "source_counts": {
117
+ "train": 10003,
118
+ "test": 3080,
119
+ "validation": 384
120
+ },
121
+ "retained_train": 9608,
122
+ "group_definition": "normalized premise/passage/request/question/context"
123
+ },
124
+ "social": {
125
+ "source_counts": {
126
+ "validation": 1949
127
+ },
128
+ "retained_train": 0,
129
+ "group_definition": "normalized premise/passage/request/question/context"
130
+ }
131
+ },
132
+ "banking_protocol": "target plus three uniformly sampled negatives, seeded and shuffled; not 77-way accuracy",
133
+ "holdout": "Social IQA family; no training or calibration rows",
134
+ "group_leakage": false,
135
+ "limitations": "Exact normalized groups only; pretrained contamination and semantic duplicates not excluded."
136
+ }
snapshots/20260917-expanded-pilot-hf-backup/artifacts/warm-start-parent/validation_step_001500_predictions.json ADDED
The diff for this file is too large to render. See raw diff
 
snapshots/20260917-expanded-pilot-hf-backup/backup-manifest.json ADDED
@@ -0,0 +1,402 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "artifacts": [
3
+ {
4
+ "artifact_directory": "artifacts/expanded-gx10-4b-pilot",
5
+ "base_model": {
6
+ "license": "apache-2.0",
7
+ "model_id": "Qwen/Qwen3-4B-Instruct-2507",
8
+ "revision": "cdbee75f17c01a7cc42f958dc650907174af0554"
9
+ },
10
+ "data_signature": "76183c642668602f42b7f3e71a3fe03bd5bd76f064fce4ba92351d8703396207",
11
+ "initialization": {
12
+ "data_transition": {
13
+ "data_signature": "76183c642668602f42b7f3e71a3fe03bd5bd76f064fce4ba92351d8703396207",
14
+ "dataset": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917",
15
+ "kind": "training_split_only",
16
+ "manifest_sha256": "fde6ee7ce2eca20cb22cdbbe4db0ddbdb29a8ea9d597906d88e545939a5b602c",
17
+ "parent_data_signature": "c00527e505fba87aabbfa65ba5fb7a67cf628ac7d46d7598f270936a894e3884",
18
+ "parent_dataset": "/home/andy/ai/opensysone/data/public-decisions-v1-20260916",
19
+ "parent_manifest_sha256": "adf5a8ca2bab60cf7429a82b1b7a2d1ae7d3cc812de04dd195a478dfd7c6f628",
20
+ "protected_split_sha256": {
21
+ "calibration": "58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4",
22
+ "holdout": "0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3",
23
+ "test": "ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819",
24
+ "validation": "411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f"
25
+ }
26
+ },
27
+ "kind": "warm_start",
28
+ "parent_checkpoint": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/best.pt",
29
+ "parent_checkpoint_sha256": "5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024",
30
+ "parent_source_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1",
31
+ "parent_step": 1500,
32
+ "restores_optimizer": false,
33
+ "restores_rng": false
34
+ },
35
+ "name": "expanded-gx10-4b-pilot",
36
+ "resumable": {
37
+ "config": {
38
+ "adapters": true,
39
+ "allow_train_data_change": true,
40
+ "alpha": 16.0,
41
+ "branch_batch_size": 1,
42
+ "command": "train",
43
+ "dataset": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917",
44
+ "deadline": "2026-09-17T16:00:00Z",
45
+ "effective_batch": 4,
46
+ "epochs": 3,
47
+ "eval_steps": 500,
48
+ "head_lr": 2e-05,
49
+ "head_only": false,
50
+ "lr": 2e-05,
51
+ "max_tokens": 512,
52
+ "model": "/home/andy/ai/models/opensysone/Qwen3-4B-Instruct-2507-cdbee75f",
53
+ "output": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts",
54
+ "patience": 8,
55
+ "rank": 8,
56
+ "resume": null,
57
+ "save_seconds": 900,
58
+ "save_steps": 250,
59
+ "schedule_steps": 3500,
60
+ "seed": 433,
61
+ "selection_metric": "crossfit_temperature_nll_v1",
62
+ "steps": 8,
63
+ "two_pass": true,
64
+ "validation_per_family": 128,
65
+ "warm_start": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/best.pt"
66
+ },
67
+ "optimizer_steps": [
68
+ 8.0
69
+ ],
70
+ "paired_selected_best_step": 0,
71
+ "path": "artifacts/expanded-gx10-4b-pilot/checkpoint.pt",
72
+ "sha256": "cf4cd775165cc2bb02113d17b94facd0db7b94fcff87b27389b3280fbb712469",
73
+ "step": 8
74
+ },
75
+ "selected": {
76
+ "accuracy": 0.947265625,
77
+ "path": "artifacts/expanded-gx10-4b-pilot/best.pt",
78
+ "prediction_count": 512,
79
+ "predictions_path": "artifacts/expanded-gx10-4b-pilot/validation_step_000000_predictions.json",
80
+ "raw_macro_nll": 0.190872636672039,
81
+ "selection_metric": "crossfit_temperature_nll_v1",
82
+ "selection_score": 0.1701497127614862,
83
+ "sha256": "5b9eccc4c4e2bf1306e2def3e663b66e9f0cd1f3791dff04313039eb07426be5",
84
+ "step": 0
85
+ },
86
+ "source_commit": "24b8ccf60d388f9cbb184e03a6ae260a1f5a8b86",
87
+ "source_file_sha256": {
88
+ "data_transition.py": "93aaa89b4de3aa78c34f03a5643e31f91f738c832395334966368df9b902621c",
89
+ "decision_model.py": "a3d8aeb02a1ac765c6cc30ff175acad0664560f01ab5403e22cade924d17371e",
90
+ "experiment.py": "c779c3936aa1c2c51052f035df7bc0895a2de79c9ffc6c50fb0ee848832e17c7",
91
+ "jev_harness.py": "4d4e979cb7ae352bcdacaaa6d64045e6b5e550b1a9721d4bad545045bee6c67f",
92
+ "playground.py": "b10c400421dd8558a7fef8ddde632676cdfe7f63edf94184f299f9ed569c010a",
93
+ "scripts/campaign_status.py": "1500b5e24f06231c7aafd7277aefd840582e35997e265f79db93614828d34411",
94
+ "scripts/diagnose_parity.py": "08b5d66d316ebda98a2226251a4f952701f86a1d5726ce7a4d7e8fb22755da5a",
95
+ "scripts/download_candidate.py": "d06a2c01be0cf6577f927fb37e3bc1eab014949fd934e4f4d9825674adba608e",
96
+ "scripts/download_model.py": "72ad9a5de44d09e2ee4ed8afb7c3c0ff6fb48987410a3f7f0368353572bf1f4d",
97
+ "scripts/fleet_campaign.py": "e69fdff96f92c6943b7895be11df444f016d0b744a1b9441995a5f8bb7af9d54",
98
+ "scripts/fleet_status.py": "2519ced157ef4ac4fa449eebabae5740d7527778d578b4ac6720583010fa5217",
99
+ "scripts/investigate_precision.py": "609b744a926d8a45b87ba8d225e5312ee0c71a7096b21bd3589c5846e1dc847c",
100
+ "scripts/launch_24h.py": "39c26dc10535d3adac09209b0743edf2ed384563e732512ad62f8c6f37161a83",
101
+ "scripts/prepare_expanded_data.py": "5c05478b29c84218784690f3c7c3ec994615fec1c57826591607189f191c56b4",
102
+ "scripts/prepare_public_data.py": "32ea84aa719818e1141b958b6ef27a85f7ddb86bfcd7c1585fc487b25253d977",
103
+ "scripts/publish_hf_final.py": "278efc5878d7ebc5d1171f3a735d6c7d78275d93650e5a8a16c3f55b8353a577",
104
+ "scripts/publish_hf_snapshot.py": "b5fe16a00fcbc5ab97121428c6ce750275ee193438e3ffe24aac5e3bb325c018",
105
+ "scripts/run_experiment.sh": "661a6309fc54a2a8aff918f14a553c72dcb21730bd6a3cfc55d6ccd4700d11d3",
106
+ "scripts/run_precision.sh": "766d82b30cf3e83951f662685b4472ee053c7fbc171c4ff140823bf4d1c2782f",
107
+ "scripts/run_smoke.sh": "39d59f2120f362729d1c2e384391b82be1e580dcc1aca0eaa7ab231115225574",
108
+ "scripts/start_spark_candidate.sh": "c2ca18b008a144de7cb264c9fcca634d68c4a8a317db3638e8a70b3dbcff064b",
109
+ "scripts/verify_artifact.py": "9833350e9d72c0065b15206bb71c5a8b5a6b3185219db90074e369ede563a985",
110
+ "scripts/verify_playground.cjs": "a55aa0e945baeb9e536c2b9cce7a9aa2364244b587e90477190add96f78601dc",
111
+ "scripts/verify_playground_layout.cjs": "dea3fa07c568c141cd58fead8c547db7196f4a48fe4ea4cea0312d2ba7bd8bb4",
112
+ "selection.py": "be0a7a8496b5b830aa572ceba93606320f442063fd38180503fd6980dc1c578f",
113
+ "smoke_data.py": "06b3cbac1c8c4a86b8aecbee4459073cc3e46d4ddcd576392f3cb4805924f815",
114
+ "smoke_train.py": "8cdeb2b397177fc9c26638aaa871501ddab8e3871aa1573ecd98f66590f5c228",
115
+ "training_model.py": "d5b0aefeeb5290816bc0b669aa0a8cbbe27f6a12b9cb23c141ac9b9ae9ee4e65"
116
+ },
117
+ "source_training_directory": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts"
118
+ },
119
+ {
120
+ "artifact_directory": "artifacts/warm-start-parent",
121
+ "base_model": {
122
+ "license": "apache-2.0",
123
+ "model_id": "Qwen/Qwen3-4B-Instruct-2507",
124
+ "revision": "cdbee75f17c01a7cc42f958dc650907174af0554"
125
+ },
126
+ "data_signature": "c00527e505fba87aabbfa65ba5fb7a67cf628ac7d46d7598f270936a894e3884",
127
+ "name": "warm-start-parent",
128
+ "role": "weights-only initialization; no optimizer carry-over",
129
+ "selected": {
130
+ "accuracy": 0.947265625,
131
+ "path": "artifacts/warm-start-parent/best.pt",
132
+ "prediction_count": 512,
133
+ "predictions_path": "artifacts/warm-start-parent/validation_step_001500_predictions.json",
134
+ "raw_macro_nll": 0.190872636672039,
135
+ "selection_metric": "crossfit_temperature_nll_v1",
136
+ "selection_score": 0.1701497127614862,
137
+ "sha256": "5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024",
138
+ "step": 1500
139
+ },
140
+ "source_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1",
141
+ "source_file_sha256": {
142
+ "decision_model.py": "a3d8aeb02a1ac765c6cc30ff175acad0664560f01ab5403e22cade924d17371e",
143
+ "experiment.py": "c036cbe9fbf6f71c1bb141dcfbecfb06a2f4bd697b070d2274b1931480e92031",
144
+ "jev_harness.py": "4d4e979cb7ae352bcdacaaa6d64045e6b5e550b1a9721d4bad545045bee6c67f",
145
+ "scripts/campaign_status.py": "1500b5e24f06231c7aafd7277aefd840582e35997e265f79db93614828d34411",
146
+ "scripts/diagnose_parity.py": "08b5d66d316ebda98a2226251a4f952701f86a1d5726ce7a4d7e8fb22755da5a",
147
+ "scripts/download_candidate.py": "d06a2c01be0cf6577f927fb37e3bc1eab014949fd934e4f4d9825674adba608e",
148
+ "scripts/download_model.py": "72ad9a5de44d09e2ee4ed8afb7c3c0ff6fb48987410a3f7f0368353572bf1f4d",
149
+ "scripts/fleet_campaign.py": "9e8d0511a579ace145b7ad079c11ce9b7465950569e86b22099caec2831d5f21",
150
+ "scripts/investigate_precision.py": "609b744a926d8a45b87ba8d225e5312ee0c71a7096b21bd3589c5846e1dc847c",
151
+ "scripts/launch_24h.py": "39c26dc10535d3adac09209b0743edf2ed384563e732512ad62f8c6f37161a83",
152
+ "scripts/prepare_public_data.py": "32ea84aa719818e1141b958b6ef27a85f7ddb86bfcd7c1585fc487b25253d977",
153
+ "scripts/run_experiment.sh": "661a6309fc54a2a8aff918f14a553c72dcb21730bd6a3cfc55d6ccd4700d11d3",
154
+ "scripts/run_precision.sh": "766d82b30cf3e83951f662685b4472ee053c7fbc171c4ff140823bf4d1c2782f",
155
+ "scripts/run_smoke.sh": "39d59f2120f362729d1c2e384391b82be1e580dcc1aca0eaa7ab231115225574",
156
+ "scripts/start_spark_candidate.sh": "c2ca18b008a144de7cb264c9fcca634d68c4a8a317db3638e8a70b3dbcff064b",
157
+ "scripts/verify_artifact.py": "9833350e9d72c0065b15206bb71c5a8b5a6b3185219db90074e369ede563a985",
158
+ "selection.py": "be0a7a8496b5b830aa572ceba93606320f442063fd38180503fd6980dc1c578f",
159
+ "smoke_data.py": "06b3cbac1c8c4a86b8aecbee4459073cc3e46d4ddcd576392f3cb4805924f815",
160
+ "smoke_train.py": "8cdeb2b397177fc9c26638aaa871501ddab8e3871aa1573ecd98f66590f5c228",
161
+ "training_model.py": "d5b0aefeeb5290816bc0b669aa0a8cbbe27f6a12b9cb23c141ac9b9ae9ee4e65"
162
+ },
163
+ "source_training_directory": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent"
164
+ }
165
+ ],
166
+ "calibrated": false,
167
+ "created_utc": "2026-09-17T07:22:43.416924+00:00",
168
+ "dataset": {
169
+ "diagnostics_selection_eligible": false,
170
+ "manifest_sha256": "fde6ee7ce2eca20cb22cdbbe4db0ddbdb29a8ea9d597906d88e545939a5b602c",
171
+ "path": "data/public-decisions-v2-20260917",
172
+ "payload_included": true,
173
+ "split_sha256": {
174
+ "calibration": "58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4",
175
+ "holdout": "0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3",
176
+ "test": "ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819",
177
+ "train": "d6d6a57c3aaaf6527647d0a21cc6027a58a6b40e2099eebb9a0689c0212ee372",
178
+ "validation": "411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f"
179
+ }
180
+ },
181
+ "files": {
182
+ "artifacts/expanded-gx10-4b-pilot/best.pt": {
183
+ "bytes": 66203019,
184
+ "sha256": "5b9eccc4c4e2bf1306e2def3e663b66e9f0cd1f3791dff04313039eb07426be5",
185
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts/best.pt"
186
+ },
187
+ "artifacts/expanded-gx10-4b-pilot/best_validation_selection.json": {
188
+ "bytes": 19700,
189
+ "sha256": "ced9908b7a6a9d6f682c7afac42d12316467b7a8aa83dcfd4b09b6af8aee6d6e",
190
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts/best_validation_selection.json"
191
+ },
192
+ "artifacts/expanded-gx10-4b-pilot/checkpoint.pt": {
193
+ "bytes": 198820477,
194
+ "sha256": "cf4cd775165cc2bb02113d17b94facd0db7b94fcff87b27389b3280fbb712469",
195
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts/checkpoint.pt"
196
+ },
197
+ "artifacts/expanded-gx10-4b-pilot/correctness_final.json": {
198
+ "bytes": 355,
199
+ "sha256": "d5a9fc3e02e6e5d204d4c5154c21d30dcf92aa7621620f1ea7f64e824a4ecfef",
200
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts/correctness_final.json"
201
+ },
202
+ "artifacts/expanded-gx10-4b-pilot/correctness_initial.json": {
203
+ "bytes": 356,
204
+ "sha256": "cd6b7a551708c15a09099a58d7863fc0eee17e8b5ff36fd3ed2fdb7147ffbf3a",
205
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts/correctness_initial.json"
206
+ },
207
+ "artifacts/expanded-gx10-4b-pilot/data_filter.json": {
208
+ "bytes": 2462,
209
+ "sha256": "f07eef84b3081ad86bb5b48f810bbed569a76ee8e82cba9228beec908232d79e",
210
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts/data_filter.json"
211
+ },
212
+ "artifacts/expanded-gx10-4b-pilot/exit_code": {
213
+ "bytes": 2,
214
+ "sha256": "9a271f2a916b0b6ee6cecb2426f0b3206ef074578be55d9bc94f6f3fe3ab86aa",
215
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/exit_code"
216
+ },
217
+ "artifacts/expanded-gx10-4b-pilot/manifest.json": {
218
+ "bytes": 14253,
219
+ "sha256": "80dc3efef131f59bc7bb5005bc6d1de46350c405604dd3a710aa1f2d34c3762b",
220
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts/manifest.json"
221
+ },
222
+ "artifacts/expanded-gx10-4b-pilot/summary.json": {
223
+ "bytes": 11417,
224
+ "sha256": "bc5e3ad91ecf3ef13ed3b82e408ebdd7110102a924f8b046e5ac3631f7fcdda7",
225
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts/summary.json"
226
+ },
227
+ "artifacts/expanded-gx10-4b-pilot/validation_step_000000_predictions.json": {
228
+ "bytes": 331142,
229
+ "sha256": "1e0a2f838251c8eb411b59f4fa061db67ec308ce249d166901760b097c64bd6c",
230
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts/best_validation_predictions.json"
231
+ },
232
+ "artifacts/warm-start-parent/best.pt": {
233
+ "bytes": 66202635,
234
+ "sha256": "5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024",
235
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/best.pt"
236
+ },
237
+ "artifacts/warm-start-parent/best_validation_selection.json": {
238
+ "bytes": 19700,
239
+ "sha256": "ced9908b7a6a9d6f682c7afac42d12316467b7a8aa83dcfd4b09b6af8aee6d6e",
240
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/best_validation_selection.json"
241
+ },
242
+ "artifacts/warm-start-parent/correctness_initial.json": {
243
+ "bytes": 354,
244
+ "sha256": "f9ad8b0308a09577abc390412f214cc6289b1108ced3907570c6d4cf58539158",
245
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/correctness_initial.json"
246
+ },
247
+ "artifacts/warm-start-parent/data_filter.json": {
248
+ "bytes": 2360,
249
+ "sha256": "3a514e34a8a6f7e35776bafe5b19015d6c5cec0eaabea5558aa896471aba3b56",
250
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/data_filter.json"
251
+ },
252
+ "artifacts/warm-start-parent/manifest.json": {
253
+ "bytes": 13002,
254
+ "sha256": "0fe2c1bde4a47879d3ba040b97aa26c3025101dec10192e86188fe2cd98c33cf",
255
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/manifest.json"
256
+ },
257
+ "artifacts/warm-start-parent/parent-snapshot.json": {
258
+ "bytes": 11628,
259
+ "sha256": "fa43228a19a32fc2caf5480799a2746b4da619b42406c0f3898d36059107050f",
260
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/parent-snapshot.json"
261
+ },
262
+ "artifacts/warm-start-parent/source-data-manifest.json": {
263
+ "bytes": 5266,
264
+ "sha256": "adf5a8ca2bab60cf7429a82b1b7a2d1ae7d3cc812de04dd195a478dfd7c6f628",
265
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/source-data-manifest.json"
266
+ },
267
+ "artifacts/warm-start-parent/validation_step_001500_predictions.json": {
268
+ "bytes": 331142,
269
+ "sha256": "1e0a2f838251c8eb411b59f4fa061db67ec308ce249d166901760b097c64bd6c",
270
+ "source_path": "/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/validation_step_001500_predictions.json"
271
+ },
272
+ "backup-tools/prepare_expansion_backup.py": {
273
+ "bytes": 21154,
274
+ "sha256": "0b546fd6b96e34316fddcb06314fa072d1ddb9c06d7e6979ccd790421b93f419",
275
+ "source_path": "/home/andy/projects/opensysone/scripts/prepare_expansion_backup.py"
276
+ },
277
+ "backup-tools/test_expansion_backup.py": {
278
+ "bytes": 11522,
279
+ "sha256": "3845483f6469dc30122ae31a2bcbf6b1ad7ce69663308266f831142cd5d4fd1e",
280
+ "source_path": "/home/andy/projects/opensysone/tests/test_expansion_backup.py"
281
+ },
282
+ "data/public-decisions-v1-20260916-manifest.json": {
283
+ "bytes": 5266,
284
+ "sha256": "adf5a8ca2bab60cf7429a82b1b7a2d1ae7d3cc812de04dd195a478dfd7c6f628",
285
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v1-20260916/manifest.json"
286
+ },
287
+ "data/public-decisions-v2-20260917/ATTRIBUTION.md": {
288
+ "bytes": 1345,
289
+ "sha256": "87f4430d84f2a6157593f9af44a356ede87bad18cc51fa9ae3db2fee96830691"
290
+ },
291
+ "data/public-decisions-v2-20260917/calibration.jsonl": {
292
+ "bytes": 303868,
293
+ "sha256": "58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4",
294
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917/calibration.jsonl"
295
+ },
296
+ "data/public-decisions-v2-20260917/diagnostics/new_sources.jsonl": {
297
+ "bytes": 318001,
298
+ "sha256": "93ec461769c925a7f10de76ff7b04118ab731ed5f49ebb5517b2d448fa1fdafc",
299
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917/diagnostics/new_sources.jsonl"
300
+ },
301
+ "data/public-decisions-v2-20260917/holdout.jsonl": {
302
+ "bytes": 307068,
303
+ "sha256": "0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3",
304
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917/holdout.jsonl"
305
+ },
306
+ "data/public-decisions-v2-20260917/manifest.json": {
307
+ "bytes": 12761,
308
+ "sha256": "fde6ee7ce2eca20cb22cdbbe4db0ddbdb29a8ea9d597906d88e545939a5b602c",
309
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917/manifest.json"
310
+ },
311
+ "data/public-decisions-v2-20260917/proof/data_filter.json": {
312
+ "bytes": 2462,
313
+ "sha256": "f07eef84b3081ad86bb5b48f810bbed569a76ee8e82cba9228beec908232d79e",
314
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917-proof/data_filter.json"
315
+ },
316
+ "data/public-decisions-v2-20260917/proof/postbuild-audit.json": {
317
+ "bytes": 6956,
318
+ "sha256": "a72ee77ba11017679548b06a2b956f7d2e56ca07e95eba165bb48aefbb75da18",
319
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917-proof/postbuild-audit.json"
320
+ },
321
+ "data/public-decisions-v2-20260917/proof/prepare_expanded_data.py": {
322
+ "bytes": 18386,
323
+ "sha256": "5c05478b29c84218784690f3c7c3ec994615fec1c57826591607189f191c56b4",
324
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917-proof/prepare_expanded_data.py"
325
+ },
326
+ "data/public-decisions-v2-20260917/proof/tokenization-proof.json": {
327
+ "bytes": 2220,
328
+ "sha256": "c2c506ad72513d573eec723adb3960a00f9416fb974e9359f7b61b108f1ae2bb",
329
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917-proof/tokenization-proof.json"
330
+ },
331
+ "data/public-decisions-v2-20260917/proof/tokenize_only.py": {
332
+ "bytes": 4027,
333
+ "sha256": "f30b0dab7c917f74676fd1cffc988ffb175ddb9a843f2029ce715d7e44bc7f96",
334
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917-proof/tokenize_only.py"
335
+ },
336
+ "data/public-decisions-v2-20260917/raw/commonsenseqa/README.md": {
337
+ "bytes": 7395,
338
+ "sha256": "172917e887dcc013fe0f0bd6aa8c810aea1e2be67d3bc7ff63e9b9d92075cc34",
339
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917/raw/commonsenseqa/README.md"
340
+ },
341
+ "data/public-decisions-v2-20260917/raw/hellaswag/README.md": {
342
+ "bytes": 7019,
343
+ "sha256": "cfe6e26e7e936a447a12f7eee50f2bffb0355c3b97459496d8c7296f65c5b353",
344
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917/raw/hellaswag/README.md"
345
+ },
346
+ "data/public-decisions-v2-20260917/raw/piqa/piqa/README.md": {
347
+ "bytes": 244,
348
+ "sha256": "f7bcb808a64959151698f2bca621572c4809459733d04b95d53f3c62588f1593",
349
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917/raw/piqa/piqa/README.md"
350
+ },
351
+ "data/public-decisions-v2-20260917/test.jsonl": {
352
+ "bytes": 1204426,
353
+ "sha256": "ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819",
354
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917/test.jsonl"
355
+ },
356
+ "data/public-decisions-v2-20260917/train.jsonl": {
357
+ "bytes": 55474225,
358
+ "sha256": "d6d6a57c3aaaf6527647d0a21cc6027a58a6b40e2099eebb9a0689c0212ee372",
359
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917/train.jsonl"
360
+ },
361
+ "data/public-decisions-v2-20260917/validation.jsonl": {
362
+ "bytes": 292566,
363
+ "sha256": "411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f",
364
+ "source_path": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917/validation.jsonl"
365
+ },
366
+ "fleet/plan.json": {
367
+ "bytes": 2790,
368
+ "sha256": "657979189140a298f383cbcf722c425c64f7b07b7e0332c32ac1b076445028e2",
369
+ "source_path": "/home/andy/ai/opensysone/runs/20260916T194403396250Z-fleet/plan.json"
370
+ },
371
+ "fleet/reference-validation-predictions.json": {
372
+ "bytes": 333136,
373
+ "sha256": "e671e1508185765552b0f933ba03f356be62143c531d8ef534457d34b1645c9b",
374
+ "source_path": "/home/andy/ai/opensysone/runs/20260916T183823Z-train/artifacts/validation_step_000040_predictions.json"
375
+ }
376
+ },
377
+ "fleet_plan": {
378
+ "captured_utc": "2026-09-17T07:22:42.214857+00:00",
379
+ "path": "fleet/plan.json",
380
+ "sha256": "657979189140a298f383cbcf722c425c64f7b07b7e0332c32ac1b076445028e2"
381
+ },
382
+ "format": "opensysone-backup-v1",
383
+ "gpu_initialized_during_backup": false,
384
+ "helper_sha256": "0b546fd6b96e34316fddcb06314fa072d1ddb9c06d7e6979ccd790421b93f419",
385
+ "purpose": "Completed expanded-training pilot with selected and resumable weights plus exact transformed dataset",
386
+ "reserved_predictions_accessed": false,
387
+ "restore_requirements": [
388
+ "Restore the pinned base weights separately.",
389
+ "Use the archived exact training source revision and recreate dataset paths or verify signatures after path relocation.",
390
+ "Keep checkpoint.pt, best.pt and matching validation evidence together; checkpoint.pt includes Adam/RNG state.",
391
+ "Each upstream dataset retains its recorded licence; no common project or dataset licence is asserted."
392
+ ],
393
+ "staging_directory": "/home/andy/ai/opensysone/exports/20260917-expanded-pilot-hf-backup",
394
+ "target_repository": "andyshu/opensysone",
395
+ "total_payload_bytes": 390326107,
396
+ "uploaded": false,
397
+ "validation_reference": {
398
+ "path": "fleet/reference-validation-predictions.json",
399
+ "sha256": "e671e1508185765552b0f933ba03f356be62143c531d8ef534457d34b1645c9b",
400
+ "source_path": "/home/andy/ai/opensysone/runs/20260916T183823Z-train/artifacts/validation_step_000040_predictions.json"
401
+ }
402
+ }
snapshots/20260917-expanded-pilot-hf-backup/backup-tools/prepare_expansion_backup.py ADDED
@@ -0,0 +1,331 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Prepare an immutable expanded-data pilot backup; never upload or load a model.
2
+
3
+ Usage: python scripts/prepare_expansion_backup.py --pilot /path/to/completed-run
4
+ --output /path/to/new-export --fleet-plan /path/to/fleet/plan.json
5
+
6
+ The output is accepted by scripts/publish_hf_snapshot.py. Publication is separate.
7
+ Only trusted local project checkpoints may be supplied (PyTorch pickle format).
8
+ """
9
+ import argparse
10
+ from datetime import datetime, timezone
11
+ import hashlib
12
+ import json
13
+ from pathlib import Path
14
+ import re
15
+ import shutil
16
+ import subprocess
17
+ import sys
18
+ import tempfile
19
+
20
+ PROJECT = Path(__file__).resolve().parents[1]
21
+ sys.path.insert(0, str(PROJECT))
22
+ from scripts.fleet_campaign import verify_candidate, verify_correctness
23
+ from scripts.publish_hf_snapshot import relative_name, verify_artifacts
24
+
25
+ SPLITS = ('train', 'validation', 'calibration', 'test', 'holdout')
26
+ COMPLETE = {'completed_step_target', 'completed_epochs', 'validation_early_stop'}
27
+
28
+
29
+ def digest(path):
30
+ result = hashlib.sha256()
31
+ with Path(path).open('rb') as handle:
32
+ for chunk in iter(lambda: handle.read(1024 * 1024), b''):
33
+ result.update(chunk)
34
+ return result.hexdigest()
35
+
36
+
37
+ def read_json(path):
38
+ return json.loads(Path(path).read_text())
39
+
40
+
41
+ def write_json(path, value):
42
+ Path(path).write_text(json.dumps(value, indent=2, sort_keys=True) + '\n')
43
+
44
+
45
+ def verified_dataset(path):
46
+ value = read_json(path / 'manifest.json')
47
+ if set(value.get('split_sha256', {})) != set(SPLITS):
48
+ raise ValueError('Dataset must contain exactly five registered splits')
49
+ for split, checksum in value['split_sha256'].items():
50
+ if digest(path / (split + '.jsonl')) != checksum:
51
+ raise ValueError('Dataset split checksum mismatch')
52
+ return value
53
+
54
+
55
+ def verify_source(root, manifest):
56
+ revision = manifest['git_commit']
57
+ if not re.fullmatch('[0-9a-f]{40}', revision):
58
+ raise ValueError('Training source revision is not exact')
59
+ if not manifest.get('source_sha256'):
60
+ raise ValueError('Training source hash inventory is missing')
61
+ for name, expected in manifest['source_sha256'].items():
62
+ relative_name(name)
63
+ content = subprocess.check_output(['git', 'show', revision + ':' + name], cwd=root,
64
+ stderr=subprocess.DEVNULL, timeout=30)
65
+ if hashlib.sha256(content).hexdigest() != expected:
66
+ raise ValueError('Training source does not match committed revision')
67
+ return revision
68
+
69
+
70
+ def finite_tensors(value, torch):
71
+ if isinstance(value, torch.Tensor):
72
+ return value.device.type == 'cpu' and bool(torch.isfinite(value).all())
73
+ if isinstance(value, dict):
74
+ return all(finite_tensors(item, torch) for item in value.values())
75
+ if isinstance(value, (list, tuple)):
76
+ return all(finite_tensors(item, torch) for item in value)
77
+ return True
78
+
79
+
80
+ def prepare(pilot, output, fleet_plan, repo_id='andyshu/opensysone', source_root=PROJECT):
81
+ import torch
82
+ torch.set_num_threads(4)
83
+ if torch.cuda.is_initialized():
84
+ raise ValueError('Backup must run without initialized CUDA')
85
+ pilot, output, source_root = Path(pilot).resolve(), Path(output).resolve(), Path(source_root).resolve()
86
+ if (pilot / 'artifacts').is_dir():
87
+ pilot = pilot / 'artifacts'
88
+ if output.exists() or output.is_relative_to(source_root) or output == source_root:
89
+ raise ValueError('Output must be a new directory outside source checkout')
90
+ if (pilot.parent / 'exit_code').read_text().strip() != '0':
91
+ raise ValueError('Pilot must have completed with exit code zero')
92
+ manifest, summary = read_json(pilot / 'manifest.json'), read_json(pilot / 'summary.json')
93
+ if summary.get('status') not in COMPLETE or summary.get('final_correctness_status') != 'passed':
94
+ raise ValueError('Pilot is not complete with final correctness gates passed')
95
+ verify_correctness(pilot)
96
+ revision = verify_source(source_root, manifest)
97
+ selected_hash, checkpoint_hash = digest(pilot / 'best.pt'), digest(pilot / 'checkpoint.pt')
98
+ if summary.get('best_sha256') != selected_hash or summary.get('checkpoint_sha256') != checkpoint_hash:
99
+ raise ValueError('Pilot summary does not match durable checkpoints')
100
+ best = torch.load(pilot / 'best.pt', map_location='cpu', weights_only=False)
101
+ latest = torch.load(pilot / 'checkpoint.pt', map_location='cpu', weights_only=False)
102
+ for saved in (best, latest):
103
+ if saved.get('format') != 'opensysone-adapter-v1' or not saved.get('trainable_state') or 'temperature' in saved:
104
+ raise ValueError('Expected raw trained adapter/head checkpoint')
105
+ if saved.get('source_commit') != revision or saved.get('config') != manifest['config']:
106
+ raise ValueError('Checkpoint source/config differs from pilot manifest')
107
+ if saved.get('data_signature') != manifest['data_signature'] or saved.get('model_provenance') != manifest['model_provenance']:
108
+ raise ValueError('Checkpoint data/model provenance differs from pilot manifest')
109
+ if not finite_tensors(saved.get('trainable_state'), torch):
110
+ raise ValueError('Checkpoint trainable tensors are nonfinite or non-CPU')
111
+ if latest['step'] != summary['completed_steps'] or not 0 <= best['step'] <= latest['step']:
112
+ raise ValueError('Checkpoint steps disagree with completed pilot')
113
+ for key in ('best_validation_macro_nll', 'best_validation_selection_score', 'selection_metric'):
114
+ if latest.get(key) != best.get(key) or summary.get(key) != best.get(key):
115
+ raise ValueError('Resumable checkpoint does not preserve selected best metrics')
116
+ if not {'optimizer', 'random_state', 'torch_rng', 'cuda_rng'}.issubset(latest):
117
+ raise ValueError('Resumable checkpoint lacks optimizer or random states')
118
+ optimizer = latest['optimizer']
119
+ if not optimizer.get('state') or not finite_tensors(optimizer, torch):
120
+ raise ValueError('Resumable optimizer state is missing or nonfinite')
121
+ optimizer_steps = sorted({float(item['step']) for item in optimizer['state'].values()})
122
+ if optimizer_steps != [float(latest['step'])] or latest['step'] <= 0:
123
+ raise ValueError('Optimizer step does not match the completed pilot')
124
+
125
+ dataset = Path(best['config']['dataset']).resolve()
126
+ data = verified_dataset(dataset)
127
+ base_root = Path(data['base_dataset']['path']).resolve()
128
+ base = verified_dataset(base_root)
129
+ if (data['base_dataset']['manifest_sha256'] != digest(base_root / 'manifest.json') or
130
+ data['base_dataset']['split_sha256'] != base['split_sha256'] or
131
+ any(data['split_sha256'][s] != base['split_sha256'][s] for s in SPLITS[1:])):
132
+ raise ValueError('Expanded dataset parent/protected split lineage mismatch')
133
+ def signature(value):
134
+ return hashlib.sha256(json.dumps({'data': value['split_sha256'], 'model': best['model_provenance'],
135
+ 'implementation': manifest['source_sha256']['training_model.py'],
136
+ 'max_tokens': best['config']['max_tokens']}, sort_keys=True).encode()).hexdigest()
137
+ expected_transition = {'kind': 'training_split_only', 'parent_dataset': str(base_root),
138
+ 'dataset': str(dataset), 'parent_data_signature': signature(base), 'data_signature': signature(data),
139
+ 'parent_manifest_sha256': digest(base_root / 'manifest.json'),
140
+ 'manifest_sha256': digest(dataset / 'manifest.json'),
141
+ 'protected_split_sha256': {s: base['split_sha256'][s] for s in SPLITS[1:]}}
142
+ for saved in (best, latest):
143
+ initial = saved.get('initialization', {})
144
+ if (initial.get('kind') != 'warm_start' or initial.get('restores_optimizer') is not False or
145
+ initial.get('restores_rng') is not False or initial.get('data_transition') != expected_transition or
146
+ saved['data_signature'] != signature(data)):
147
+ raise ValueError('Expanded pilot lacks verified fresh-optimizer data transition')
148
+ if data['source_script_sha256'] != manifest['source_sha256']['scripts/prepare_expanded_data.py']:
149
+ raise ValueError('Expanded dataset builder differs from committed training source')
150
+ diagnostics = data['diagnostics']
151
+ diagnostic_name = relative_name(diagnostics['path'])
152
+ if diagnostics.get('selection_eligible') is not False or digest(dataset / diagnostic_name) != diagnostics['sha256']:
153
+ raise ValueError('Diagnostic data checksum or selection exclusion mismatch')
154
+ plan_hash = digest(fleet_plan)
155
+ plan = read_json(fleet_plan)
156
+ plan_captured_utc = datetime.now(timezone.utc).isoformat()
157
+ reference_path = Path(plan['reference_predictions'])
158
+ if digest(reference_path) != plan['reference_sha256']:
159
+ raise ValueError('Fixed fleet validation reference checksum mismatch')
160
+ reference = read_json(reference_path)
161
+ parent_path = Path(best['initialization']['parent_checkpoint'])
162
+ parent_hash = best['initialization']['parent_checkpoint_sha256']
163
+ if digest(parent_path) != parent_hash:
164
+ raise ValueError('Warm-start parent checkpoint checksum mismatch')
165
+ parent = torch.load(parent_path, map_location='cpu', weights_only=False)
166
+ parent_manifest = read_json(parent_path.parent / 'manifest.json')
167
+ parent_revision = verify_source(source_root, parent_manifest)
168
+ if (parent['source_commit'] != parent_revision or parent_revision != best['initialization']['parent_source_commit'] or
169
+ parent['step'] != best['initialization']['parent_step'] or parent['data_signature'] != signature(base) or
170
+ parent['model_provenance'] != best['model_provenance'] or not finite_tensors(parent['trainable_state'], torch)):
171
+ raise ValueError('Warm-start parent provenance mismatch')
172
+ verify_correctness(parent_path.parent)
173
+ parent_predictions = parent_path.parent / f"validation_step_{parent['step']:06d}_predictions.json"
174
+ parent_metrics = verify_candidate(parent, read_json(parent_predictions), reference)
175
+ if read_json(parent_path.parent / 'best_validation_selection.json') != parent_metrics['selection']:
176
+ raise ValueError('Parent selection evidence disagrees with durable checkpoint')
177
+ predictions_path = None
178
+ choices = [pilot / f"validation_step_{best['step']:06d}_predictions.json",
179
+ pilot / 'best_validation_predictions.json']
180
+ if best['step'] == 0:
181
+ choices.append(pilot / 'initial_validation_predictions.json')
182
+ for path in choices:
183
+ if path.is_file():
184
+ try:
185
+ metrics = verify_candidate(best, read_json(path), reference)
186
+ except ValueError:
187
+ continue
188
+ predictions_path = path
189
+ break
190
+ if predictions_path is None:
191
+ raise ValueError('No matching fixed-validation evidence for durable best')
192
+ if read_json(pilot / 'best_validation_selection.json') != metrics['selection']:
193
+ raise ValueError('Pilot selection evidence disagrees with durable checkpoint')
194
+
195
+ output.parent.mkdir(parents=True, exist_ok=True)
196
+ staging = Path(tempfile.mkdtemp(prefix=output.name + '.staging-', dir=output.parent))
197
+ files = {}
198
+ def copy(source, name, expected=None):
199
+ source = Path(source)
200
+ name = relative_name(name)
201
+ before = digest(source)
202
+ if expected is not None and expected != before:
203
+ raise ValueError('Source checksum differs from recorded provenance')
204
+ target = staging / name
205
+ target.parent.mkdir(parents=True, exist_ok=True)
206
+ shutil.copyfile(source, target)
207
+ if digest(source) != before or digest(target) != before:
208
+ raise ValueError('Source changed while creating immutable backup')
209
+ files[name] = {'bytes': target.stat().st_size, 'sha256': before, 'source_path': str(source)}
210
+ artifact_directory = 'artifacts/expanded-gx10-4b-pilot'
211
+ try:
212
+ copy(fleet_plan, 'fleet/plan.json', plan_hash)
213
+ copy(reference_path, 'fleet/reference-validation-predictions.json', plan['reference_sha256'])
214
+ copy(Path(__file__).resolve(), 'backup-tools/prepare_expansion_backup.py')
215
+ test_helper = PROJECT / 'tests/test_expansion_backup.py'
216
+ if test_helper.is_file():
217
+ copy(test_helper, 'backup-tools/test_expansion_backup.py')
218
+ copy(pilot / 'best.pt', artifact_directory + '/best.pt', selected_hash)
219
+ copy(pilot / 'checkpoint.pt', artifact_directory + '/checkpoint.pt', checkpoint_hash)
220
+ prediction_name = artifact_directory + f"/validation_step_{best['step']:06d}_predictions.json"
221
+ copy(predictions_path, prediction_name)
222
+ for name in ('manifest.json', 'summary.json', 'correctness_initial.json', 'correctness_final.json',
223
+ 'data_filter.json', 'best_validation_selection.json'):
224
+ copy(pilot / name, artifact_directory + '/' + name)
225
+ copy(pilot.parent / 'exit_code', artifact_directory + '/exit_code')
226
+ parent_directory = 'artifacts/warm-start-parent'
227
+ copy(parent_path, parent_directory + '/best.pt', parent_hash)
228
+ parent_prediction_name = parent_directory + '/' + parent_predictions.name
229
+ copy(parent_predictions, parent_prediction_name)
230
+ for name in ('manifest.json', 'correctness_initial.json', 'data_filter.json', 'best_validation_selection.json',
231
+ 'parent-snapshot.json', 'source-data-manifest.json'):
232
+ copy(parent_path.parent / name, parent_directory + '/' + name)
233
+ data_prefix = 'data/' + dataset.name
234
+ copy(dataset / 'manifest.json', data_prefix + '/manifest.json', expected_transition['manifest_sha256'])
235
+ copy(base_root / 'manifest.json', 'data/' + base_root.name + '-manifest.json', expected_transition['parent_manifest_sha256'])
236
+ for split in SPLITS:
237
+ copy(dataset / (split + '.jsonl'), data_prefix + '/' + split + '.jsonl', data['split_sha256'][split])
238
+ copy(dataset / diagnostic_name, data_prefix + '/' + diagnostic_name, diagnostics['sha256'])
239
+ proof_directory = dataset.with_name(dataset.name + '-proof')
240
+ if proof_directory.is_dir():
241
+ for name in ('tokenization-proof.json', 'data_filter.json', 'postbuild-audit.json',
242
+ 'prepare_expanded_data.py', 'tokenize_only.py'):
243
+ copy(proof_directory / name, data_prefix + '/proof/' + name)
244
+ for family, info in data['sources'].items():
245
+ if 'license_reference' in info:
246
+ recorded = info['files'][info['license_reference']]
247
+ relative = relative_name(recorded['local_path'])
248
+ copy(dataset / relative, data_prefix + '/' + relative, recorded['sha256'])
249
+ notices = ['# Dataset attribution and transformations', '',
250
+ 'This private backup retains the upstream licences recorded in the dataset manifest.',
251
+ 'No blanket licence is assigned to this mixed-source collection or to the model.',
252
+ 'The transformed five splits and diagnostics are copied exactly; raw source datasets and token caches are omitted.',
253
+ 'Source pins, raw-file hashes, URLs, normalization, sampling and grouping transformations are recorded in manifest.json.',
254
+ 'Calibration/test/holdout files are copied as opaque bytes; their predictions are never used by this helper.', '',
255
+ '| Family | Upstream repository | Pinned revision | Recorded licence |', '| --- | --- | --- | --- |']
256
+ notices += [f"| {family} | {info['repo']} | {info['revision']} | {info['license']} |"
257
+ for family, info in sorted(data['sources'].items())]
258
+ notice_path = staging / data_prefix / 'ATTRIBUTION.md'
259
+ notice_path.write_text('\n'.join(notices) + '\n')
260
+ files[str(notice_path.relative_to(staging))] = {'bytes': notice_path.stat().st_size, 'sha256': digest(notice_path)}
261
+ artifact = {'name': 'expanded-gx10-4b-pilot', 'source_training_directory': str(pilot),
262
+ 'artifact_directory': artifact_directory, 'source_commit': revision,
263
+ 'source_file_sha256': manifest['source_sha256'], 'base_model': best['model_provenance'],
264
+ 'data_signature': best['data_signature'], 'initialization': best['initialization'],
265
+ 'selected': {'path': artifact_directory + '/best.pt', 'step': best['step'], 'sha256': selected_hash,
266
+ 'predictions_path': prediction_name, 'raw_macro_nll': metrics['macro_nll'],
267
+ 'selection_metric': metrics['selection_metric'], 'selection_score': metrics['selection_score'],
268
+ 'accuracy': metrics['accuracy'], 'prediction_count': metrics['count']},
269
+ 'resumable': {'path': artifact_directory + '/checkpoint.pt', 'step': latest['step'],
270
+ 'sha256': checkpoint_hash, 'optimizer_steps': optimizer_steps,
271
+ 'paired_selected_best_step': best['step'], 'config': latest['config']}}
272
+ parent_artifact = {'name': 'warm-start-parent', 'role': 'weights-only initialization; no optimizer carry-over',
273
+ 'source_training_directory': str(parent_path.parent), 'artifact_directory': parent_directory,
274
+ 'source_commit': parent_revision, 'source_file_sha256': parent_manifest['source_sha256'],
275
+ 'base_model': parent['model_provenance'], 'data_signature': parent['data_signature'],
276
+ 'selected': {'path': parent_directory + '/best.pt', 'step': parent['step'], 'sha256': parent_hash,
277
+ 'predictions_path': parent_prediction_name, 'raw_macro_nll': parent_metrics['macro_nll'],
278
+ 'selection_metric': parent_metrics['selection_metric'], 'selection_score': parent_metrics['selection_score'],
279
+ 'accuracy': parent_metrics['accuracy'], 'prediction_count': parent_metrics['count']}}
280
+ backup = {'format': 'opensysone-backup-v1', 'created_utc': datetime.now(timezone.utc).isoformat(),
281
+ 'target_repository': repo_id, 'staging_directory': str(output), 'uploaded': False,
282
+ 'purpose': 'Completed expanded-training pilot with selected and resumable weights plus exact transformed dataset',
283
+ 'calibrated': False, 'reserved_predictions_accessed': False, 'gpu_initialized_during_backup': False,
284
+ 'dataset': {'path': data_prefix, 'payload_included': True, 'diagnostics_selection_eligible': False,
285
+ 'manifest_sha256': expected_transition['manifest_sha256'], 'split_sha256': data['split_sha256']},
286
+ 'validation_reference': {'path': 'fleet/reference-validation-predictions.json',
287
+ 'sha256': plan['reference_sha256'], 'source_path': str(reference_path)},
288
+ 'fleet_plan': {'path': 'fleet/plan.json', 'sha256': plan_hash, 'captured_utc': plan_captured_utc},
289
+ 'artifacts': [artifact, parent_artifact], 'files': files, 'total_payload_bytes': sum(item['bytes'] for item in files.values()),
290
+ 'helper_sha256': digest(__file__),
291
+ 'restore_requirements': ['Restore the pinned base weights separately.',
292
+ 'Use the archived exact training source revision and recreate dataset paths or verify signatures after path relocation.',
293
+ 'Keep checkpoint.pt, best.pt and matching validation evidence together; checkpoint.pt includes Adam/RNG state.',
294
+ 'Each upstream dataset retains its recorded licence; no common project or dataset licence is asserted.']}
295
+ write_json(staging / 'backup-manifest.json', backup)
296
+ sums = {name: record['sha256'] for name, record in files.items()}
297
+ sums['backup-manifest.json'] = digest(staging / 'backup-manifest.json')
298
+ (staging / 'SHA256SUMS').write_text(''.join(f'{checksum} {name}\n' for name, checksum in sorted(sums.items())))
299
+ verify_artifacts(staging, repo_id)
300
+ if torch.cuda.is_initialized():
301
+ raise ValueError('Backup unexpectedly initialized CUDA')
302
+ for path in staging.rglob('*'):
303
+ if path.is_file():
304
+ path.chmod(0o444)
305
+ staging.rename(output)
306
+ return backup
307
+ except BaseException:
308
+ shutil.rmtree(staging)
309
+ raise
310
+
311
+
312
+ def main():
313
+ parser = argparse.ArgumentParser(description=__doc__)
314
+ parser.add_argument('--pilot', required=True, help='Completed local pilot run or its artifacts directory')
315
+ parser.add_argument('--output', required=True, help='Fresh immutable artifact directory outside the source checkout')
316
+ parser.add_argument('--fleet-plan', required=True)
317
+ parser.add_argument('--repo-id', default='andyshu/opensysone')
318
+ args = parser.parse_args()
319
+ Path('/proc/self/oom_score_adj').write_text('0')
320
+ try:
321
+ result = prepare(args.pilot, args.output, args.fleet_plan, args.repo_id)
322
+ except Exception as error:
323
+ print(json.dumps({'status': 'failed', 'error_type': type(error).__name__}), file=sys.stderr)
324
+ return 1
325
+ print(json.dumps({'status': 'prepared', 'output': args.output, 'payload_bytes': result['total_payload_bytes'],
326
+ 'source_commit': result['artifacts'][0]['source_commit'], 'uploaded': False}))
327
+ return 0
328
+
329
+
330
+ if __name__ == '__main__':
331
+ raise SystemExit(main())
snapshots/20260917-expanded-pilot-hf-backup/backup-tools/test_expansion_backup.py ADDED
@@ -0,0 +1,175 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import hashlib
2
+ import json
3
+ import math
4
+ from pathlib import Path
5
+ import subprocess
6
+ import tempfile
7
+ import unittest
8
+
9
+ import torch
10
+ from scripts import prepare_expansion_backup as backup
11
+ from scripts.publish_hf_snapshot import prepare as publication_prepare, verify_stage
12
+ from selection import validation_selection, SELECTION_METRIC
13
+
14
+
15
+ class ExpansionBackupTests(unittest.TestCase):
16
+ def setUp(self):
17
+ self.temp = tempfile.TemporaryDirectory()
18
+ self.addCleanup(self.temp.cleanup)
19
+ self.root = Path(self.temp.name)
20
+ self.source = self.root / 'source'
21
+ self.source.mkdir()
22
+ (self.source / 'scripts').mkdir()
23
+ (self.source / 'training_model.py').write_text('# unchanged implementation\n')
24
+ (self.source / 'scripts/prepare_expanded_data.py').write_text('# frozen data builder\n')
25
+ (self.source / 'HF_MODEL_CARD.md').write_text('---\nlicense: unknown\n---\nFixture model card\n')
26
+ for command in (['init', '-q'], ['add', '.'], ['-c', 'user.name=Fixture', '-c', 'user.email=fixture@example.invalid',
27
+ 'commit', '-qm', 'fixture']):
28
+ subprocess.run(['git', *command], cwd=self.source, check=True, capture_output=True)
29
+ self.revision = subprocess.check_output(['git', 'rev-parse', 'HEAD'], cwd=self.source, text=True).strip()
30
+ source_sha = {name: backup.digest(self.source / name) for name in ('training_model.py', 'scripts/prepare_expanded_data.py')}
31
+ self.old, self.new = self.root / 'v1', self.root / 'v2'
32
+ for directory in (self.old, self.new):
33
+ directory.mkdir()
34
+ for split in backup.SPLITS:
35
+ (directory / (split + '.jsonl')).write_bytes(b'opaque frozen payload\n')
36
+ (self.new / 'train.jsonl').write_bytes(b'opaque frozen payload\nexpanded train\n')
37
+ hashes = lambda directory: {s: backup.digest(directory / (s + '.jsonl')) for s in backup.SPLITS}
38
+ base = {'split_sha256': hashes(self.old)}
39
+ backup.write_json(self.old / 'manifest.json', base)
40
+ (self.new / 'diagnostics').mkdir()
41
+ (self.new / 'diagnostics/new_sources.jsonl').write_text('opaque diagnostic payload\n')
42
+ (self.new / 'raw/piqa').mkdir(parents=True)
43
+ (self.new / 'raw/piqa/README.md').write_text('Fixture upstream license evidence\n')
44
+ self.data = {'split_sha256': hashes(self.new), 'base_dataset': {'path': str(self.old),
45
+ 'manifest_sha256': backup.digest(self.old / 'manifest.json'), 'split_sha256': base['split_sha256']},
46
+ 'source_script_sha256': source_sha['scripts/prepare_expanded_data.py'],
47
+ 'diagnostics': {'path': 'diagnostics/new_sources.jsonl', 'selection_eligible': False,
48
+ 'sha256': backup.digest(self.new / 'diagnostics/new_sources.jsonl')},
49
+ 'sources': {'piqa': {'repo': 'fixture/upstream', 'revision': 'pinned', 'license': 'AFL-3.0',
50
+ 'license_reference': 'README.md', 'files': {'README.md': {'local_path': 'raw/piqa/README.md',
51
+ 'sha256': backup.digest(self.new / 'raw/piqa/README.md')}}}}}
52
+ backup.write_json(self.new / 'manifest.json', self.data)
53
+ proof = self.new.with_name(self.new.name + '-proof')
54
+ proof.mkdir()
55
+ for name in ('tokenization-proof.json', 'data_filter.json', 'postbuild-audit.json',
56
+ 'prepare_expanded_data.py', 'tokenize_only.py'):
57
+ (proof / name).write_bytes(b'opaque provenance fixture\n')
58
+ provenance = {'model_id': 'fixture', 'revision': 'pin'}
59
+ signature = lambda data: hashlib.sha256(json.dumps({'data': data['split_sha256'], 'model': provenance,
60
+ 'implementation': source_sha['training_model.py'], 'max_tokens': 512}, sort_keys=True).encode()).hexdigest()
61
+ self.rows = []
62
+ for family in ('arc', 'banking', 'boolq', 'snli'):
63
+ for index in range(128):
64
+ logits = [float(index % 3), 0.0]
65
+ logsum = math.log(sum(math.exp(v) for v in logits))
66
+ logs = [v - logsum for v in logits]
67
+ self.rows.append({'id': f'{family}-{index}', 'group': f'{family}-{index // 2}', 'family': family,
68
+ 'choices': ['a', 'b'], 'target': index % 2, 'logits': logits,
69
+ 'log_probabilities': logs, 'probabilities': [math.exp(v) for v in logs]})
70
+ scores = validation_selection(self.rows)
71
+ self.parent = self.root / 'parent'
72
+ self.parent.mkdir()
73
+ parent_config = {'dataset': str(self.old), 'validation_per_family': 128, 'max_tokens': 512}
74
+ saved = {'format': 'opensysone-adapter-v1', 'step': 20, 'config': parent_config,
75
+ 'trainable_state': {'head': torch.ones(2)}, 'source_commit': self.revision,
76
+ 'data_signature': signature(base), 'model_provenance': provenance,
77
+ 'selection_metric': SELECTION_METRIC, 'best_validation_selection_score': scores['score'],
78
+ 'best_validation_macro_nll': scores['raw_macro_nll']}
79
+ torch.save(saved, self.parent / 'best.pt')
80
+ self.run = self.root / 'run'
81
+ self.pilot = self.run / 'artifacts'
82
+ self.pilot.mkdir(parents=True)
83
+ (self.run / 'exit_code').write_text('0\n')
84
+ initial = {'kind': 'warm_start', 'restores_optimizer': False, 'restores_rng': False,
85
+ 'parent_checkpoint': str(self.parent / 'best.pt'), 'parent_checkpoint_sha256': backup.digest(self.parent / 'best.pt'),
86
+ 'parent_step': 20, 'parent_source_commit': self.revision,
87
+ 'data_transition': {'kind': 'training_split_only', 'parent_dataset': str(self.old), 'dataset': str(self.new),
88
+ 'parent_data_signature': signature(base), 'data_signature': signature(self.data),
89
+ 'parent_manifest_sha256': backup.digest(self.old / 'manifest.json'),
90
+ 'manifest_sha256': backup.digest(self.new / 'manifest.json'),
91
+ 'protected_split_sha256': {s: base['split_sha256'][s] for s in backup.SPLITS[1:]}}}
92
+ config = {**parent_config, 'dataset': str(self.new)}
93
+ self.best = {**saved, 'step': 0, 'config': config, 'initialization': initial, 'data_signature': signature(self.data)}
94
+ torch.save(self.best, self.pilot / 'best.pt')
95
+ latest = {**self.best, 'step': 8, 'optimizer': {'state': {0: {'step': torch.tensor(8.), 'exp_avg': torch.zeros(2)}}},
96
+ 'random_state': (), 'torch_rng': torch.zeros(2, dtype=torch.uint8), 'cuda_rng': []}
97
+ torch.save(latest, self.pilot / 'checkpoint.pt')
98
+ self.manifest = {'git_commit': self.revision, 'source_sha256': source_sha, 'config': config,
99
+ 'data_signature': signature(self.data), 'model_provenance': provenance}
100
+ backup.write_json(self.pilot / 'manifest.json', self.manifest)
101
+ backup.write_json(self.parent / 'manifest.json', {**self.manifest, 'config': parent_config, 'data_signature': signature(base)})
102
+ self.summary = {'status': 'completed_step_target', 'final_correctness_status': 'passed', 'completed_steps': 8,
103
+ 'best_sha256': backup.digest(self.pilot / 'best.pt'), 'checkpoint_sha256': backup.digest(self.pilot / 'checkpoint.pt'),
104
+ **{k: self.best[k] for k in ('best_validation_macro_nll', 'best_validation_selection_score', 'selection_metric')}}
105
+ backup.write_json(self.pilot / 'summary.json', self.summary)
106
+ checks = {name + '_probability_max_abs': 0.0 for name in ('branch_chunks_1', 'branch_chunks_2', 'branch_chunks_4',
107
+ 'question_isolation', 'candidate_permutation', 'repeat')}
108
+ checks['tolerance_probability_abs'] = 1e-5
109
+ for directory in (self.parent, self.pilot):
110
+ for name, value in [('correctness_initial.json', checks), ('correctness_final.json', checks),
111
+ ('data_filter.json', {}), ('best_validation_selection.json', scores)]:
112
+ backup.write_json(directory / name, value)
113
+ for name in ('parent-snapshot.json', 'source-data-manifest.json'):
114
+ backup.write_json(self.parent / name, {})
115
+ backup.write_json(self.pilot / 'best_validation_predictions.json', self.rows)
116
+ backup.write_json(self.parent / 'validation_step_000020_predictions.json', self.rows)
117
+ reference = self.root / 'reference.json'
118
+ backup.write_json(reference, self.rows)
119
+ self.plan = self.root / 'plan.json'
120
+ backup.write_json(self.plan, {'reference_predictions': str(reference), 'reference_sha256': backup.digest(reference)})
121
+ self.output = self.root / 'exports/snapshot'
122
+
123
+ def prepare(self):
124
+ return backup.prepare(self.run, self.output, self.plan, source_root=self.source)
125
+
126
+ def test_complete_backup_preserves_bytes_and_passes_existing_publisher(self):
127
+ result = self.prepare()
128
+ self.assertEqual(len(result['artifacts']), 2)
129
+ self.assertFalse(result['gpu_initialized_during_backup'])
130
+ self.assertEqual((self.output / 'data/v2/test.jsonl').read_bytes(), (self.new / 'test.jsonl').read_bytes())
131
+ self.assertTrue((self.output / 'data/v2/diagnostics/new_sources.jsonl').is_file())
132
+ self.assertEqual((self.output / 'data/v2/proof/tokenization-proof.json').read_bytes(), b'opaque provenance fixture\n')
133
+ self.assertTrue((self.output / 'artifacts/warm-start-parent/best.pt').is_file())
134
+ self.assertEqual(backup.read_json(self.output / 'fleet/reference-validation-predictions.json'), self.rows)
135
+ stage = publication_prepare(self.output, 'andyshu/opensysone', self.source, self.root / 'publication')
136
+ _, _, proof = verify_stage(stage)
137
+ self.assertEqual(proof['training_source_commits'], [self.revision])
138
+ self.assertFalse(torch.cuda.is_initialized())
139
+
140
+ def test_incomplete_pilot_rejected_before_any_stage(self):
141
+ self.summary['status'] = 'interrupted'
142
+ backup.write_json(self.pilot / 'summary.json', self.summary)
143
+ with self.assertRaisesRegex(ValueError, 'not complete'):
144
+ self.prepare()
145
+ self.assertFalse(self.output.parent.exists())
146
+
147
+ def test_source_revision_and_parent_corruption_rejected(self):
148
+ self.manifest['source_sha256']['training_model.py'] = '0' * 64
149
+ backup.write_json(self.pilot / 'manifest.json', self.manifest)
150
+ with self.assertRaisesRegex(ValueError, 'source does not match'):
151
+ self.prepare()
152
+ self.manifest['source_sha256']['training_model.py'] = backup.digest(self.source / 'training_model.py')
153
+ backup.write_json(self.pilot / 'manifest.json', self.manifest)
154
+ (self.parent / 'best.pt').write_bytes(b'corrupt')
155
+ with self.assertRaisesRegex(ValueError, 'parent checkpoint checksum'):
156
+ self.prepare()
157
+
158
+ def test_protected_data_and_diagnostics_mismatch_rejected(self):
159
+ (self.new / 'test.jsonl').write_bytes(b'changed reserved data')
160
+ self.data['split_sha256']['test'] = backup.digest(self.new / 'test.jsonl')
161
+ backup.write_json(self.new / 'manifest.json', self.data)
162
+ with self.assertRaisesRegex(ValueError, 'protected split'):
163
+ self.prepare()
164
+
165
+ def test_corrupted_completed_stage_rejected_by_publisher(self):
166
+ self.prepare()
167
+ path = self.output / 'data/v2/train.jsonl'
168
+ path.chmod(0o644)
169
+ path.write_bytes(b'changed')
170
+ with self.assertRaisesRegex(ValueError, 'checksum'):
171
+ backup.verify_artifacts(self.output, 'andyshu/opensysone')
172
+
173
+
174
+ if __name__ == '__main__':
175
+ unittest.main()
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v1-20260916-manifest.json ADDED
@@ -0,0 +1,136 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "version": "public-decisions-v1",
3
+ "sources": {
4
+ "snli": {
5
+ "repo": "stanfordnlp/snli",
6
+ "revision": "cdb5c3d5eed6ead6e5a341c8e56e669bb666725b",
7
+ "license": "CC-BY-SA-4.0",
8
+ "path": "plain_text"
9
+ },
10
+ "boolq": {
11
+ "repo": "google/boolq",
12
+ "revision": "35b264d03638db9f4ce671b711558bf7ff0f80d5",
13
+ "license": "CC-BY-SA-3.0",
14
+ "path": "data"
15
+ },
16
+ "arc": {
17
+ "repo": "allenai/ai2_arc",
18
+ "revision": "210d026faf9955653af8916fad021475a3f00453",
19
+ "license": "CC-BY-SA-4.0"
20
+ },
21
+ "banking": {
22
+ "repo": "PolyAI-LDN/task-specific-datasets",
23
+ "revision": "57ec275d8078af65b7731c2a98be812d844a6d6b",
24
+ "license": "CC-BY-4.0"
25
+ },
26
+ "social": {
27
+ "repo": "allenai/social_i_qa",
28
+ "revision": "8835ceb9141d7896d9d968634a9b21ae440e3ec5",
29
+ "url": "https://storage.googleapis.com/ai2-mosaic/public/socialiqa/socialiqa-train-dev.zip",
30
+ "license": "CC-BY-4.0",
31
+ "holdout_only": true
32
+ }
33
+ },
34
+ "raw_sha256": {
35
+ "raw/snli/plain_text/train-00000-of-00001.parquet": "ef9a7b25d97390a62aeda7abe26aec8640600f50b818eaeb9107097d60ac6620",
36
+ "raw/snli/plain_text/validation-00000-of-00001.parquet": "00f5ed8deaed007fef3022f0215b287efbab815b1bb31ac3f3ff4f4129d41ffe",
37
+ "raw/snli/plain_text/test-00000-of-00001.parquet": "4696deda851c4d2385f26b58f2e13f9ed9f08ea7b42a3f4c2b97a9d08448878c",
38
+ "raw/boolq/data/train-00000-of-00001.parquet": "4f028e992c0bd4df30b9f056f4946b64f5c23028034ff0ed5ea467d8538cc623",
39
+ "raw/boolq/data/validation-00000-of-00001.parquet": "52355d11524b4b874a9b9dcc278feb10f672d52c4f4eff9872e695ede59820f8",
40
+ "raw/arc/ARC-Challenge/train-00000-of-00001.parquet": "e488c1587ffdcfc8443f916c53488a95cd471c5790e0746c6bfe4cecf20962cb",
41
+ "raw/arc/ARC-Challenge/validation-00000-of-00001.parquet": "395a5c88d1580d69855fbaee9450270578df1ad5af6259771cd0a42c20e99f05",
42
+ "raw/arc/ARC-Challenge/test-00000-of-00001.parquet": "62f03257e737aed263f55c6abf87c7bb0028a44a6bdd2a26eb1279eb42c1d1e9",
43
+ "raw/arc/ARC-Easy/train-00000-of-00001.parquet": "b315db8a4be597dc7daa50a4e70d48dd7c990c32085629e6ccd8c926beaa80b5",
44
+ "raw/arc/ARC-Easy/validation-00000-of-00001.parquet": "ed890ff1e4cef7a7140d3a30dcea3ed2c9d467c6458f447ad9ef0176d8dcbb74",
45
+ "raw/arc/ARC-Easy/test-00000-of-00001.parquet": "4160597d618ae851c7eb04e281574f3f654776216ac6b6641588d64527b47177",
46
+ "raw/banking-categories.json": "53261da888122daf2d120d925458631d9619e15d82e56052e7a42e535ce32b63",
47
+ "raw/banking-train.csv": "b06e26ac675513959a63135f11b94ea7786ed02da65db93a5650d8838cbc664b",
48
+ "raw/banking-test.csv": "d12d6e3bc4c3103966ae786dc435913c0c563dfa328f5a3646d0e62cfeeb474d",
49
+ "raw/socialiqa-train-dev.zip": "ee073914a0fc33265cfbcfc50ec20df9b2e07809c3f420f599138d1f394ef5c3"
50
+ },
51
+ "source_script_sha256": "32ea84aa719818e1141b958b6ef27a85f7ddb86bfcd7c1585fc487b25253d977",
52
+ "split_sha256": {
53
+ "test": "ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819",
54
+ "validation": "411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f",
55
+ "calibration": "58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4",
56
+ "train": "314f2978aeeddec7b03a50412dae82595966ad4a0f8003f2e124b48e2497b550",
57
+ "holdout": "0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3"
58
+ },
59
+ "split_family_counts": {
60
+ "test": {
61
+ "banking": 512,
62
+ "boolq": 512,
63
+ "arc": 512,
64
+ "snli": 512
65
+ },
66
+ "validation": {
67
+ "boolq": 128,
68
+ "snli": 128,
69
+ "banking": 128,
70
+ "arc": 128
71
+ },
72
+ "calibration": {
73
+ "boolq": 128,
74
+ "snli": 128,
75
+ "banking": 128,
76
+ "arc": 128
77
+ },
78
+ "train": {
79
+ "banking": 9608,
80
+ "snli": 20000,
81
+ "boolq": 7988,
82
+ "arc": 3345
83
+ },
84
+ "holdout": {
85
+ "social": 768
86
+ }
87
+ },
88
+ "audit": {
89
+ "snli": {
90
+ "source_counts": {
91
+ "train": 549367,
92
+ "validation": 9842,
93
+ "test": 9824
94
+ },
95
+ "retained_train": 20000,
96
+ "group_definition": "normalized premise/passage/request/question/context"
97
+ },
98
+ "boolq": {
99
+ "source_counts": {
100
+ "train": 9427,
101
+ "validation": 3270
102
+ },
103
+ "retained_train": 7988,
104
+ "group_definition": "normalized premise/passage/request/question/context"
105
+ },
106
+ "arc": {
107
+ "source_counts": {
108
+ "train": 3370,
109
+ "validation": 869,
110
+ "test": 3548
111
+ },
112
+ "retained_train": 3345,
113
+ "group_definition": "normalized premise/passage/request/question/context"
114
+ },
115
+ "banking": {
116
+ "source_counts": {
117
+ "train": 10003,
118
+ "test": 3080,
119
+ "validation": 384
120
+ },
121
+ "retained_train": 9608,
122
+ "group_definition": "normalized premise/passage/request/question/context"
123
+ },
124
+ "social": {
125
+ "source_counts": {
126
+ "validation": 1949
127
+ },
128
+ "retained_train": 0,
129
+ "group_definition": "normalized premise/passage/request/question/context"
130
+ }
131
+ },
132
+ "banking_protocol": "target plus three uniformly sampled negatives, seeded and shuffled; not 77-way accuracy",
133
+ "holdout": "Social IQA family; no training or calibration rows",
134
+ "group_leakage": false,
135
+ "limitations": "Exact normalized groups only; pretrained contamination and semantic duplicates not excluded."
136
+ }
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/ATTRIBUTION.md ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Dataset attribution and transformations
2
+
3
+ This private backup retains the upstream licences recorded in the dataset manifest.
4
+ No blanket licence is assigned to this mixed-source collection or to the model.
5
+ The transformed five splits and diagnostics are copied exactly; raw source datasets and token caches are omitted.
6
+ Source pins, raw-file hashes, URLs, normalization, sampling and grouping transformations are recorded in manifest.json.
7
+ Calibration/test/holdout files are copied as opaque bytes; their predictions are never used by this helper.
8
+
9
+ | Family | Upstream repository | Pinned revision | Recorded licence |
10
+ | --- | --- | --- | --- |
11
+ | arc | allenai/ai2_arc | 210d026faf9955653af8916fad021475a3f00453 | CC-BY-SA-4.0 |
12
+ | banking | PolyAI-LDN/task-specific-datasets | 57ec275d8078af65b7731c2a98be812d844a6d6b | CC-BY-4.0 |
13
+ | boolq | google/boolq | 35b264d03638db9f4ce671b711558bf7ff0f80d5 | CC-BY-SA-3.0 |
14
+ | commonsenseqa | tau/commonsense_qa | 94630fe30dad47192a8546eb75f094926d47e155 | MIT |
15
+ | hellaswag | Rowan/hellaswag | 218ec52e09a7e7462a5400043bb9a69a41d06b76 | MIT |
16
+ | piqa | ybisk/ybisk.github.io | 21edab439af693b961be2f069e8690a88d3b4e37 | AFL-3.0 |
17
+ | snli | stanfordnlp/snli | cdb5c3d5eed6ead6e5a341c8e56e669bb666725b | CC-BY-SA-4.0 |
18
+ | social | allenai/social_i_qa | 8835ceb9141d7896d9d968634a9b21ae440e3ec5 | CC-BY-4.0 |
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/calibration.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/diagnostics/new_sources.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/holdout.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/manifest.json ADDED
@@ -0,0 +1,303 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "audit": {
3
+ "commonsenseqa": {
4
+ "available_training_rows": 9490,
5
+ "cap_excluded_rows": 0,
6
+ "diagnostic_group_rows_excluded_from_training": 128,
7
+ "diagnostic_groups": 128,
8
+ "official_reserved_groups": 2361,
9
+ "raw_train_rows": 9741,
10
+ "removed": {
11
+ "duplicate_training_identity_or_state": 1,
12
+ "invalid_source_row": 122
13
+ },
14
+ "retained_training_rows": 9490
15
+ },
16
+ "hellaswag": {
17
+ "available_training_rows": 32340,
18
+ "cap_excluded_rows": 16340,
19
+ "diagnostic_group_rows_excluded_from_training": 168,
20
+ "diagnostic_groups": 128,
21
+ "official_reserved_groups": 16580,
22
+ "raw_train_rows": 39905,
23
+ "removed": {
24
+ "duplicate_training_identity_or_state": 7333,
25
+ "official_evaluation_overlap": 64
26
+ },
27
+ "retained_training_rows": 16000
28
+ },
29
+ "piqa": {
30
+ "available_training_rows": 14361,
31
+ "cap_excluded_rows": 0,
32
+ "diagnostic_group_rows_excluded_from_training": 128,
33
+ "diagnostic_groups": 128,
34
+ "official_reserved_groups": 4711,
35
+ "raw_train_rows": 16113,
36
+ "removed": {
37
+ "duplicate_training_identity_or_state": 502,
38
+ "invalid_source_row": 1,
39
+ "official_evaluation_overlap": 1121
40
+ },
41
+ "retained_training_rows": 14361
42
+ }
43
+ },
44
+ "base_dataset": {
45
+ "manifest_sha256": "adf5a8ca2bab60cf7429a82b1b7a2d1ae7d3cc812de04dd195a478dfd7c6f628",
46
+ "path": "/home/andy/ai/opensysone/data/public-decisions-v1-20260916",
47
+ "split_sha256": {
48
+ "calibration": "58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4",
49
+ "holdout": "0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3",
50
+ "test": "ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819",
51
+ "train": "314f2978aeeddec7b03a50412dae82595966ad4a0f8003f2e124b48e2497b550",
52
+ "validation": "411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f"
53
+ }
54
+ },
55
+ "created_utc": "2026-09-17T07:03:41.972837+00:00",
56
+ "diagnostics": {
57
+ "family_counts": {
58
+ "commonsenseqa": 128,
59
+ "hellaswag": 128,
60
+ "piqa": 128
61
+ },
62
+ "path": "diagnostics/new_sources.jsonl",
63
+ "selection_eligible": false,
64
+ "sha256": "93ec461769c925a7f10de76ff7b04118ab731ed5f49ebb5517b2d448fa1fdafc",
65
+ "source_split": "train"
66
+ },
67
+ "group_leakage": false,
68
+ "holdout": "Original Social IQA holdout preserved byte-for-byte; no Social IQA training",
69
+ "limitations": "Exact normalized states/groups only; semantic duplicates and pretraining contamination not excluded. New-source diagnostics are outside fixed checkpoint selection.",
70
+ "normalization": "Unicode NFKC, casefold, collapsed whitespace, SHA256",
71
+ "protected_split_sha256": {
72
+ "calibration": "58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4",
73
+ "holdout": "0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3",
74
+ "test": "ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819",
75
+ "validation": "411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f"
76
+ },
77
+ "sampling": {
78
+ "diagnostic_groups_per_family": 128,
79
+ "max_new_per_family": 16000,
80
+ "new_training_rows": 39851,
81
+ "original_replay_bytes": 22856350,
82
+ "original_replay_rows": 40941,
83
+ "original_replay_sha256": "314f2978aeeddec7b03a50412dae82595966ad4a0f8003f2e124b48e2497b550",
84
+ "seed": 917,
85
+ "trainer_sampler": "unchanged uniform rows without replacement per epoch"
86
+ },
87
+ "source_git_commit": "b156513bf6ac1475eaa24fd79e7beb94d159364f",
88
+ "source_git_status": " M scripts/fleet_campaign.py\n M tests/test_fleet_campaign.py\n?? scripts/prepare_expanded_data.py\n?? tests/test_expanded_data.py\n",
89
+ "source_script_sha256": "5c05478b29c84218784690f3c7c3ec994615fec1c57826591607189f191c56b4",
90
+ "sources": {
91
+ "arc": {
92
+ "license": "CC-BY-SA-4.0",
93
+ "repo": "allenai/ai2_arc",
94
+ "revision": "210d026faf9955653af8916fad021475a3f00453"
95
+ },
96
+ "banking": {
97
+ "license": "CC-BY-4.0",
98
+ "repo": "PolyAI-LDN/task-specific-datasets",
99
+ "revision": "57ec275d8078af65b7731c2a98be812d844a6d6b"
100
+ },
101
+ "boolq": {
102
+ "license": "CC-BY-SA-3.0",
103
+ "path": "data",
104
+ "repo": "google/boolq",
105
+ "revision": "35b264d03638db9f4ce671b711558bf7ff0f80d5"
106
+ },
107
+ "commonsenseqa": {
108
+ "evaluation_access": "group/text metadata only; no evaluation labels used",
109
+ "expected_train_rows": 9741,
110
+ "files": {
111
+ "README.md": {
112
+ "bytes": 7395,
113
+ "local_path": "raw/commonsenseqa/README.md",
114
+ "sha256": "172917e887dcc013fe0f0bd6aa8c810aea1e2be67d3bc7ff63e9b9d92075cc34",
115
+ "url": "https://huggingface.co/datasets/tau/commonsense_qa/resolve/94630fe30dad47192a8546eb75f094926d47e155/README.md"
116
+ },
117
+ "data/test-00000-of-00001.parquet": {
118
+ "bytes": 151227,
119
+ "local_path": "raw/commonsenseqa/data/test-00000-of-00001.parquet",
120
+ "sha256": "19efe93223d1712397aaa44b3adc07f4fa50349206a611ca4f33afbec661fa5e",
121
+ "url": "https://huggingface.co/datasets/tau/commonsense_qa/resolve/94630fe30dad47192a8546eb75f094926d47e155/data/test-00000-of-00001.parquet"
122
+ },
123
+ "data/train-00000-of-00001.parquet": {
124
+ "bytes": 1247103,
125
+ "local_path": "raw/commonsenseqa/data/train-00000-of-00001.parquet",
126
+ "sha256": "b0449767ed986bfc2ca52b1244a46ef12f732756727f3cb0a4ab69ac8b3d282b",
127
+ "url": "https://huggingface.co/datasets/tau/commonsense_qa/resolve/94630fe30dad47192a8546eb75f094926d47e155/data/train-00000-of-00001.parquet"
128
+ },
129
+ "data/validation-00000-of-00001.parquet": {
130
+ "bytes": 160240,
131
+ "local_path": "raw/commonsenseqa/data/validation-00000-of-00001.parquet",
132
+ "sha256": "bdbd9bf9cc4d2349b24901038b2ab2f58e10e4e507ad2fd425dca55cd3cb6660",
133
+ "url": "https://huggingface.co/datasets/tau/commonsense_qa/resolve/94630fe30dad47192a8546eb75f094926d47e155/data/validation-00000-of-00001.parquet"
134
+ }
135
+ },
136
+ "group_columns": [
137
+ "question"
138
+ ],
139
+ "license": "MIT",
140
+ "license_reference": "README.md",
141
+ "provider": "huggingface",
142
+ "question": "Which candidate correctly answers the question in the state?",
143
+ "repo": "tau/commonsense_qa",
144
+ "revision": "94630fe30dad47192a8546eb75f094926d47e155",
145
+ "source_counts": {
146
+ "test": 1140,
147
+ "train": 9741,
148
+ "validation": 1221
149
+ },
150
+ "transform": "question -> state; choices.text -> choices; answerKey mapped through choices.label; question groups"
151
+ },
152
+ "hellaswag": {
153
+ "evaluation_access": "group/text metadata only; no evaluation labels used",
154
+ "expected_train_rows": 39905,
155
+ "files": {
156
+ "README.md": {
157
+ "bytes": 7019,
158
+ "local_path": "raw/hellaswag/README.md",
159
+ "sha256": "cfe6e26e7e936a447a12f7eee50f2bffb0355c3b97459496d8c7296f65c5b353",
160
+ "url": "https://huggingface.co/datasets/Rowan/hellaswag/resolve/218ec52e09a7e7462a5400043bb9a69a41d06b76/README.md"
161
+ },
162
+ "data/test-00000-of-00001.parquet": {
163
+ "bytes": 6112397,
164
+ "local_path": "raw/hellaswag/data/test-00000-of-00001.parquet",
165
+ "sha256": "e572fd5579bd1768b1985f47234f8bbe29247aca200a778b635bffc637714a41",
166
+ "url": "https://huggingface.co/datasets/Rowan/hellaswag/resolve/218ec52e09a7e7462a5400043bb9a69a41d06b76/data/test-00000-of-00001.parquet"
167
+ },
168
+ "data/train-00000-of-00001.parquet": {
169
+ "bytes": 24365524,
170
+ "local_path": "raw/hellaswag/data/train-00000-of-00001.parquet",
171
+ "sha256": "cacb12587faa63d7f723a72d61d12bfa94b140446f5a6a0a2e1c6906ab88bf02",
172
+ "url": "https://huggingface.co/datasets/Rowan/hellaswag/resolve/218ec52e09a7e7462a5400043bb9a69a41d06b76/data/train-00000-of-00001.parquet"
173
+ },
174
+ "data/validation-00000-of-00001.parquet": {
175
+ "bytes": 6315951,
176
+ "local_path": "raw/hellaswag/data/validation-00000-of-00001.parquet",
177
+ "sha256": "899813071e1e95efafec90f856e1987d2150fa4d020fc005df6962c259f660cd",
178
+ "url": "https://huggingface.co/datasets/Rowan/hellaswag/resolve/218ec52e09a7e7462a5400043bb9a69a41d06b76/data/validation-00000-of-00001.parquet"
179
+ }
180
+ },
181
+ "group_columns": [
182
+ "ctx",
183
+ "source_id"
184
+ ],
185
+ "license": "MIT",
186
+ "license_reference": "README.md",
187
+ "provider": "huggingface",
188
+ "question": "Which continuation is most plausible given this context?",
189
+ "repo": "Rowan/hellaswag",
190
+ "revision": "218ec52e09a7e7462a5400043bb9a69a41d06b76",
191
+ "source_counts": {
192
+ "test": 10003,
193
+ "train": 39905,
194
+ "validation": 10042
195
+ },
196
+ "transform": "ctx -> state; four endings -> choices; label -> target; source_id groups"
197
+ },
198
+ "piqa": {
199
+ "evaluation_access": "group/text metadata only; no evaluation labels used",
200
+ "expected_train_rows": 16113,
201
+ "files": {
202
+ "piqa/README.md": {
203
+ "bytes": 244,
204
+ "local_path": "raw/piqa/piqa/README.md",
205
+ "sha256": "f7bcb808a64959151698f2bca621572c4809459733d04b95d53f3c62588f1593",
206
+ "url": "https://raw.githubusercontent.com/ybisk/ybisk.github.io/21edab439af693b961be2f069e8690a88d3b4e37/piqa/README.md"
207
+ },
208
+ "piqa/data/tests.jsonl": {
209
+ "bytes": 814616,
210
+ "local_path": "raw/piqa/piqa/data/tests.jsonl",
211
+ "sha256": "402f1e2e61347db773e6e5e0a6b24f97396b59f6fd046dcdcbc12f483ac8553b",
212
+ "url": "https://raw.githubusercontent.com/ybisk/ybisk.github.io/21edab439af693b961be2f069e8690a88d3b4e37/piqa/data/tests.jsonl"
213
+ },
214
+ "piqa/data/train-labels.lst": {
215
+ "bytes": 32226,
216
+ "local_path": "raw/piqa/piqa/data/train-labels.lst",
217
+ "sha256": "df8ae080c41f02f88e81f6932d425f4b7ee72f433ad552e6a30d45e358e8f8bb",
218
+ "url": "https://raw.githubusercontent.com/ybisk/ybisk.github.io/21edab439af693b961be2f069e8690a88d3b4e37/piqa/data/train-labels.lst"
219
+ },
220
+ "piqa/data/train.jsonl": {
221
+ "bytes": 4382162,
222
+ "local_path": "raw/piqa/piqa/data/train.jsonl",
223
+ "sha256": "f805f9c96518d7b59f2f870d90f712457cda949979469d9fa3fd706ce22e0a1a",
224
+ "url": "https://raw.githubusercontent.com/ybisk/ybisk.github.io/21edab439af693b961be2f069e8690a88d3b4e37/piqa/data/train.jsonl"
225
+ },
226
+ "piqa/data/valid.jsonl": {
227
+ "bytes": 495929,
228
+ "local_path": "raw/piqa/piqa/data/valid.jsonl",
229
+ "sha256": "93503cc97c679e459b065c3d13e848282e44b2a25213985bed3e5d458abef72d",
230
+ "url": "https://raw.githubusercontent.com/ybisk/ybisk.github.io/21edab439af693b961be2f069e8690a88d3b4e37/piqa/data/valid.jsonl"
231
+ }
232
+ },
233
+ "group_columns": [
234
+ "goal"
235
+ ],
236
+ "license": "AFL-3.0",
237
+ "license_reference": "piqa/README.md",
238
+ "provider": "github",
239
+ "question": "Which solution best achieves the goal?",
240
+ "repo": "ybisk/ybisk.github.io",
241
+ "revision": "21edab439af693b961be2f069e8690a88d3b4e37",
242
+ "source_counts": {
243
+ "test": 3084,
244
+ "train": 16113,
245
+ "validation": 1838
246
+ },
247
+ "transform": "goal -> state; sol1/sol2 -> choices; train-labels.lst -> target; goal groups"
248
+ },
249
+ "snli": {
250
+ "license": "CC-BY-SA-4.0",
251
+ "path": "plain_text",
252
+ "repo": "stanfordnlp/snli",
253
+ "revision": "cdb5c3d5eed6ead6e5a341c8e56e669bb666725b"
254
+ },
255
+ "social": {
256
+ "holdout_only": true,
257
+ "license": "CC-BY-4.0",
258
+ "repo": "allenai/social_i_qa",
259
+ "revision": "8835ceb9141d7896d9d968634a9b21ae440e3ec5",
260
+ "url": "https://storage.googleapis.com/ai2-mosaic/public/socialiqa/socialiqa-train-dev.zip"
261
+ }
262
+ },
263
+ "split_family_counts": {
264
+ "calibration": {
265
+ "arc": 128,
266
+ "banking": 128,
267
+ "boolq": 128,
268
+ "snli": 128
269
+ },
270
+ "holdout": {
271
+ "social": 768
272
+ },
273
+ "test": {
274
+ "arc": 512,
275
+ "banking": 512,
276
+ "boolq": 512,
277
+ "snli": 512
278
+ },
279
+ "train": {
280
+ "arc": 3345,
281
+ "banking": 9608,
282
+ "boolq": 7988,
283
+ "commonsenseqa": 9490,
284
+ "hellaswag": 16000,
285
+ "piqa": 14361,
286
+ "snli": 20000
287
+ },
288
+ "validation": {
289
+ "arc": 128,
290
+ "banking": 128,
291
+ "boolq": 128,
292
+ "snli": 128
293
+ }
294
+ },
295
+ "split_sha256": {
296
+ "calibration": "58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4",
297
+ "holdout": "0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3",
298
+ "test": "ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819",
299
+ "train": "d6d6a57c3aaaf6527647d0a21cc6027a58a6b40e2099eebb9a0689c0212ee372",
300
+ "validation": "411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f"
301
+ },
302
+ "version": "public-decisions-v2"
303
+ }
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/data_filter.json ADDED
@@ -0,0 +1,100 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "calibration": {
3
+ "retained": 510,
4
+ "dropped_ids": [
5
+ "boolq:validation:200",
6
+ "boolq:validation:1836"
7
+ ],
8
+ "family_counts": {
9
+ "boolq": 126,
10
+ "snli": 128,
11
+ "banking": 128,
12
+ "arc": 128
13
+ },
14
+ "retained_id_sha256": "8a0d4add2dd95717d34915195dc7100878b8f443bf714656d54b336b629f6476",
15
+ "max_branch_tokens": 481
16
+ },
17
+ "holdout": {
18
+ "retained": 768,
19
+ "dropped_ids": [],
20
+ "family_counts": {
21
+ "social": 768
22
+ },
23
+ "retained_id_sha256": "805387bd9156d12d4d40b33e5926f209451f323814ad876a3f04a8abece3ae0b",
24
+ "max_branch_tokens": 123
25
+ },
26
+ "test": {
27
+ "retained": 2042,
28
+ "dropped_ids": [
29
+ "boolq:validation:1681",
30
+ "boolq:validation:1661",
31
+ "boolq:validation:2150",
32
+ "boolq:validation:2153",
33
+ "boolq:validation:3154",
34
+ "boolq:validation:561"
35
+ ],
36
+ "family_counts": {
37
+ "banking": 512,
38
+ "boolq": 506,
39
+ "arc": 512,
40
+ "snli": 512
41
+ },
42
+ "retained_id_sha256": "343960f63f954a0c05884459a13a3ef8560fd98c2caead3aa3c3b70022917ec5",
43
+ "max_branch_tokens": 440
44
+ },
45
+ "train": {
46
+ "retained": 80765,
47
+ "dropped_ids": [
48
+ "boolq:train:5085",
49
+ "boolq:train:1430",
50
+ "boolq:train:353",
51
+ "boolq:train:3547",
52
+ "boolq:train:5618",
53
+ "boolq:train:3872",
54
+ "boolq:train:6128",
55
+ "boolq:train:899",
56
+ "boolq:train:711",
57
+ "boolq:train:6969",
58
+ "boolq:train:7410",
59
+ "boolq:train:9405",
60
+ "boolq:train:3362",
61
+ "boolq:train:8317",
62
+ "boolq:train:4726",
63
+ "boolq:train:3163",
64
+ "boolq:train:7445",
65
+ "boolq:train:2181",
66
+ "boolq:train:8140",
67
+ "boolq:train:2141",
68
+ "boolq:train:204",
69
+ "boolq:train:1505",
70
+ "boolq:train:2352",
71
+ "boolq:train:9421",
72
+ "boolq:train:5517",
73
+ "boolq:train:6869",
74
+ "piqa:train:13223"
75
+ ],
76
+ "family_counts": {
77
+ "banking": 9608,
78
+ "snli": 20000,
79
+ "boolq": 7962,
80
+ "arc": 3345,
81
+ "hellaswag": 16000,
82
+ "piqa": 14360,
83
+ "commonsenseqa": 9490
84
+ },
85
+ "retained_id_sha256": "79e791c0e8b0f4312fdadcd62042a689d32d2bcf04c80419c8eebd94db99249c",
86
+ "max_branch_tokens": 509
87
+ },
88
+ "validation": {
89
+ "retained": 512,
90
+ "dropped_ids": [],
91
+ "family_counts": {
92
+ "boolq": 128,
93
+ "snli": 128,
94
+ "banking": 128,
95
+ "arc": 128
96
+ },
97
+ "retained_id_sha256": "b0ea1f550bee363d58849a00295771707fca92d9d8e630840d0f870a91a9f8a7",
98
+ "max_branch_tokens": 477
99
+ }
100
+ }
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/postbuild-audit.json ADDED
@@ -0,0 +1,239 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "added_training_rows_reconstructed": 39851,
3
+ "audited_utc": "2026-09-17T07:10:03.832729+00:00",
4
+ "dataset": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917",
5
+ "diagnostic_rows_reconstructed": 384,
6
+ "diagnostics": {
7
+ "family_counts": {
8
+ "commonsenseqa": 128,
9
+ "hellaswag": 128,
10
+ "piqa": 128
11
+ },
12
+ "training_group_overlap": 0,
13
+ "training_normalized_state_overlap": 0,
14
+ "unique_groups": 384
15
+ },
16
+ "family_histograms": {
17
+ "commonsenseqa": {
18
+ "choice_count": {
19
+ "5": 9490
20
+ },
21
+ "permuted_target_index": {
22
+ "0": 1871,
23
+ "1": 1940,
24
+ "2": 1923,
25
+ "3": 1850,
26
+ "4": 1906
27
+ },
28
+ "rows": 9490,
29
+ "source_target_index": {
30
+ "0": 1855,
31
+ "1": 1924,
32
+ "2": 1889,
33
+ "3": 1936,
34
+ "4": 1886
35
+ }
36
+ },
37
+ "hellaswag": {
38
+ "choice_count": {
39
+ "4": 16000
40
+ },
41
+ "permuted_target_index": {
42
+ "0": 3895,
43
+ "1": 4129,
44
+ "2": 3903,
45
+ "3": 4073
46
+ },
47
+ "rows": 16000,
48
+ "source_target_index": {
49
+ "0": 3923,
50
+ "1": 4015,
51
+ "2": 4026,
52
+ "3": 4036
53
+ }
54
+ },
55
+ "piqa": {
56
+ "choice_count": {
57
+ "2": 14361
58
+ },
59
+ "permuted_target_index": {
60
+ "0": 7252,
61
+ "1": 7109
62
+ },
63
+ "rows": 14361,
64
+ "source_target_index": {
65
+ "0": 7188,
66
+ "1": 7173
67
+ }
68
+ }
69
+ },
70
+ "family_mix": {
71
+ "filtered_counts": {
72
+ "arc": 3345,
73
+ "banking": 9608,
74
+ "boolq": 7962,
75
+ "commonsenseqa": 9490,
76
+ "hellaswag": 16000,
77
+ "piqa": 14360,
78
+ "snli": 20000
79
+ },
80
+ "filtered_original_replay_percent": 50.65932025010834,
81
+ "filtered_percent": {
82
+ "arc": 4.1416455147650595,
83
+ "banking": 11.896242184114406,
84
+ "boolq": 9.85823066922553,
85
+ "commonsenseqa": 11.750139293010585,
86
+ "hellaswag": 19.810561505602674,
87
+ "piqa": 17.7799789512784,
88
+ "snli": 24.763201882003344
89
+ },
90
+ "raw_counts": {
91
+ "arc": 3345,
92
+ "banking": 9608,
93
+ "boolq": 7988,
94
+ "commonsenseqa": 9490,
95
+ "hellaswag": 16000,
96
+ "piqa": 14361,
97
+ "snli": 20000
98
+ },
99
+ "raw_percent": {
100
+ "arc": 4.140261412020992,
101
+ "banking": 11.892266561045648,
102
+ "boolq": 9.887117536389741,
103
+ "commonsenseqa": 11.746212496286761,
104
+ "hellaswag": 19.803940984255867,
105
+ "piqa": 17.77527477968116,
106
+ "snli": 24.754926230319835
107
+ }
108
+ },
109
+ "manifest_sha256": "fde6ee7ce2eca20cb22cdbbe4db0ddbdb29a8ea9d597906d88e545939a5b602c",
110
+ "method": "Independent raw-source reconstruction; no preparer imports, model, torch, CUDA, predictions, or GPU calls",
111
+ "original_replay": {
112
+ "byte_exact_prefix": true,
113
+ "bytes": 22856350,
114
+ "rows": 40941,
115
+ "sha256": "314f2978aeeddec7b03a50412dae82595966ad4a0f8003f2e124b48e2497b550"
116
+ },
117
+ "pilot_seed433_first32": {
118
+ "all_three_new_families_present": true,
119
+ "family_counts": {
120
+ "arc": 1,
121
+ "banking": 2,
122
+ "boolq": 7,
123
+ "commonsenseqa": 3,
124
+ "hellaswag": 2,
125
+ "piqa": 7,
126
+ "snli": 10
127
+ },
128
+ "ordered_families": [
129
+ "commonsenseqa",
130
+ "boolq",
131
+ "boolq",
132
+ "snli",
133
+ "piqa",
134
+ "banking",
135
+ "snli",
136
+ "piqa",
137
+ "boolq",
138
+ "banking",
139
+ "boolq",
140
+ "arc",
141
+ "snli",
142
+ "boolq",
143
+ "snli",
144
+ "piqa",
145
+ "piqa",
146
+ "snli",
147
+ "commonsenseqa",
148
+ "commonsenseqa",
149
+ "snli",
150
+ "snli",
151
+ "hellaswag",
152
+ "boolq",
153
+ "snli",
154
+ "hellaswag",
155
+ "boolq",
156
+ "snli",
157
+ "piqa",
158
+ "piqa",
159
+ "snli",
160
+ "piqa"
161
+ ],
162
+ "ordered_ids": [
163
+ "commonsenseqa:train:73d3614d5927a6a9613a9c1f0834fb0f",
164
+ "boolq:train:4702",
165
+ "boolq:train:5568",
166
+ "snli:train:31785",
167
+ "piqa:train:339",
168
+ "banking:train:2818",
169
+ "snli:train:499223",
170
+ "piqa:train:1897",
171
+ "boolq:train:3754",
172
+ "banking:train:5442",
173
+ "boolq:train:124",
174
+ "arc:train:ARC-Easy:1303",
175
+ "snli:train:369827",
176
+ "boolq:train:5551",
177
+ "snli:train:256682",
178
+ "piqa:train:8016",
179
+ "piqa:train:3199",
180
+ "snli:train:465814",
181
+ "commonsenseqa:train:c0891ed8c5a43838f2f35f9b502621c6",
182
+ "commonsenseqa:train:efb02992f984a3ff2c06133eeab1e5ee",
183
+ "snli:train:343199",
184
+ "snli:train:219421",
185
+ "hellaswag:train:28480",
186
+ "boolq:train:6731",
187
+ "snli:train:219444",
188
+ "hellaswag:train:14613",
189
+ "boolq:train:392",
190
+ "snli:train:504298",
191
+ "piqa:train:13034",
192
+ "piqa:train:2477",
193
+ "snli:train:121240",
194
+ "piqa:train:4632"
195
+ ],
196
+ "sampler": "random.Random(433).shuffle over exact token-filtered row order; effective_batch4 means8updates"
197
+ },
198
+ "protected_files": {
199
+ "calibration": {
200
+ "byte_identical": true,
201
+ "sha256": "58fea4f180f16e5e0f2c9fd5f57487d3106415ec8e37bdd003e629ea944c51b4"
202
+ },
203
+ "holdout": {
204
+ "byte_identical": true,
205
+ "sha256": "0fb1bf6baf32374cf5dd8059428f058929884c903daa26be518ef14d33510cc3"
206
+ },
207
+ "test": {
208
+ "byte_identical": true,
209
+ "sha256": "ea61477a192d0a7174bcf0536f7664b547fdefe0bca9fe02c8e13c621e380819"
210
+ },
211
+ "validation": {
212
+ "byte_identical": true,
213
+ "sha256": "411199524c930d33fed8e1afa24597c59d400a500195fbec33d96afafd7ce74f"
214
+ }
215
+ },
216
+ "seconds": 2.2738394110056106,
217
+ "source_files_verified": {
218
+ "commonsenseqa/README.md": "172917e887dcc013fe0f0bd6aa8c810aea1e2be67d3bc7ff63e9b9d92075cc34",
219
+ "commonsenseqa/data/test-00000-of-00001.parquet": "19efe93223d1712397aaa44b3adc07f4fa50349206a611ca4f33afbec661fa5e",
220
+ "commonsenseqa/data/train-00000-of-00001.parquet": "b0449767ed986bfc2ca52b1244a46ef12f732756727f3cb0a4ab69ac8b3d282b",
221
+ "commonsenseqa/data/validation-00000-of-00001.parquet": "bdbd9bf9cc4d2349b24901038b2ab2f58e10e4e507ad2fd425dca55cd3cb6660",
222
+ "hellaswag/README.md": "cfe6e26e7e936a447a12f7eee50f2bffb0355c3b97459496d8c7296f65c5b353",
223
+ "hellaswag/data/test-00000-of-00001.parquet": "e572fd5579bd1768b1985f47234f8bbe29247aca200a778b635bffc637714a41",
224
+ "hellaswag/data/train-00000-of-00001.parquet": "cacb12587faa63d7f723a72d61d12bfa94b140446f5a6a0a2e1c6906ab88bf02",
225
+ "hellaswag/data/validation-00000-of-00001.parquet": "899813071e1e95efafec90f856e1987d2150fa4d020fc005df6962c259f660cd",
226
+ "piqa/piqa/README.md": "f7bcb808a64959151698f2bca621572c4809459733d04b95d53f3c62588f1593",
227
+ "piqa/piqa/data/tests.jsonl": "402f1e2e61347db773e6e5e0a6b24f97396b59f6fd046dcdcbc12f483ac8553b",
228
+ "piqa/piqa/data/train-labels.lst": "df8ae080c41f02f88e81f6932d425f4b7ee72f433ad552e6a30d45e358e8f8bb",
229
+ "piqa/piqa/data/train.jsonl": "f805f9c96518d7b59f2f870d90f712457cda949979469d9fa3fd706ce22e0a1a",
230
+ "piqa/piqa/data/valid.jsonl": "93503cc97c679e459b065c3d13e848282e44b2a25213985bed3e5d458abef72d"
231
+ },
232
+ "source_label_and_permutation_exact": true,
233
+ "source_train_counts": {
234
+ "commonsenseqa": 9741,
235
+ "hellaswag": 39905,
236
+ "piqa": 16113
237
+ },
238
+ "status": "passed"
239
+ }
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/prepare_expanded_data.py ADDED
@@ -0,0 +1,343 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Build an immutable train-only expansion while preserving frozen evaluation.
2
+
3
+ Only labeled official TRAIN rows become training or diagnostic decisions. Source
4
+ validation/test files are projected onto grouping metadata for exclusion only.
5
+ The existing evaluation files are copied byte-for-byte; no predictions are read.
6
+ """
7
+ import argparse
8
+ from collections import Counter
9
+ from datetime import datetime, timezone
10
+ import hashlib
11
+ import json
12
+ import os
13
+ from pathlib import Path
14
+ import random
15
+ import shutil
16
+ import subprocess
17
+ import tempfile
18
+ import unicodedata
19
+ import urllib.request
20
+
21
+
22
+ VERSION = "public-decisions-v2"
23
+ PROTECTED = ("validation", "calibration", "test", "holdout")
24
+ SPLITS = ("train", *PROTECTED)
25
+ SOURCES = {
26
+ "hellaswag": {
27
+ "repo": "Rowan/hellaswag", "provider": "huggingface",
28
+ "revision": "218ec52e09a7e7462a5400043bb9a69a41d06b76",
29
+ "license": "MIT", "license_reference": "README.md",
30
+ "expected_train_rows": 39905, "group_columns": ["ctx", "source_id"],
31
+ "transform": "ctx -> state; four endings -> choices; label -> target; source_id groups",
32
+ "question": "Which continuation is most plausible given this context?",
33
+ },
34
+ "piqa": {
35
+ "repo": "ybisk/ybisk.github.io", "provider": "github",
36
+ "revision": "21edab439af693b961be2f069e8690a88d3b4e37",
37
+ "license": "AFL-3.0", "license_reference": "piqa/README.md",
38
+ "expected_train_rows": 16113, "group_columns": ["goal"],
39
+ "transform": "goal -> state; sol1/sol2 -> choices; train-labels.lst -> target; goal groups",
40
+ "question": "Which solution best achieves the goal?",
41
+ },
42
+ "commonsenseqa": {
43
+ "repo": "tau/commonsense_qa", "provider": "huggingface",
44
+ "revision": "94630fe30dad47192a8546eb75f094926d47e155",
45
+ "license": "MIT", "license_reference": "README.md",
46
+ "expected_train_rows": 9741, "group_columns": ["question"],
47
+ "transform": "question -> state; choices.text -> choices; answerKey mapped through choices.label; question groups",
48
+ "question": "Which candidate correctly answers the question in the state?",
49
+ },
50
+ }
51
+
52
+
53
+ def sha256(path):
54
+ digest = hashlib.sha256()
55
+ with Path(path).open("rb") as handle:
56
+ for chunk in iter(lambda: handle.read(1024 * 1024), b""):
57
+ digest.update(chunk)
58
+ return digest.hexdigest()
59
+
60
+
61
+ def normalized(text):
62
+ return " ".join(unicodedata.normalize("NFKC", text).casefold().split())
63
+
64
+
65
+ def fingerprint(text):
66
+ return hashlib.sha256(normalized(text).encode()).hexdigest()
67
+
68
+
69
+ def json_write(path, value):
70
+ Path(path).write_text(json.dumps(value, indent=2, sort_keys=True) + "\n")
71
+
72
+
73
+ def read_rows(path):
74
+ with Path(path).open() as handle:
75
+ for line in handle:
76
+ if line.strip():
77
+ yield json.loads(line)
78
+
79
+
80
+ def source_keys(family, row):
81
+ """Return only group/text identities; never examine evaluation labels."""
82
+ state = row[{"hellaswag": "ctx", "piqa": "goal", "commonsenseqa": "question"}[family]]
83
+ if not isinstance(state, str) or not state.strip():
84
+ raise ValueError(f"{family}: missing nonempty state")
85
+ source_group = row.get("source_id") if family == "hellaswag" else state
86
+ if not isinstance(source_group, str) or not source_group.strip():
87
+ raise ValueError(f"{family}: missing source group")
88
+ return family + ":" + fingerprint(source_group), fingerprint(state)
89
+
90
+
91
+ def convert_row(family, index, raw):
92
+ source = SOURCES[family]
93
+ group, _ = source_keys(family, raw)
94
+ if family == "hellaswag":
95
+ state, choices, target = raw["ctx"], raw["endings"], int(raw["label"])
96
+ original_id = str(raw["ind"])
97
+ elif family == "piqa":
98
+ state, choices, target = raw["goal"], [raw["sol1"], raw["sol2"]], int(raw["label"])
99
+ original_id = str(index)
100
+ elif family == "commonsenseqa":
101
+ state, choices = raw["question"], raw["choices"]["text"]
102
+ target = raw["choices"]["label"].index(raw["answerKey"])
103
+ original_id = str(raw["id"])
104
+ else:
105
+ raise ValueError("Unknown source family")
106
+ expected_choices = {"hellaswag": 4, "piqa": 2, "commonsenseqa": 5}[family]
107
+ if (len(choices) != expected_choices or not 0 <= target < len(choices)
108
+ or any(not isinstance(choice, str) or not choice.strip() for choice in choices)
109
+ or len({normalized(choice) for choice in choices}) != len(choices)):
110
+ raise ValueError(f"{family}: invalid choices or target at row {index}")
111
+ row_id = f"{family}:train:{original_id}"
112
+ order = list(range(len(choices)))
113
+ random.Random(fingerprint("expanded-choices:" + row_id)).shuffle(order)
114
+ return {"id": row_id, "group": group, "family": family, "source_split": "train",
115
+ "source_record_id": original_id, "source_row_index": index,
116
+ "source_revision": source["revision"], "state": state,
117
+ "question": source["question"], "choices": [choices[i] for i in order],
118
+ "target": order.index(target)}
119
+
120
+
121
+ def download_sources(raw_directory):
122
+ """Download pinned artifacts and project nontraining files to group metadata."""
123
+ os.environ.setdefault("HF_HUB_DISABLE_XET", "1")
124
+ from huggingface_hub import hf_hub_download
125
+ import pyarrow.parquet as pq
126
+
127
+ raw_directory = Path(raw_directory)
128
+ raw_directory.mkdir(parents=True, exist_ok=False)
129
+ train, reserved, provenance = {}, {}, {}
130
+ for family, source in SOURCES.items():
131
+ directory = raw_directory / family
132
+ directory.mkdir()
133
+ files = {}
134
+
135
+ def acquire(filename):
136
+ if source["provider"] == "huggingface":
137
+ path = Path(hf_hub_download(source["repo"], filename, repo_type="dataset",
138
+ revision=source["revision"], local_dir=directory))
139
+ url = f"https://huggingface.co/datasets/{source['repo']}/resolve/{source['revision']}/{filename}"
140
+ else:
141
+ path = directory / filename
142
+ path.parent.mkdir(parents=True, exist_ok=True)
143
+ url = f"https://raw.githubusercontent.com/{source['repo']}/{source['revision']}/{filename}"
144
+ with urllib.request.urlopen(url, timeout=120) as response:
145
+ with path.open("xb") as handle:
146
+ shutil.copyfileobj(response, handle)
147
+ files[filename] = {"sha256": sha256(path), "bytes": path.stat().st_size,
148
+ "url": url, "local_path": str(path.relative_to(raw_directory.parent))}
149
+ return path
150
+
151
+ acquire(source["license_reference"])
152
+ reserved[family] = []
153
+ split_counts = {}
154
+ if source["provider"] == "huggingface":
155
+ for split in ("train", "validation", "test"):
156
+ path = acquire(f"data/{split}-00000-of-00001.parquet")
157
+ columns = None if split == "train" else source["group_columns"]
158
+ rows = pq.read_table(path, columns=columns).to_pylist()
159
+ split_counts[split] = len(rows)
160
+ if split == "train":
161
+ train[family] = rows
162
+ else:
163
+ reserved[family].extend(rows)
164
+ else:
165
+ rows = list(read_rows(acquire("piqa/data/train.jsonl")))
166
+ labels = acquire("piqa/data/train-labels.lst").read_text().splitlines()
167
+ if len(rows) != len(labels):
168
+ raise ValueError("PIQA train/label count mismatch")
169
+ train[family] = [{**row, "label": int(label)} for row, label in zip(rows, labels)]
170
+ split_counts["train"] = len(rows)
171
+ for split, name in (("validation", "valid.jsonl"), ("test", "tests.jsonl")):
172
+ # Only the goal participates in exclusions; no evaluation label file is fetched.
173
+ rows = [{"goal": row["goal"]} for row in read_rows(acquire("piqa/data/" + name))]
174
+ split_counts[split] = len(rows)
175
+ reserved[family].extend(rows)
176
+ if len(train[family]) != source["expected_train_rows"]:
177
+ raise ValueError(f"{family}: pinned train row count changed")
178
+ provenance[family] = {**source, "files": files, "source_counts": split_counts,
179
+ "evaluation_access": "group/text metadata only; no evaluation labels used"}
180
+ print(json.dumps({"event": "source_ready", "family": family, "counts": split_counts}), flush=True)
181
+ return train, reserved, provenance
182
+
183
+
184
+ def prepare_dataset(base, output, source_train, source_reserved, source_provenance,
185
+ max_new_per_family=16000, diagnostic_per_family=128, seed=917):
186
+ """Write a complete new dataset. All source rows passed here are in-memory data."""
187
+ base, output = Path(base).resolve(), Path(output).resolve()
188
+ if output == base or output.is_relative_to(base):
189
+ raise ValueError("Expanded output must be separate from its immutable base")
190
+ if max_new_per_family < 0 or diagnostic_per_family < 0:
191
+ raise ValueError("Sampling limits must be nonnegative")
192
+ base_manifest = json.loads((base / "manifest.json").read_text())
193
+ if set(base_manifest["split_sha256"]) != set(SPLITS):
194
+ raise ValueError("Base dataset must have exactly the five frozen decision splits")
195
+ for split, checksum in base_manifest["split_sha256"].items():
196
+ if sha256(base / f"{split}.jsonl") != checksum:
197
+ raise ValueError("Base dataset hash mismatch: " + split)
198
+ original = list(read_rows(base / "train.jsonl"))
199
+ if any(row["family"] in {"social", "socialiqa", "social_i_qa"} for row in original):
200
+ raise ValueError("Social IQA must never enter training")
201
+ original_ids = {row["id"] for row in original}
202
+ if len(original_ids) != len(original):
203
+ raise ValueError("Duplicate original training ID")
204
+ old_train_text = {fingerprint(row["state"]) for row in original}
205
+ reserved_groups, reserved_text = set(), set()
206
+ for split in PROTECTED:
207
+ for row in read_rows(base / f"{split}.jsonl"):
208
+ # Label/choice content does not affect exclusions or sampling.
209
+ reserved_groups.add(row["group"])
210
+ reserved_text.add(fingerprint(row["state"]))
211
+ if {row["group"] for row in original} & reserved_groups:
212
+ raise ValueError("Base dataset has train/evaluation group overlap")
213
+
214
+ additions, diagnostics, audit = [], [], {}
215
+ seen_ids, seen_text = set(original_ids), set(old_train_text)
216
+ for family in SOURCES:
217
+ official_groups, official_text = set(), set()
218
+ for row in source_reserved[family]:
219
+ group, text = source_keys(family, row)
220
+ official_groups.add(group)
221
+ official_text.add(text)
222
+ removed, candidates = Counter(), []
223
+ for index, raw in enumerate(source_train[family]):
224
+ try:
225
+ row = convert_row(family, index, raw)
226
+ except (ValueError, TypeError, KeyError, IndexError):
227
+ removed["invalid_source_row"] += 1
228
+ continue
229
+ text = fingerprint(row["state"])
230
+ if row["group"] in official_groups or text in official_text:
231
+ removed["official_evaluation_overlap"] += 1
232
+ elif row["group"] in reserved_groups or text in reserved_text:
233
+ removed["original_evaluation_overlap"] += 1
234
+ elif row["id"] in seen_ids or text in seen_text:
235
+ removed["duplicate_training_identity_or_state"] += 1
236
+ else:
237
+ seen_ids.add(row["id"])
238
+ seen_text.add(text)
239
+ candidates.append(row)
240
+ groups = sorted({row["group"] for row in candidates},
241
+ key=lambda group: fingerprint(f"diagnostics:{seed}:{group}"))
242
+ if len(groups) <= diagnostic_per_family:
243
+ raise ValueError(f"{family}: insufficient groups after exclusions")
244
+ diagnostic_groups = set(groups[:diagnostic_per_family])
245
+ diagnostic_rows = {}
246
+ train_rows = []
247
+ for row in candidates:
248
+ if row["group"] in diagnostic_groups:
249
+ diagnostic_rows.setdefault(row["group"], row)
250
+ else:
251
+ train_rows.append(row)
252
+ # Sampling is deterministic and independent of input/parquet row ordering.
253
+ train_rows.sort(key=lambda row: (fingerprint(f"train:{seed}:{row['id']}"), row["id"]))
254
+ selected = train_rows[:max_new_per_family] if max_new_per_family else train_rows
255
+ additions.extend(selected)
256
+ diagnostics.extend(diagnostic_rows[group] for group in groups[:diagnostic_per_family])
257
+ audit[family] = {"raw_train_rows": len(source_train[family]), "removed": dict(removed),
258
+ "official_reserved_groups": len(official_groups),
259
+ "diagnostic_groups": len(diagnostic_groups),
260
+ "diagnostic_group_rows_excluded_from_training": len(candidates) - len(train_rows),
261
+ "available_training_rows": len(train_rows), "retained_training_rows": len(selected),
262
+ "cap_excluded_rows": len(train_rows) - len(selected)}
263
+ if {row["group"] for row in additions} & {row["group"] for row in diagnostics}:
264
+ raise ValueError("Expanded train/diagnostic group overlap")
265
+ if {fingerprint(row["state"]) for row in additions} & reserved_text:
266
+ raise ValueError("Expanded training state overlaps frozen evaluation")
267
+
268
+ output.mkdir(parents=True, exist_ok=False)
269
+ for split in PROTECTED:
270
+ shutil.copyfile(base / f"{split}.jsonl", output / f"{split}.jsonl")
271
+ original_bytes = (base / "train.jsonl").read_bytes()
272
+ if original_bytes and not original_bytes.endswith(b"\n"):
273
+ raise ValueError("Original training JSONL must end in a newline for byte-exact replay")
274
+ # Original replay is a byte-exact prefix. The trainer shuffles all rows per epoch.
275
+ with (output / "train.jsonl").open("wb") as handle:
276
+ handle.write(original_bytes)
277
+ for row in additions:
278
+ handle.write((json.dumps(row, sort_keys=True) + "\n").encode())
279
+ diagnostic_path = output / "diagnostics" / "new_sources.jsonl"
280
+ diagnostic_path.parent.mkdir()
281
+ diagnostic_path.write_text("".join(json.dumps(row, sort_keys=True) + "\n" for row in diagnostics))
282
+ split_hashes = {split: sha256(output / f"{split}.jsonl") for split in SPLITS}
283
+ protected_hashes = {split: base_manifest["split_sha256"][split] for split in PROTECTED}
284
+ if any(split_hashes[split] != expected for split, expected in protected_hashes.items()):
285
+ raise ValueError("Frozen evaluation changed during copy")
286
+ counts = {**base_manifest["split_family_counts"],
287
+ "train": dict(Counter(row["family"] for row in [*original, *additions]))}
288
+ manifest = {"version": VERSION, "created_utc": datetime.now(timezone.utc).isoformat(),
289
+ "base_dataset": {"path": str(base), "manifest_sha256": sha256(base / "manifest.json"),
290
+ "split_sha256": base_manifest["split_sha256"]},
291
+ "sources": {**base_manifest["sources"], **source_provenance},
292
+ "source_script_sha256": sha256(__file__), "split_sha256": split_hashes,
293
+ "protected_split_sha256": protected_hashes, "split_family_counts": counts,
294
+ "diagnostics": {"path": str(diagnostic_path.relative_to(output)), "sha256": sha256(diagnostic_path),
295
+ "family_counts": dict(Counter(row["family"] for row in diagnostics)),
296
+ "selection_eligible": False, "source_split": "train"},
297
+ "sampling": {"seed": seed, "max_new_per_family": max_new_per_family,
298
+ "diagnostic_groups_per_family": diagnostic_per_family,
299
+ "trainer_sampler": "unchanged uniform rows without replacement per epoch",
300
+ "original_replay_rows": len(original), "original_replay_bytes": len(original_bytes),
301
+ "original_replay_sha256": base_manifest["split_sha256"]["train"],
302
+ "new_training_rows": len(additions)},
303
+ "audit": audit, "group_leakage": False,
304
+ "holdout": "Original Social IQA holdout preserved byte-for-byte; no Social IQA training",
305
+ "normalization": "Unicode NFKC, casefold, collapsed whitespace, SHA256",
306
+ "limitations": "Exact normalized states/groups only; semantic duplicates and pretraining contamination not excluded. New-source diagnostics are outside fixed checkpoint selection."}
307
+ json_write(output / "manifest.json", manifest)
308
+ return manifest
309
+
310
+
311
+ def main():
312
+ parser = argparse.ArgumentParser(description=__doc__)
313
+ parser.add_argument("--base", required=True)
314
+ parser.add_argument("--output", required=True)
315
+ parser.add_argument("--max-new-per-family", type=int, default=16000)
316
+ parser.add_argument("--diagnostic-per-family", type=int, default=128)
317
+ parser.add_argument("--seed", type=int, default=917)
318
+ args = parser.parse_args()
319
+ output = Path(args.output).expanduser().resolve()
320
+ repository = Path(__file__).resolve().parents[1]
321
+ if output.exists() or output.is_relative_to(repository):
322
+ parser.error("Output must be a new directory outside the source repository")
323
+ output.parent.mkdir(parents=True, exist_ok=True)
324
+ staging = Path(tempfile.mkdtemp(prefix=f".{output.name}-preparing-", dir=output.parent))
325
+ train, reserved, provenance = download_sources(staging / "raw")
326
+ prepared = staging / "dataset"
327
+ manifest = prepare_dataset(args.base, prepared, train, reserved, provenance,
328
+ args.max_new_per_family, args.diagnostic_per_family, args.seed)
329
+ shutil.move(staging / "raw", prepared / "raw")
330
+ manifest["source_git_commit"] = subprocess.check_output(["git", "rev-parse", "HEAD"], cwd=repository, text=True).strip()
331
+ manifest["source_git_status"] = subprocess.check_output(["git", "status", "--porcelain"], cwd=repository, text=True)
332
+ json_write(prepared / "manifest.json", manifest)
333
+ if output.exists():
334
+ raise ValueError("Output appeared during preparation; refusing overwrite")
335
+ prepared.rename(output)
336
+ staging.rmdir()
337
+ print(json.dumps({"event": "expanded_dataset_ready", "path": str(output),
338
+ "manifest_sha256": sha256(output / "manifest.json"),
339
+ "counts": manifest["split_family_counts"], "audit": manifest["audit"]}, indent=2), flush=True)
340
+
341
+
342
+ if __name__ == "__main__":
343
+ main()
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/tokenization-proof.json ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cuda_initialized": false,
3
+ "data_signature": "76183c642668602f42b7f3e71a3fe03bd5bd76f064fce4ba92351d8703396207",
4
+ "dataset": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917",
5
+ "diagnostics": {
6
+ "dropped_ids": [
7
+ "piqa:train:5528"
8
+ ],
9
+ "family_counts": {
10
+ "commonsenseqa": 128,
11
+ "hellaswag": 128,
12
+ "piqa": 127
13
+ },
14
+ "max_branch_tokens": 242,
15
+ "retained": 383
16
+ },
17
+ "max_tokens": 512,
18
+ "model": "/home/andy/ai/models/opensysone/Qwen3-4B-Instruct-2507-cdbee75f",
19
+ "model_provenance": {
20
+ "license": "apache-2.0",
21
+ "model_id": "Qwen/Qwen3-4B-Instruct-2507",
22
+ "revision": "cdbee75f17c01a7cc42f958dc650907174af0554"
23
+ },
24
+ "original_training_replay_exact": true,
25
+ "original_training_replay_retained": 40915,
26
+ "production_sequences": "training_model.TrainableScorer.sequences (same method, tokenizer only)",
27
+ "protected_token_replay": {
28
+ "calibration": {
29
+ "exact_equal": true,
30
+ "retained": 510,
31
+ "retained_id_sha256": "8a0d4add2dd95717d34915195dc7100878b8f443bf714656d54b336b629f6476"
32
+ },
33
+ "holdout": {
34
+ "exact_equal": true,
35
+ "retained": 768,
36
+ "retained_id_sha256": "805387bd9156d12d4d40b33e5926f209451f323814ad876a3f04a8abece3ae0b"
37
+ },
38
+ "test": {
39
+ "exact_equal": true,
40
+ "retained": 2042,
41
+ "retained_id_sha256": "343960f63f954a0c05884459a13a3ef8560fd98c2caead3aa3c3b70022917ec5"
42
+ },
43
+ "validation": {
44
+ "exact_equal": true,
45
+ "retained": 512,
46
+ "retained_id_sha256": "b0ea1f550bee363d58849a00295771707fca92d9d8e630840d0f870a91a9f8a7"
47
+ }
48
+ },
49
+ "seconds": 60.95969182101544,
50
+ "token_cache": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917/tokens-76183c642668602f.pt",
51
+ "token_cache_bytes": 166924109,
52
+ "token_cache_sha256": "4513bf6dbe68aab0503f917d895eb3b2b644386d28876a9946045082f7835d42",
53
+ "tokenizer_helper_sha256": "f30b0dab7c917f74676fd1cffc988ffb175ddb9a843f2029ce715d7e44bc7f96",
54
+ "train_family_counts": {
55
+ "arc": 3345,
56
+ "banking": 9608,
57
+ "boolq": 7962,
58
+ "commonsenseqa": 9490,
59
+ "hellaswag": 16000,
60
+ "piqa": 14360,
61
+ "snli": 20000
62
+ },
63
+ "training_model_sha256": "d5b0aefeeb5290816bc0b669aa0a8cbbe27f6a12b9cb23c141ac9b9ae9ee4e65"
64
+ }
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/proof/tokenize_only.py ADDED
@@ -0,0 +1,79 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """CPU-only cache preparation using the unmodified production sequence method."""
2
+ from collections import Counter
3
+ import hashlib
4
+ import json
5
+ from pathlib import Path
6
+ import sys
7
+ import time
8
+
9
+ ROOT = Path('/home/andy/projects/opensysone')
10
+ sys.path.insert(0, str(ROOT))
11
+ import torch
12
+ from transformers import AutoTokenizer
13
+ from decision_model import DecisionScorer
14
+ from training_model import TrainableScorer
15
+ from experiment import data_for, sha256
16
+
17
+ MODEL = Path('/home/andy/ai/models/opensysone/Qwen3-4B-Instruct-2507-cdbee75f')
18
+ DATASET = Path('/home/andy/ai/opensysone/data/public-decisions-v2-20260917')
19
+ PROOF = DATASET.parent / (DATASET.name + '-proof')
20
+
21
+ class TokenizerOnly:
22
+ _text = staticmethod(DecisionScorer._text)
23
+ _choices = DecisionScorer._choices
24
+ sequences = TrainableScorer.sequences
25
+
26
+ started = time.monotonic()
27
+ torch.set_num_threads(1)
28
+ assert not torch.cuda.is_initialized()
29
+ scorer = TokenizerOnly()
30
+ scorer.tokenizer = AutoTokenizer.from_pretrained(MODEL, local_files_only=True)
31
+ scorer.provenance = json.loads((MODEL / 'opensysone-provenance.json').read_text())
32
+ scorer.max_tokens = 512
33
+ data, signature = data_for(scorer, DATASET, PROOF)
34
+ manifest = json.loads((DATASET / 'manifest.json').read_text())
35
+ base_signature = hashlib.sha256(json.dumps({'data':manifest['base_dataset']['split_sha256'],
36
+ 'model':scorer.provenance,'implementation':sha256(ROOT/'training_model.py'),
37
+ 'max_tokens':scorer.max_tokens},sort_keys=True).encode()).hexdigest()
38
+ base_cache = Path(manifest['base_dataset']['path']) / f'tokens-{base_signature[:16]}.pt'
39
+ if not base_cache.is_file():
40
+ raise ValueError('Expected verified original4B token cache is unavailable')
41
+ original = torch.load(base_cache, map_location='cpu', weights_only=False)
42
+ assert original['signature'] == base_signature
43
+ protected = {}
44
+ for split in ('validation','calibration','test','holdout'):
45
+ before, after = original['data'][split], data[split]
46
+ assert before == after, f'Protected tokenized split changed: {split}'
47
+ protected[split] = {'exact_equal':True,'retained':len(after),
48
+ 'retained_id_sha256':hashlib.sha256('\n'.join(row['id'] for row in after).encode()).hexdigest()}
49
+ original_ids = {row['id'] for row in original['data']['train']}
50
+ new_replay = [row for row in data['train'] if row['id'] in original_ids]
51
+ assert new_replay == original['data']['train'], 'Original retained training replay changed'
52
+ retained, dropped = [], []
53
+ for line in (DATASET / manifest['diagnostics']['path']).read_text().splitlines():
54
+ row = json.loads(line)
55
+ try:
56
+ row['_sequences'] = scorer.sequences(row)
57
+ except ValueError as error:
58
+ if 'no truncation' not in str(error): raise
59
+ dropped.append(row['id'])
60
+ else:
61
+ retained.append(row)
62
+ assert not torch.cuda.is_initialized()
63
+ result = {'dataset':str(DATASET),'model':str(MODEL),'data_signature':signature,
64
+ 'max_tokens':scorer.max_tokens,'model_provenance':scorer.provenance,
65
+ 'production_sequences':'training_model.TrainableScorer.sequences (same method, tokenizer only)',
66
+ 'training_model_sha256':sha256(ROOT/'training_model.py'),
67
+ 'tokenizer_helper_sha256':sha256(__file__),'cuda_initialized':False,
68
+ 'token_cache':str(DATASET/f'tokens-{signature[:16]}.pt'),
69
+ 'token_cache_sha256':sha256(DATASET/f'tokens-{signature[:16]}.pt'),
70
+ 'token_cache_bytes':(DATASET/f'tokens-{signature[:16]}.pt').stat().st_size,
71
+ 'protected_token_replay':protected,'original_training_replay_exact':True,
72
+ 'original_training_replay_retained':len(new_replay),
73
+ 'train_family_counts':dict(Counter(row['family'] for row in data['train'])),
74
+ 'diagnostics':{'retained':len(retained),'dropped_ids':dropped,
75
+ 'family_counts':dict(Counter(row['family'] for row in retained)),
76
+ 'max_branch_tokens':max(len(s) for row in retained for s in row['_sequences'])},
77
+ 'seconds':time.monotonic()-started}
78
+ (PROOF/'tokenization-proof.json').write_text(json.dumps(result,indent=2,sort_keys=True)+'\n')
79
+ print(json.dumps(result,indent=2,sort_keys=True),flush=True)
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/raw/commonsenseqa/README.md ADDED
@@ -0,0 +1,233 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ annotations_creators:
3
+ - crowdsourced
4
+ language_creators:
5
+ - crowdsourced
6
+ language:
7
+ - en
8
+ license:
9
+ - mit
10
+ multilinguality:
11
+ - monolingual
12
+ size_categories:
13
+ - 1K<n<10K
14
+ source_datasets:
15
+ - original
16
+ task_categories:
17
+ - question-answering
18
+ task_ids:
19
+ - open-domain-qa
20
+ paperswithcode_id: commonsenseqa
21
+ pretty_name: CommonsenseQA
22
+ dataset_info:
23
+ features:
24
+ - name: id
25
+ dtype: string
26
+ - name: question
27
+ dtype: string
28
+ - name: question_concept
29
+ dtype: string
30
+ - name: choices
31
+ sequence:
32
+ - name: label
33
+ dtype: string
34
+ - name: text
35
+ dtype: string
36
+ - name: answerKey
37
+ dtype: string
38
+ splits:
39
+ - name: train
40
+ num_bytes: 2207794
41
+ num_examples: 9741
42
+ - name: validation
43
+ num_bytes: 273848
44
+ num_examples: 1221
45
+ - name: test
46
+ num_bytes: 257842
47
+ num_examples: 1140
48
+ download_size: 1558570
49
+ dataset_size: 2739484
50
+ configs:
51
+ - config_name: default
52
+ data_files:
53
+ - split: train
54
+ path: data/train-*
55
+ - split: validation
56
+ path: data/validation-*
57
+ - split: test
58
+ path: data/test-*
59
+ ---
60
+
61
+ # Dataset Card for "commonsense_qa"
62
+
63
+ ## Table of Contents
64
+ - [Table of Contents](#table-of-contents)
65
+ - [Dataset Description](#dataset-description)
66
+ - [Dataset Summary](#dataset-summary)
67
+ - [Supported Tasks and Leaderboards](#supported-tasks-and-leaderboards)
68
+ - [Languages](#languages)
69
+ - [Dataset Structure](#dataset-structure)
70
+ - [Data Instances](#data-instances)
71
+ - [Data Fields](#data-fields)
72
+ - [Data Splits](#data-splits)
73
+ - [Dataset Creation](#dataset-creation)
74
+ - [Curation Rationale](#curation-rationale)
75
+ - [Source Data](#source-data)
76
+ - [Annotations](#annotations)
77
+ - [Personal and Sensitive Information](#personal-and-sensitive-information)
78
+ - [Considerations for Using the Data](#considerations-for-using-the-data)
79
+ - [Social Impact of Dataset](#social-impact-of-dataset)
80
+ - [Discussion of Biases](#discussion-of-biases)
81
+ - [Other Known Limitations](#other-known-limitations)
82
+ - [Additional Information](#additional-information)
83
+ - [Dataset Curators](#dataset-curators)
84
+ - [Licensing Information](#licensing-information)
85
+ - [Citation Information](#citation-information)
86
+ - [Contributions](#contributions)
87
+
88
+ ## Dataset Description
89
+
90
+ - **Homepage:** https://www.tau-nlp.org/commonsenseqa
91
+ - **Repository:** https://github.com/jonathanherzig/commonsenseqa
92
+ - **Paper:** https://arxiv.org/abs/1811.00937
93
+ - **Point of Contact:** [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
94
+ - **Size of downloaded dataset files:** 4.68 MB
95
+ - **Size of the generated dataset:** 2.18 MB
96
+ - **Total amount of disk used:** 6.86 MB
97
+
98
+ ### Dataset Summary
99
+
100
+ CommonsenseQA is a new multiple-choice question answering dataset that requires different types of commonsense knowledge
101
+ to predict the correct answers . It contains 12,102 questions with one correct answer and four distractor answers.
102
+ The dataset is provided in two major training/validation/testing set splits: "Random split" which is the main evaluation
103
+ split, and "Question token split", see paper for details.
104
+
105
+ ### Supported Tasks and Leaderboards
106
+
107
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
108
+
109
+ ### Languages
110
+
111
+ The dataset is in English (`en`).
112
+
113
+ ## Dataset Structure
114
+
115
+ ### Data Instances
116
+
117
+ #### default
118
+
119
+ - **Size of downloaded dataset files:** 4.68 MB
120
+ - **Size of the generated dataset:** 2.18 MB
121
+ - **Total amount of disk used:** 6.86 MB
122
+
123
+ An example of 'train' looks as follows:
124
+ ```
125
+ {'id': '075e483d21c29a511267ef62bedc0461',
126
+ 'question': 'The sanctions against the school were a punishing blow, and they seemed to what the efforts the school had made to change?',
127
+ 'question_concept': 'punishing',
128
+ 'choices': {'label': ['A', 'B', 'C', 'D', 'E'],
129
+ 'text': ['ignore', 'enforce', 'authoritarian', 'yell at', 'avoid']},
130
+ 'answerKey': 'A'}
131
+ ```
132
+
133
+ ### Data Fields
134
+
135
+ The data fields are the same among all splits.
136
+
137
+ #### default
138
+ - `id` (`str`): Unique ID.
139
+ - `question`: a `string` feature.
140
+ - `question_concept` (`str`): ConceptNet concept associated to the question.
141
+ - `choices`: a dictionary feature containing:
142
+ - `label`: a `string` feature.
143
+ - `text`: a `string` feature.
144
+ - `answerKey`: a `string` feature.
145
+
146
+ ### Data Splits
147
+
148
+ | name | train | validation | test |
149
+ |---------|------:|-----------:|-----:|
150
+ | default | 9741 | 1221 | 1140 |
151
+
152
+ ## Dataset Creation
153
+
154
+ ### Curation Rationale
155
+
156
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
157
+
158
+ ### Source Data
159
+
160
+ #### Initial Data Collection and Normalization
161
+
162
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
163
+
164
+ #### Who are the source language producers?
165
+
166
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
167
+
168
+ ### Annotations
169
+
170
+ #### Annotation process
171
+
172
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
173
+
174
+ #### Who are the annotators?
175
+
176
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
177
+
178
+ ### Personal and Sensitive Information
179
+
180
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
181
+
182
+ ## Considerations for Using the Data
183
+
184
+ ### Social Impact of Dataset
185
+
186
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
187
+
188
+ ### Discussion of Biases
189
+
190
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
191
+
192
+ ### Other Known Limitations
193
+
194
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
195
+
196
+ ## Additional Information
197
+
198
+ ### Dataset Curators
199
+
200
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
201
+
202
+ ### Licensing Information
203
+
204
+ The dataset is licensed under the MIT License.
205
+
206
+ See: https://github.com/jonathanherzig/commonsenseqa/issues/5
207
+
208
+ ### Citation Information
209
+
210
+ ```
211
+ @inproceedings{talmor-etal-2019-commonsenseqa,
212
+ title = "{C}ommonsense{QA}: A Question Answering Challenge Targeting Commonsense Knowledge",
213
+ author = "Talmor, Alon and
214
+ Herzig, Jonathan and
215
+ Lourie, Nicholas and
216
+ Berant, Jonathan",
217
+ booktitle = "Proceedings of the 2019 Conference of the North {A}merican Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)",
218
+ month = jun,
219
+ year = "2019",
220
+ address = "Minneapolis, Minnesota",
221
+ publisher = "Association for Computational Linguistics",
222
+ url = "https://aclanthology.org/N19-1421",
223
+ doi = "10.18653/v1/N19-1421",
224
+ pages = "4149--4158",
225
+ archivePrefix = "arXiv",
226
+ eprint = "1811.00937",
227
+ primaryClass = "cs",
228
+ }
229
+ ```
230
+
231
+ ### Contributions
232
+
233
+ Thanks to [@thomwolf](https://github.com/thomwolf), [@lewtun](https://github.com/lewtun), [@albertvillanova](https://github.com/albertvillanova), [@patrickvonplaten](https://github.com/patrickvonplaten) for adding this dataset.
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/raw/hellaswag/README.md ADDED
@@ -0,0 +1,218 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ language:
3
+ - en
4
+ paperswithcode_id: hellaswag
5
+ pretty_name: HellaSwag
6
+ dataset_info:
7
+ features:
8
+ - name: ind
9
+ dtype: int32
10
+ - name: activity_label
11
+ dtype: string
12
+ - name: ctx_a
13
+ dtype: string
14
+ - name: ctx_b
15
+ dtype: string
16
+ - name: ctx
17
+ dtype: string
18
+ - name: endings
19
+ sequence: string
20
+ - name: source_id
21
+ dtype: string
22
+ - name: split
23
+ dtype: string
24
+ - name: split_type
25
+ dtype: string
26
+ - name: label
27
+ dtype: string
28
+ splits:
29
+ - name: train
30
+ num_bytes: 43232624
31
+ num_examples: 39905
32
+ - name: test
33
+ num_bytes: 10791853
34
+ num_examples: 10003
35
+ - name: validation
36
+ num_bytes: 11175717
37
+ num_examples: 10042
38
+ download_size: 36793872
39
+ dataset_size: 65200194
40
+ configs:
41
+ - config_name: default
42
+ data_files:
43
+ - split: train
44
+ path: data/train-*
45
+ - split: test
46
+ path: data/test-*
47
+ - split: validation
48
+ path: data/validation-*
49
+ ---
50
+
51
+ # Dataset Card for "hellaswag"
52
+
53
+ ## Table of Contents
54
+ - [Dataset Description](#dataset-description)
55
+ - [Dataset Summary](#dataset-summary)
56
+ - [Supported Tasks and Leaderboards](#supported-tasks-and-leaderboards)
57
+ - [Languages](#languages)
58
+ - [Dataset Structure](#dataset-structure)
59
+ - [Data Instances](#data-instances)
60
+ - [Data Fields](#data-fields)
61
+ - [Data Splits](#data-splits)
62
+ - [Dataset Creation](#dataset-creation)
63
+ - [Curation Rationale](#curation-rationale)
64
+ - [Source Data](#source-data)
65
+ - [Annotations](#annotations)
66
+ - [Personal and Sensitive Information](#personal-and-sensitive-information)
67
+ - [Considerations for Using the Data](#considerations-for-using-the-data)
68
+ - [Social Impact of Dataset](#social-impact-of-dataset)
69
+ - [Discussion of Biases](#discussion-of-biases)
70
+ - [Other Known Limitations](#other-known-limitations)
71
+ - [Additional Information](#additional-information)
72
+ - [Dataset Curators](#dataset-curators)
73
+ - [Licensing Information](#licensing-information)
74
+ - [Citation Information](#citation-information)
75
+ - [Contributions](#contributions)
76
+
77
+ ## Dataset Description
78
+
79
+ - **Homepage:** [https://rowanzellers.com/hellaswag/](https://rowanzellers.com/hellaswag/)
80
+ - **Repository:** [https://github.com/rowanz/hellaswag/](https://github.com/rowanz/hellaswag/)
81
+ - **Paper:** [HellaSwag: Can a Machine Really Finish Your Sentence?](https://arxiv.org/abs/1905.07830)
82
+ - **Point of Contact:** [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
83
+ - **Size of downloaded dataset files:** 71.49 MB
84
+ - **Size of the generated dataset:** 65.32 MB
85
+ - **Total amount of disk used:** 136.81 MB
86
+
87
+ ### Dataset Summary
88
+
89
+ HellaSwag: Can a Machine Really Finish Your Sentence? is a new dataset for commonsense NLI. A paper was published at ACL2019.
90
+
91
+ ### Supported Tasks and Leaderboards
92
+
93
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
94
+
95
+ ### Languages
96
+
97
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
98
+
99
+ ## Dataset Structure
100
+
101
+ ### Data Instances
102
+
103
+ #### default
104
+
105
+ - **Size of downloaded dataset files:** 71.49 MB
106
+ - **Size of the generated dataset:** 65.32 MB
107
+ - **Total amount of disk used:** 136.81 MB
108
+
109
+ An example of 'train' looks as follows.
110
+ ```
111
+ This example was too long and was cropped:
112
+
113
+ {
114
+ "activity_label": "Removing ice from car",
115
+ "ctx": "Then, the man writes over the snow covering the window of a car, and a woman wearing winter clothes smiles. then",
116
+ "ctx_a": "Then, the man writes over the snow covering the window of a car, and a woman wearing winter clothes smiles.",
117
+ "ctx_b": "then",
118
+ "endings": "[\", the man adds wax to the windshield and cuts it.\", \", a person board a ski lift, while two men supporting the head of the per...",
119
+ "ind": 4,
120
+ "label": "3",
121
+ "source_id": "activitynet~v_-1IBHYS3L-Y",
122
+ "split": "train",
123
+ "split_type": "indomain"
124
+ }
125
+ ```
126
+
127
+ ### Data Fields
128
+
129
+ The data fields are the same among all splits.
130
+
131
+ #### default
132
+ - `ind`: a `int32` feature.
133
+ - `activity_label`: a `string` feature.
134
+ - `ctx_a`: a `string` feature.
135
+ - `ctx_b`: a `string` feature.
136
+ - `ctx`: a `string` feature.
137
+ - `endings`: a `list` of `string` features.
138
+ - `source_id`: a `string` feature.
139
+ - `split`: a `string` feature.
140
+ - `split_type`: a `string` feature.
141
+ - `label`: a `string` feature.
142
+
143
+ ### Data Splits
144
+
145
+ | name |train|validation|test |
146
+ |-------|----:|---------:|----:|
147
+ |default|39905| 10042|10003|
148
+
149
+ ## Dataset Creation
150
+
151
+ ### Curation Rationale
152
+
153
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
154
+
155
+ ### Source Data
156
+
157
+ #### Initial Data Collection and Normalization
158
+
159
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
160
+
161
+ #### Who are the source language producers?
162
+
163
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
164
+
165
+ ### Annotations
166
+
167
+ #### Annotation process
168
+
169
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
170
+
171
+ #### Who are the annotators?
172
+
173
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
174
+
175
+ ### Personal and Sensitive Information
176
+
177
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
178
+
179
+ ## Considerations for Using the Data
180
+
181
+ ### Social Impact of Dataset
182
+
183
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
184
+
185
+ ### Discussion of Biases
186
+
187
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
188
+
189
+ ### Other Known Limitations
190
+
191
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
192
+
193
+ ## Additional Information
194
+
195
+ ### Dataset Curators
196
+
197
+ [More Information Needed](https://github.com/huggingface/datasets/blob/master/CONTRIBUTING.md#how-to-contribute-to-the-dataset-cards)
198
+
199
+ ### Licensing Information
200
+
201
+ MIT https://github.com/rowanz/hellaswag/blob/master/LICENSE
202
+
203
+ ### Citation Information
204
+
205
+ ```
206
+ @inproceedings{zellers2019hellaswag,
207
+ title={HellaSwag: Can a Machine Really Finish Your Sentence?},
208
+ author={Zellers, Rowan and Holtzman, Ari and Bisk, Yonatan and Farhadi, Ali and Choi, Yejin},
209
+ booktitle ={Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics},
210
+ year={2019}
211
+ }
212
+
213
+ ```
214
+
215
+
216
+ ### Contributions
217
+
218
+ Thanks to [@albertvillanova](https://github.com/albertvillanova), [@mariamabarham](https://github.com/mariamabarham), [@thomwolf](https://github.com/thomwolf), [@patrickvonplaten](https://github.com/patrickvonplaten), [@lewtun](https://github.com/lewtun) for adding this dataset.
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/raw/piqa/piqa/README.md ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ Leaderboard: [https://leaderboard.allenai.org/physicaliqa/submissions/public](https://leaderboard.allenai.org/physicaliqa/submissions/public)
2
+
3
+ License: [https://opensource.org/licenses/AFL-3.0](Academic Free License v. 3.0)
4
+
5
+ Data, in the data/
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/test.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/train.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d6d6a57c3aaaf6527647d0a21cc6027a58a6b40e2099eebb9a0689c0212ee372
3
+ size 55474225
snapshots/20260917-expanded-pilot-hf-backup/data/public-decisions-v2-20260917/validation.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
snapshots/20260917-expanded-pilot-hf-backup/fleet/plan.json ADDED
@@ -0,0 +1,67 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "selection_cutoff": "2026-09-17T16:00:00Z",
3
+ "final_deadline": "2026-09-17T18:16:10Z",
4
+ "dataset": "/home/andy/ai/opensysone/data/public-decisions-v1-20260916",
5
+ "reference_predictions": "/home/andy/ai/opensysone/runs/20260916T183823Z-train/artifacts/validation_step_000040_predictions.json",
6
+ "reference_sha256": "e671e1508185765552b0f933ba03f356be62143c531d8ef534457d34b1645c9b",
7
+ "port": 18081,
8
+ "inference_max_tokens": 1024,
9
+ "candidates": [
10
+ {
11
+ "project": "/home/andy/projects/opensysone",
12
+ "python": "/home/andy/ai/envs/opensysone/bin/python",
13
+ "name": "gx10-4b",
14
+ "host": "local",
15
+ "campaign": "/home/andy/ai/opensysone/runs/20260916T193741Z-24h",
16
+ "training": "/home/andy/ai/opensysone/runs/20260916T193741Z-24h/training",
17
+ "evidence_dirs": [
18
+ "/home/andy/ai/opensysone/runs/20260916T193721Z-selection-parent"
19
+ ]
20
+ },
21
+ {
22
+ "project": "/home/andy/projects/opensysone",
23
+ "python": "/home/andy/ai/envs/opensysone/bin/python",
24
+ "name": "spark-a-4b-low-lr",
25
+ "host": "andy@192.168.8.111",
26
+ "campaign": "/home/andy/ai/opensysone/runs/20260916T194258Z-24h",
27
+ "training": "/home/andy/ai/opensysone/runs/20260916T194258Z-24h/training",
28
+ "evidence_dirs": [
29
+ "/home/andy/ai/opensysone/runs/20260916T192910Z-train/artifacts"
30
+ ]
31
+ },
32
+ {
33
+ "project": "/home/andy/projects/opensysone",
34
+ "python": "/home/andy/ai/envs/opensysone/bin/python",
35
+ "name": "spark-b-2b",
36
+ "host": "andy@192.168.8.204",
37
+ "campaign": "/home/andy/ai/opensysone/runs/20260916T193803Z-24h",
38
+ "training": "/home/andy/ai/opensysone/runs/20260916T193803Z-24h/training",
39
+ "evidence_dirs": [
40
+ "/home/andy/ai/opensysone/runs/20260916T182352Z-train/artifacts"
41
+ ]
42
+ },
43
+ {
44
+ "project": "/home/andy/projects/opensysone",
45
+ "python": "/home/andy/ai/envs/opensysone/bin/python",
46
+ "name": "spark-b-4b-refinement",
47
+ "host": "andy@192.168.8.204",
48
+ "campaign": "/home/andy/ai/opensysone/runs/20260917T023137Z-24h",
49
+ "training": "/home/andy/ai/opensysone/runs/20260917T023137Z-24h/training",
50
+ "evidence_dirs": [
51
+ "/home/andy/ai/opensysone/runs/20260917T021605Z-train/artifacts"
52
+ ]
53
+ },
54
+ {
55
+ "project": "/home/andy/ai/opensysone/source/expanded-24b8ccf",
56
+ "python": "/home/andy/ai/envs/opensysone/bin/python",
57
+ "name": "gx10-4b-expanded",
58
+ "host": "local",
59
+ "campaign": "/home/andy/ai/opensysone/runs/20260917T072142Z-24h",
60
+ "training": "/home/andy/ai/opensysone/runs/20260917T072142Z-24h/training",
61
+ "dataset": "/home/andy/ai/opensysone/data/public-decisions-v2-20260917",
62
+ "evidence_dirs": [
63
+ "/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts"
64
+ ]
65
+ }
66
+ ]
67
+ }
snapshots/20260917-expanded-pilot-hf-backup/fleet/reference-validation-predictions.json ADDED
The diff for this file is too large to render. See raw diff
 
source/EXPANDED_DATA.md ADDED
@@ -0,0 +1,147 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Training-data expansion, 2026-09-17
2
+
3
+ The user requested a broader training set. Version 2 adds official training data
4
+ from HellaSwag, PIQA and CommonsenseQA to the complete version-1 replay set.
5
+ These sources add plausible continuations, physical problem solving and everyday
6
+ commonsense questions with two to five answer choices. They train the existing
7
+ decision scorer, not a text-generation model.
8
+
9
+ Data directory: `/home/andy/ai/opensysone/data/public-decisions-v2-20260917`.
10
+ Build with `scripts/prepare_expanded_data.py`; raw downloads, transformed rows,
11
+ weights and token caches stay outside Git. Source revisions, licenses, file hashes,
12
+ sampling decisions and exclusions are recorded in the dataset manifest and
13
+ `results/20260917-expanded-data/dataset-manifest.json`.
14
+
15
+ | Source | Training decisions after 512-token filtering | License |
16
+ | --- | ---: | --- |
17
+ | HellaSwag | 16,000 | MIT |
18
+ | PIQA | 14,360 | AFL-3.0, creator's README |
19
+ | CommonsenseQA | 9,490 | MIT |
20
+ | Original replay | 40,915 | Original per-source provenance |
21
+ | Total | 80,765 | Mixed; no blanket relicensing |
22
+
23
+ Only labeled official training rows are added. Exclusions use normalized source
24
+ groups and exact normalized context text, including official evaluation metadata.
25
+ Duplicate choices and invalid labels are rejected. Original training bytes form
26
+ an unchanged prefix; the existing deterministic per-epoch shuffle samples the
27
+ combined rows uniformly. Model-specific token filtering rejects oversize examples
28
+ without truncation. Exact matching does not exclude semantic duplicates or
29
+ contamination already present in the pretrained model.
30
+
31
+ CPU tokenization completed in 60.96 seconds without loading model weights or
32
+ initializing CUDA. Of 80,792 raw rows, 27 exceed the limit (26 original BoolQ and
33
+ one new PIQA); the longest retained branch is 509 tokens. Original tokenized
34
+ training replay and all four protected tokenized splits compare exactly with the
35
+ verified version-1 cache. Data signature:
36
+ `76183c642668602f42b7f3e71a3fe03bd5bd76f064fce4ba92351d8703396207`.
37
+
38
+ The original validation, calibration, test and Social IQA holdout files remain
39
+ byte-for-byte identical. Another 128 source groups per new family are withheld
40
+ as separate diagnostics, excluded from training and from checkpoint selection.
41
+ After length filtering these contain 383 decisions (one PIQA item is too long).
42
+ The existing fixed 512-decision selection therefore measures retention on the
43
+ original four tasks; gains on the new tasks require separate diagnostic evidence.
44
+ No new-source accuracy or probability-calibration claim is made from expansion.
45
+
46
+ The new candidate will warm-start from Spark B's frozen selected step 1,500:
47
+ `/home/andy/ai/opensysone/runs/20260917T070415Z-expanded-parent/best.pt`.
48
+ Its 512-decision validation accuracy is 94.7266%, crossfit NLL 0.170149713.
49
+ Parent SHA256: `5f57ec38796d132edfa23638fbce66131fd4e7dfe87ceeadaba2b0e0d5c78024`.
50
+ The original artifact and its data signature are preserved. An explicit
51
+ `--warm-start --allow-train-data-change` operation verifies both real datasets,
52
+ parent manifest lineage, pinned model, prompt, adapter layout and token policy,
53
+ then restores only trained weights with fresh Adam and RNG. Exact resume remains
54
+ strict about the full data signature. Fleet dataset overrides require unchanged
55
+ reserved bytes and finalize against the selected candidate's verified dataset.
56
+
57
+ GX10 is the replacement target: its selected step 2,500 survived three later
58
+ non-improving validations by 07:00 UTC. Both improving Spark runs continue.
59
+ Preserve GX10's stopped run and its fleet candidacy. The planned expanded pilot
60
+ uses Qwen3-4B-Instruct-2507, FP32 rank 8 / alpha 16, branch batch 1, exact two-pass
61
+ gradients, effective batch 4, seed 433, learning rates 2e-5, schedule 3,500 steps.
62
+ The original 16 GiB allocation cap and absolute 16:00 UTC training / 18:16:10 UTC
63
+ final deadlines remain in force. Startup, progress and process controls are
64
+ recorded below after the real pilot and resume checks complete.
65
+
66
+ ## Verification and execution
67
+
68
+ Source revision `24b8ccf60d388f9cbb184e03a6ae260a1f5a8b86` is frozen in the clean
69
+ detached worktree `/home/andy/ai/opensysone/source/expanded-24b8ccf`. Training uses
70
+ the existing isolated Python at `/home/andy/ai/envs/opensysone/bin/python`.
71
+ All 86 source tests passed. The independent postbuild audit reconstructed every
72
+ added row against its pinned raw source, including the correct answer after
73
+ permutation; all 13 downloaded source artifacts matched their hashes. The final
74
+ training mix is 50.66% original replay and 49.34% additions.
75
+
76
+ GX10's original campaign `20260916T193741Z-24h` stopped gracefully at step 4,380,
77
+ with trainer and supervisor exit 0. The resumable checkpoint retains all 506 Adam
78
+ states, and its selected step 2,500 is unchanged. Its smoke lock was released and
79
+ the old trainer disappeared from the GPU process list before loading the pilot.
80
+ The playground on port 7466 remains running.
81
+
82
+ Expanded pilot: `/home/andy/ai/opensysone/runs/20260917T070758Z-train/artifacts`.
83
+ It started at 07:07:59 UTC with approximately 100 GiB available memory and OOM
84
+ adjustment 0. A frozen copy of the actual step-0 checkpoint verifies all 506
85
+ trainable tensors exactly equal the parent and the Adam state is empty; the new
86
+ Python RNG matches seed 433. No parent weights or data signature were rewritten.
87
+ Initial FP32 correctness passed, worst probability difference 2.65e-7 against the
88
+ 1e-4 tolerance. The first 32 sampled training decisions cover all seven families.
89
+
90
+ Small evidence is in `results/20260917-expanded-data/`; runtime verification
91
+ helpers and captured initialization states are outside Git under
92
+ `/home/andy/ai/opensysone/fleet-20260917/expanded-startup-verification/`.
93
+ The reusable helpers and their tests are also archived as
94
+ `scripts/verify_expanded_startup.py` and `scripts/prepare_expansion_backup.py`.
95
+
96
+ The pilot completed all eight updates and both correctness gates, **exit 0**.
97
+ All 512 initial logits/probabilities/log-probabilities exactly match the frozen
98
+ parent. Every Adam counter is eight; every loss, gradient and saved tensor is
99
+ finite. Median update time is 8.90 seconds. Peak allocation including final
100
+ checks is 15.426 GiB, under the 16 GiB cap; final correctness worst difference is
101
+ 3.58e-7. Step-8 validation crossfit NLL is 0.170108 and accuracy 94.7266%; its
102
+ improvement is below the fixed 0.001 threshold, so branch step 0 remains selected
103
+ at 0.170150. Eight updates establish startup, not new-task generalization.
104
+
105
+ ## Active campaign and controls
106
+
107
+ Expanded campaign: `/home/andy/ai/opensysone/runs/20260917T072142Z-24h`, launched
108
+ 07:21:42 UTC. Supervisor **1630617**, trainer **1630638**, source **24b8ccf**.
109
+ It resumes the complete pilot step-8 weights, Adam and RNG. A frozen copy of the
110
+ new campaign's initial checkpoint matches every trainable tensor, every Adam
111
+ state and Python/Torch/CUDA RNG exactly; all launch processes have OOM adjustment
112
+ 0. Final training exit is pending. Check live state rather than relying on PIDs.
113
+
114
+ ```bash
115
+ python3 scripts/fleet_status.py --json
116
+ tail -n 5 ~/ai/opensysone/runs/20260917T072142Z-24h/training.log
117
+ ```
118
+
119
+ To stop only this training candidate gracefully:
120
+
121
+ ```bash
122
+ ~/ai/envs/opensysone/bin/python \
123
+ ~/ai/opensysone/source/expanded-24b8ccf/scripts/campaign_status.py \
124
+ --campaign ~/ai/opensysone/runs/20260917T072142Z-24h --stop
125
+ ```
126
+
127
+ To recover after a verified stop, inspect the last saved step and free memory,
128
+ then create a fresh campaign from its training directory with the same deadlines:
129
+
130
+ ```bash
131
+ ~/ai/envs/opensysone/bin/python \
132
+ ~/ai/opensysone/source/expanded-24b8ccf/scripts/launch_24h.py \
133
+ --pilot ~/ai/opensysone/runs/20260917T072142Z-24h/training \
134
+ --train-only --training-deadline 2026-09-17T16:00:00Z \
135
+ --deadline 2026-09-17T18:16:10Z --inference-max-tokens 1024 \
136
+ --selection-metric crossfit_temperature_nll_v1
137
+ ```
138
+
139
+ Before restarting a recovered fleet candidate, stop the waiting coordinator,
140
+ update that candidate's campaign/training/evidence paths in its backed-up plan,
141
+ retain its explicit dataset override and frozen source project path, and resume
142
+ the coordinator. Preserve the original stopped candidate and previous artifacts.
143
+ The coordinator is `/home/andy/ai/opensysone/runs/20260916T194403396250Z-fleet`,
144
+ resumed as PID **1630841** after registering `gx10-4b-expanded` as candidate five.
145
+ Use `scripts/fleet_campaign.py --campaign PATH --stop` and then `--resume` from
146
+ the main checkout. It will finalize using the winner's verified dataset. Its
147
+ absolute cutoffs remain unchanged; never reset them on recovery.
source/FLEET_RUN.md CHANGED
@@ -1,5 +1,16 @@
1
  # Three-machine campaign
2
 
 
 
 
 
 
 
 
 
 
 
 
3
  **2026-09-17 continuation:** [NEXT_STEPS.md](NEXT_STEPS.md) records overnight
4
  findings and the two-Spark assessment. The original 2B campaign finished with exit
5
  0 at step 6,000; its best is step 2,000. Spark B is now running a 4B refinement
 
1
  # Three-machine campaign
2
 
3
+ **Latest, 2026-09-17 07:22 UTC:** GX10 now runs expanded-data 4B candidate
4
+ `20260917T072142Z-24h`, from frozen source worktree
5
+ `/home/andy/ai/opensysone/source/expanded-24b8ccf`. Supervisor/trainer PIDs at
6
+ launch are 1630617 / 1630638, OOM adjustment 0. The original GX10 campaign stopped
7
+ cleanly at step 4,380, preserving best 2,500. Both Spark 4B runs continue, and the
8
+ 2B remains completed. The existing fleet now has **five candidates**, including
9
+ the explicit v2 dataset override; coordinator PID 1630841 was restarted after
10
+ the plan update. See [EXPANDED_DATA.md](EXPANDED_DATA.md) for verified evidence,
11
+ dataset provenance and exact stop/resume commands. Historical startup rows below
12
+ describe the original fleet and are superseded by this update.
13
+
14
  **2026-09-17 continuation:** [NEXT_STEPS.md](NEXT_STEPS.md) records overnight
15
  findings and the two-Spark assessment. The original 2B campaign finished with exit
16
  0 at step 6,000; its best is step 2,000. Spark B is now running a 4B refinement
source/HANDOVER.md CHANGED
@@ -1,10 +1,13 @@
1
  # Continue the three-machine campaign
2
 
3
- **Continuation, 2026-09-17:** overnight results and the next-step assessment are
4
- in [NEXT_STEPS.md](NEXT_STEPS.md). GX10/A are still improving. Spark B's 2B run
5
- completed at 6,000 steps (exit 0), preserving best step 2,000; its free GPU is
6
- running a verified 4B refinement campaign from GX10 best step 2,500. All four
7
- candidates are registered in the fleet plan, including the completed 2B.
 
 
 
8
 
9
  The user assigned **GX10 and both Sparks** to this task, authorized stopping their
10
  workloads, and requested continued experimentation without permission prompts.
@@ -29,21 +32,27 @@ on each host. Read the relevant `docs/host.md`, `docs/training.md`, `docs/spark-
29
  ## Active runs and source
30
 
31
  All run IDs below are relative to `/home/andy/ai/opensysone/runs` **on that host**.
32
- The active trainers launched from clean source **`4a60423`**. GX10
33
- subsequently added the coordinator shutdown fix and this handover; the Spark
34
- checkouts remain at `4a60423`. Running trainers retain their execution revision
 
 
35
  and source hashes in their manifests. Inspect live state before
36
  using recorded PIDs. Exit statuses of active jobs remain pending.
37
 
38
  | Host | Trial | Campaign | Supervisor / trainer at launch |
39
  | --- | --- | --- | --- |
40
- | GX10 | 4B, LR 0.0001, exact resume at step 178 | `20260916T193741Z-24h` | 1096483 / 1096505 |
 
41
  | spark-a | 4B, LR 0.00003, fresh optimizer then pilot resume | `20260916T194258Z-24h` | 327084 / 327116 |
42
  | spark-b | 2B completed at step 6,000; selected step 2,000 | `20260916T193803Z-24h` | exited 0 / 0 |
43
  | spark-b | 4B refinement, LR 0.00001, seed 432 | `20260917T023137Z-24h` | 483974 / 484001 |
44
 
45
- Fleet coordinator: **`20260916T194403396250Z-fleet` on GX10**, PID **1425784**,
46
- source **`fed722e`**, running in `waiting_for_selection` with OOM adjustment 0.
 
 
 
47
  It selects the best durable candidate, then runs finalization and serves it on
48
  GX10. The individual campaigns are `train_only=true`;
49
  they cannot independently evaluate reserved data or publish competing deployments.
 
1
  # Continue the three-machine campaign
2
 
3
+ **Latest continuation, 2026-09-17 07:22 UTC:** the user requested broader training
4
+ data. [EXPANDED_DATA.md](EXPANDED_DATA.md) records the 80,765-example seven-family
5
+ mix, exact protected-split preservation, completed eight-step pilot and new GX10
6
+ campaign `20260917T072142Z-24h`. It warm-starts from Spark B's selected step 1,500
7
+ with fresh Adam, then resumes the verified pilot. GX10's original run stopped
8
+ cleanly at step 4,380, retaining best step 2,500. Both Spark 4B runs continue;
9
+ the completed 2B remains available. All **five candidates** are registered.
10
+ The earlier overnight assessment is in [NEXT_STEPS.md](NEXT_STEPS.md).
11
 
12
  The user assigned **GX10 and both Sparks** to this task, authorized stopping their
13
  workloads, and requested continued experimentation without permission prompts.
 
32
  ## Active runs and source
33
 
34
  All run IDs below are relative to `/home/andy/ai/opensysone/runs` **on that host**.
35
+ The Spark trainers launched from clean source **`4a60423`**. The expanded GX10
36
+ trainer uses clean source **`24b8ccf`** in the detached worktree
37
+ `/home/andy/ai/opensysone/source/expanded-24b8ccf`; keep that worktree for its
38
+ supervisor and recovery. The main checkout contains current documentation and
39
+ backup/verification tools. Running trainers retain their execution revision
40
  and source hashes in their manifests. Inspect live state before
41
  using recorded PIDs. Exit statuses of active jobs remain pending.
42
 
43
  | Host | Trial | Campaign | Supervisor / trainer at launch |
44
  | --- | --- | --- | --- |
45
+ | GX10 | Original 4B, stopped at 4,380; selected 2,500 | `20260916T193741Z-24h` | exited 0 / 0 |
46
+ | GX10 | Expanded 4B, LR 0.00002, seed 433, resumed pilot step 8 | `20260917T072142Z-24h` | 1630617 / 1630638 |
47
  | spark-a | 4B, LR 0.00003, fresh optimizer then pilot resume | `20260916T194258Z-24h` | 327084 / 327116 |
48
  | spark-b | 2B completed at step 6,000; selected step 2,000 | `20260916T193803Z-24h` | exited 0 / 0 |
49
  | spark-b | 4B refinement, LR 0.00001, seed 432 | `20260917T023137Z-24h` | 483974 / 484001 |
50
 
51
+ Fleet coordinator: **`20260916T194403396250Z-fleet` on GX10**, PID **1630841**,
52
+ source **`24b8ccf`**, running in `waiting_for_selection` with OOM adjustment 0.
53
+ It was stopped before the fifth candidate and its explicit dataset override were
54
+ registered, then restarted. The old stop's exit 1 can remain in `exit_code` while
55
+ the new coordinator runs; current process identity/state determines liveness.
56
  It selects the best durable candidate, then runs finalization and serves it on
57
  GX10. The individual campaigns are `train_only=true`;
58
  they cannot independently evaluate reserved data or publish competing deployments.
source/HF_MODEL_CARD.md CHANGED
@@ -32,21 +32,25 @@ calibrated deployment have not completed. The original campaign ends on
32
  `FINAL_MODEL.json` will identify the separately verified calibrated release when
33
  finalization and upload succeed; its absence means no final release is recorded.
34
 
35
- Validation audit on 17 September, 02:10–02:15 UTC:
36
 
37
  | Selected candidate | Step | Validation accuracy | Crossfit selection NLL |
38
  | --- | ---: | ---: | ---: |
39
  | Qwen3 4B, main run | 2,500 | 93.55% | 0.188640 |
40
- | Qwen3 4B, lower learning rate | 2,500 | 92.58% | 0.218012 |
41
  | Qwen3.5 2B | 2,000 | 89.84% | 0.255294 |
 
42
 
43
  All use the same 512 validation decisions. Selection uses four source-group-
44
  disjoint temperature-crossfit folds with fixed seed 431; smaller macro-family NLL
45
  is better. This is **validation-selection evidence**, not a held-out quality claim.
46
  The 2B run stopped cleanly at step 6,000 after eight evaluations without a new
47
- best. Both 4B runs continued, and another lower-rate 4B refinement was started
48
- from the main run's selected weights. See `source/NEXT_STEPS.md` for findings and
49
- why independent trials were retained over unmeasured distributed training.
 
 
 
50
 
51
  ## Contents and reconstruction
52
 
@@ -64,6 +68,8 @@ hosted inference application.
64
  and reconstruction metadata. These initial backups have no fitted temperature.
65
  - `snapshots/<id>/artifacts/<candidate>/checkpoint.pt`: resumable training state,
66
  including Adam and random states. Its step may be later than the selected best.
 
 
67
  - `final/<campaign>/`: final calibrated model and evaluation evidence, created only
68
  after successful finalization and publication.
69
 
@@ -91,7 +97,7 @@ Pinned bases, recorded as Apache-2.0 in their provenance:
91
 
92
  ## Data, evaluation and limitations
93
 
94
- Public training sources are SNLI, BoolQ, ARC and four-choice Banking77 routing.
95
  Social IQA is a wholly untrained task-family holdout. Frozen source pins, licences,
96
  raw/split hashes and group/deduplication audits are included in
97
  `source/results/public-decisions-v1-manifest.json`. That manifest credits
@@ -100,6 +106,17 @@ and records their original dataset licences. The original repository's
100
  `license: unknown` metadata is retained; no new licence for the project artifacts
101
  is assigned by this backup.
102
 
 
 
 
 
 
 
 
 
 
 
 
103
  Validation selects checkpoints. Separate calibration fits one global temperature
104
  only after selection is frozen. Final test/holdout comparisons use the unchanged
105
  pretrained scorer, with a separately fitted base temperature and source-group
@@ -112,5 +129,7 @@ The harness implements local scoring, a loopback API, hosted Jev requests and
112
  response/timing comparisons. The local model identifies itself as OpenSysOne;
113
  compatibility with the request shape does not make it Jev. Local confidence is
114
  normalized entropy, not calibrated probability of correctness. Authenticated
115
- hosted Jev calls require `TYPESAFE_API_KEY` and have not been tested. Credentials,
116
- pretrained base weights and raw training datasets are not part of this backup.
 
 
 
32
  `FINAL_MODEL.json` will identify the separately verified calibrated release when
33
  finalization and upload succeed; its absence means no final release is recorded.
34
 
35
+ Validation audit on 17 September, 07:00 UTC:
36
 
37
  | Selected candidate | Step | Validation accuracy | Crossfit selection NLL |
38
  | --- | ---: | ---: | ---: |
39
  | Qwen3 4B, main run | 2,500 | 93.55% | 0.188640 |
40
+ | Qwen3 4B, lower learning rate | 4,000 | 93.75% | 0.190020 |
41
  | Qwen3.5 2B | 2,000 | 89.84% | 0.255294 |
42
+ | Qwen3 4B, refinement | 1,500 | 94.73% | 0.170150 |
43
 
44
  All use the same 512 validation decisions. Selection uses four source-group-
45
  disjoint temperature-crossfit folds with fixed seed 431; smaller macro-family NLL
46
  is better. This is **validation-selection evidence**, not a held-out quality claim.
47
  The 2B run stopped cleanly at step 6,000 after eight evaluations without a new
48
+ best. At 07:07 UTC the original GX10 4B run stopped cleanly at step 4,380,
49
+ preserving its selected step 2,500 and full resumable state. GX10 now verifies an
50
+ expanded-data candidate from the refinement's frozen selected step 1,500, while
51
+ both Spark 4B runs continue. See `source/EXPANDED_DATA.md` for its latest run
52
+ state, verification evidence and the broader training mix. Independent trials
53
+ retain the original absolute deadlines.
54
 
55
  ## Contents and reconstruction
56
 
 
68
  and reconstruction metadata. These initial backups have no fitted temperature.
69
  - `snapshots/<id>/artifacts/<candidate>/checkpoint.pt`: resumable training state,
70
  including Adam and random states. Its step may be later than the selected best.
71
+ - Expanded-data snapshots include transformed decision files and their source
72
+ notices, split hashes, diagnostic exclusions and original dataset lineage.
73
  - `final/<campaign>/`: final calibrated model and evaluation evidence, created only
74
  after successful finalization and publication.
75
 
 
97
 
98
  ## Data, evaluation and limitations
99
 
100
+ The original public training sources are SNLI, BoolQ, ARC and four-choice Banking77 routing.
101
  Social IQA is a wholly untrained task-family holdout. Frozen source pins, licences,
102
  raw/split hashes and group/deduplication audits are included in
103
  `source/results/public-decisions-v1-manifest.json`. That manifest credits
 
106
  `license: unknown` metadata is retained; no new licence for the project artifacts
107
  is assigned by this backup.
108
 
109
+ The expanded candidate adds official training rows from HellaSwag (MIT), PIQA
110
+ (AFL-3.0 according to its creator's pinned README) and CommonsenseQA (MIT).
111
+ After the 512-token filter there are 80,765 training decisions, including all
112
+ 40,915 original retained examples. All four original reserved split files and
113
+ their tokenized rows are unchanged. Another 383 retained new-source diagnostic
114
+ decisions are excluded from both training and checkpoint selection. Source pins,
115
+ credits, licences and transformations are in
116
+ `source/results/20260917-expanded-data/dataset-manifest.json`. The unchanged
117
+ selection set measures the original tasks; expansion alone is not evidence of
118
+ better performance on the three added tasks.
119
+
120
  Validation selects checkpoints. Separate calibration fits one global temperature
121
  only after selection is frozen. Final test/holdout comparisons use the unchanged
122
  pretrained scorer, with a separately fitted base temperature and source-group
 
129
  response/timing comparisons. The local model identifies itself as OpenSysOne;
130
  compatibility with the request shape does not make it Jev. Local confidence is
131
  normalized entropy, not calibrated probability of correctness. Authenticated
132
+ hosted Jev calls require `TYPESAFE_API_KEY` and have not been tested. Credentials
133
+ and pretrained base weights are not part of this backup. Expanded-data snapshots
134
+ may include the transformed training data, with upstream notices retained;
135
+ raw upstream archives are referenced by pinned revision and checksum.
source/PLAN.md CHANGED
@@ -1,5 +1,23 @@
1
  # OpenSysOne: single-node start and GX10 handover
2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3
  ## Continuation decision — 2026-09-17
4
 
5
  See [NEXT_STEPS.md](NEXT_STEPS.md) for the overnight findings and the assessment
 
1
  # OpenSysOne: single-node start and GX10 handover
2
 
3
+ ## Training-data expansion — 2026-09-17 07:22 UTC
4
+
5
+ The user requested expansion using best judgment. Add pinned official TRAIN
6
+ rows from HellaSwag, PIQA and CommonsenseQA while retaining every original
7
+ training example and preserving all four reserved splits byte-for-byte. The
8
+ filtered mix has 80,765 decisions, approximately half replay and half additions.
9
+ Keep 383 additional new-source diagnostics outside training and the fixed
10
+ selection protocol. See [EXPANDED_DATA.md](EXPANDED_DATA.md) for provenance,
11
+ verification, current controls and the limits of the unchanged selection set.
12
+
13
+ Replace the plateaued GX10 run, preserving its resumable step 4,380 and selected
14
+ step 2,500. Initialize a new 4B candidate from frozen Spark B step 1,500 with fresh
15
+ Adam, seed 433 and LR 2e-5. The eight-step pilot passed; campaign
16
+ `20260917T072142Z-24h` resumes it from clean frozen source `24b8ccf`. Both Spark
17
+ 4B runs continue. The fifth fleet candidate explicitly registers v2 data;
18
+ protected bytes and identical validation identities remain eligibility gates.
19
+ Keep the original 16 GiB cap and absolute 16:00 / 18:16:10 UTC deadlines.
20
+
21
  ## Continuation decision — 2026-09-17
22
 
23
  See [NEXT_STEPS.md](NEXT_STEPS.md) for the overnight findings and the assessment
source/RESULTS.md CHANGED
@@ -499,3 +499,34 @@ prefix caching and the larger latency matrix remain open. The Sparks now host
499
  independent candidate experiments; GX10 does not need a ConnectX cable for this
500
  selection strategy. Architecture B and
501
  distributed training still await quality and profiling evidence in [PLAN.md](PLAN.md).
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
499
  independent candidate experiments; GX10 does not need a ConnectX cable for this
500
  selection strategy. Architecture B and
501
  distributed training still await quality and profiling evidence in [PLAN.md](PLAN.md).
502
+
503
+ ## Expanded public training data — 2026-09-17
504
+
505
+ Version 2 retains all 40,915 original 4B-compatible training decisions and adds
506
+ 16,000 HellaSwag, 14,360 PIQA and 9,490 CommonsenseQA decisions: **80,765 total**.
507
+ All four reserved source files and tokenized sequences match version 1 exactly.
508
+ The 383 retained new-source diagnostics stay outside training and checkpoint
509
+ selection. An independent reconstruction audit checked every added source label
510
+ and shuffled answer position, all downloaded hashes and diagnostic exclusions.
511
+ See [EXPANDED_DATA.md](EXPANDED_DATA.md) and its linked small evidence.
512
+
513
+ Clean source `24b8ccf`, pilot `20260917T070758Z-train`: eight finite updates,
514
+ **exit 0**, all initial/final FP32 gates passed. Frozen Spark B parent step 1,500
515
+ reproduces every initial validation logit and probability exactly. Captured step
516
+ 0 has empty Adam; every final Adam counter is eight. Median update 8.90 seconds,
517
+ peak allocation including checks 15.426 GiB, worst final probability discrepancy
518
+ 3.58e-7. The 32 sampled decisions cover all seven task families.
519
+
520
+ Step 8 scores 0.170108 validation crossfit NLL versus parent 0.170150, both
521
+ 94.7266% accuracy. The difference is below the fixed 0.001 selection threshold;
522
+ the selected branch remains step 0. This is startup evidence, not a claim of
523
+ improvement on the added tasks. The new campaign `20260917T072142Z-24h` restores
524
+ all step-8 model/Adam/Python/Torch/CUDA states exactly and retains the original
525
+ 16:00 / 18:16:10 UTC deadlines. GX10's former run stopped at step 4,380 with both
526
+ trainer and supervisor exit 0, preserving its selected step 2,500. The fleet
527
+ retains all previous candidates and explicitly registers the new dataset.
528
+
529
+ All 86 source tests passed, including rejection of reserved-data changes and
530
+ unregistered candidate datasets. Expanded checkpoints and transformed data are
531
+ staged for the existing private Hugging Face backup, with exact source revisions,
532
+ upstream notices and checksums; publication receipts are recorded separately.
source/data_transition.py ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Verify an explicit training-data expansion without changing evaluation data."""
2
+ import hashlib
3
+ import json
4
+ from pathlib import Path
5
+
6
+
7
+ SPLITS = ("train", "validation", "calibration", "test", "holdout")
8
+ PROTECTED_SPLITS = SPLITS[1:]
9
+
10
+
11
+ def sha256(path):
12
+ return hashlib.sha256(Path(path).read_bytes()).hexdigest()
13
+
14
+
15
+ def verified_manifest(dataset):
16
+ root = Path(dataset).resolve()
17
+ manifest = json.loads((root / "manifest.json").read_text())
18
+ hashes = manifest.get("split_sha256", {})
19
+ if set(hashes) != set(SPLITS):
20
+ raise ValueError("Data transition requires exactly the five registered splits")
21
+ for split, expected in hashes.items():
22
+ path = root / f"{split}.jsonl"
23
+ if not path.is_file() or sha256(path) != expected:
24
+ raise ValueError(f"Data transition checksum mismatch: {root}/{split}")
25
+ return manifest
26
+
27
+
28
+ def data_signature(manifest, provenance, max_tokens):
29
+ return hashlib.sha256(json.dumps({
30
+ "data": manifest["split_sha256"], "model": provenance,
31
+ "implementation": sha256(Path(__file__).with_name("training_model.py")),
32
+ "max_tokens": max_tokens,
33
+ }, sort_keys=True).encode()).hexdigest()
34
+
35
+
36
+ def verify_train_data_transition(saved, config, signature):
37
+ """Check both real datasets and signatures; return immutable lineage evidence."""
38
+ parent_config = saved["config"]
39
+ if parent_config.get("max_tokens") != config["max_tokens"]:
40
+ raise ValueError("Training-data expansion must preserve max_tokens")
41
+ parent_root = Path(parent_config["dataset"]).resolve()
42
+ current_root = Path(config["dataset"]).resolve()
43
+ parent = verified_manifest(parent_root)
44
+ current = verified_manifest(current_root)
45
+ provenance = saved["model_provenance"]
46
+ if data_signature(parent, provenance, config["max_tokens"]) != saved["data_signature"]:
47
+ raise ValueError("Parent data signature does not match its verified source")
48
+ if data_signature(current, provenance, config["max_tokens"]) != signature:
49
+ raise ValueError("Expanded data signature does not match its verified source")
50
+ protected = {split: parent["split_sha256"][split] for split in PROTECTED_SPLITS}
51
+ if any(current["split_sha256"][split] != digest for split, digest in protected.items()):
52
+ raise ValueError("Training-data expansion must preserve every reserved evaluation split")
53
+ base = current.get("base_dataset", {})
54
+ if (Path(base.get("path", "")).resolve() != parent_root or
55
+ base.get("manifest_sha256") != sha256(parent_root / "manifest.json") or
56
+ base.get("split_sha256") != parent["split_sha256"]):
57
+ raise ValueError("Expanded data must record the verified parent dataset lineage")
58
+ if current["split_sha256"]["train"] == parent["split_sha256"]["train"]:
59
+ raise ValueError("Training-data expansion did not change the training split")
60
+ return {"kind": "training_split_only", "parent_dataset": str(parent_root),
61
+ "dataset": str(current_root), "parent_data_signature": saved["data_signature"],
62
+ "data_signature": signature, "parent_manifest_sha256": sha256(parent_root / "manifest.json"),
63
+ "manifest_sha256": sha256(current_root / "manifest.json"),
64
+ "protected_split_sha256": protected}