YNSScarSaiyan commited on
Commit
1ce59a8
·
0 Parent(s):

Super-squash branch 'main' using huggingface_hub

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +39 -0
  2. PROGRAM_CHARTER.md +246 -0
  3. README.md +829 -0
  4. documentation/20261009-evidence.json +173 -0
  5. lineages/corrected-20260930-v2/LATEST.json +1 -0
  6. lineages/corrected-20260930-v2/curriculum.json +31 -0
  7. lineages/corrected-20260930-v2/lineage.json +11 -0
  8. lineages/corrected-20260930-v2/mix_manifest.json +187 -0
  9. lineages/corrected-20260930-v2/recovery/backup-manifest.json +119 -0
  10. lineages/corrected-20260930-v2/recovery/data/corrected-mix/mix_manifest.json +187 -0
  11. lineages/corrected-20260930-v2/recovery/data/language_baseline/recovery_manifest.json +22 -0
  12. lineages/corrected-20260930-v2/recovery/data/language_baseline/vocab_sparse.json +3 -0
  13. lineages/corrected-20260930-v2/recovery/data/slots/compiler_physics.nysa +0 -0
  14. lineages/corrected-20260930-v2/recovery/data/slots/recovery_manifest.json +108 -0
  15. lineages/corrected-20260930-v2/recovery/data/slots/speech_align.nysv +3 -0
  16. lineages/corrected-20260930-v2/recovery/data/slots/speech_slot.nysv +0 -0
  17. lineages/corrected-20260930-v2/recovery/data/slots/speech_vocoder.nyvc +0 -0
  18. lineages/corrected-20260930-v2/recovery/gpu_eqprop/build_corrected_mix.py +217 -0
  19. lineages/corrected-20260930-v2/recovery/gpu_eqprop/build_cuda.sh +7 -0
  20. lineages/corrected-20260930-v2/recovery/gpu_eqprop/cuda_self_test.h +177 -0
  21. lineages/corrected-20260930-v2/recovery/gpu_eqprop/distill_stdin.h +98 -0
  22. lineages/corrected-20260930-v2/recovery/gpu_eqprop/eqprop_gpu.cu +2157 -0
  23. lineages/corrected-20260930-v2/recovery/gpu_eqprop/hf_checkpoint_service.py +458 -0
  24. lineages/corrected-20260930-v2/recovery/gpu_eqprop/live_status.py +93 -0
  25. lineages/corrected-20260930-v2/recovery/gpu_eqprop/qwen_live.py +283 -0
  26. lineages/corrected-20260930-v2/recovery/gpu_eqprop/run_cuda_l40s.sh +13 -0
  27. lineages/corrected-20260930-v2/recovery/gpu_eqprop/run_qwen_live.py +127 -0
  28. lineages/corrected-20260930-v2/recovery/gpu_eqprop/start_backup_services.py +38 -0
  29. lineages/corrected-20260930-v2/recovery/gpu_eqprop/start_qwen_l40s.sh +73 -0
  30. lineages/corrected-20260930-v2/recovery/gpu_eqprop/validate_curriculum.py +63 -0
  31. lineages/corrected-20260930-v2/recovery/hierarchical_tokenizer.py +126 -0
  32. lineages/corrected-20260930-v2/recovery/lineage.json +11 -0
  33. lineages/corrected-20260930-v2/recovery/recovery_policy.json +95 -0
  34. lineages/corrected-20260930-v2/recovery/source_checkpoint.json +12 -0
  35. lineages/corrected-20260930-v2/recovery_policy.json +95 -0
  36. lineages/corrected-20260930/LATEST.json +1 -0
  37. lineages/corrected-20260930/curriculum.json +31 -0
  38. lineages/corrected-20260930/lineage.json +10 -0
  39. lineages/corrected-20260930/mix_manifest.json +187 -0
  40. lineages/corrected-20260930/recovery/backup-manifest.json +119 -0
  41. lineages/corrected-20260930/recovery/data/corrected-mix/mix_manifest.json +187 -0
  42. lineages/corrected-20260930/recovery/data/language_baseline/recovery_manifest.json +22 -0
  43. lineages/corrected-20260930/recovery/data/language_baseline/vocab_sparse.json +3 -0
  44. lineages/corrected-20260930/recovery/data/slots/compiler_physics.nysa +0 -0
  45. lineages/corrected-20260930/recovery/data/slots/recovery_manifest.json +108 -0
  46. lineages/corrected-20260930/recovery/data/slots/speech_align.nysv +3 -0
  47. lineages/corrected-20260930/recovery/data/slots/speech_slot.nysv +0 -0
  48. lineages/corrected-20260930/recovery/data/slots/speech_vocoder.nyvc +0 -0
  49. lineages/corrected-20260930/recovery/gpu_eqprop/build_corrected_mix.py +217 -0
  50. lineages/corrected-20260930/recovery/gpu_eqprop/build_cuda.sh +7 -0
.gitattributes ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ lineages/corrected-20260930/recovery/data/language_baseline/vocab_sparse.json filter=lfs diff=lfs merge=lfs -text
37
+ lineages/corrected-20260930/recovery/data/slots/speech_align.nysv filter=lfs diff=lfs merge=lfs -text
38
+ lineages/corrected-20260930-v2/recovery/data/language_baseline/vocab_sparse.json filter=lfs diff=lfs merge=lfs -text
39
+ lineages/corrected-20260930-v2/recovery/data/slots/speech_align.nysv filter=lfs diff=lfs merge=lfs -text
PROGRAM_CHARTER.md ADDED
@@ -0,0 +1,246 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Seven-Model and Genesis Program Charter
2
+
3
+ Charter version: 1.0
4
+
5
+ Effective date: 2026-08-08
6
+
7
+ Status: authoritative cross-repository direction
8
+
9
+ Canonical copy: <https://github.com/dakuwonmoody-lab/EsoM/blob/main/PROGRAM_CHARTER.md>
10
+
11
+ ## Read this first
12
+
13
+ This charter exists so work in one repository does not lose the direction of the
14
+ whole program.
15
+
16
+ The program is building one AI business with seven selectable model families:
17
+ **BulmaX, Beerus, NYS, AXIOM, Mech, EsoM, and WHIS**. They are not seven personas on
18
+ one hidden model. Each is a distinct attempt to build useful intelligence from a
19
+ different computational substrate.
20
+
21
+ **Genesis is not an eighth model.** Genesis is the long-term hardware compute-
22
+ substrate program intended to discover, falsify, and eventually embody new
23
+ computational principles in non-CUDA physical hardware.
24
+
25
+ Anyone—human or agent—working in any participating repository must preserve these
26
+ identities and boundaries unless an explicit cross-program decision changes this
27
+ charter.
28
+
29
+ ## Mission
30
+
31
+ Within one outside-user-oriented product, a user should be able to select any of the
32
+ seven models and interact with that model's real native computation. The models may
33
+ have very different maturity, fluency, latency, memory, and availability. Those
34
+ differences must be visible rather than hidden.
35
+
36
+ The long-term ambition is not a conventional weak-to-strong size ladder. It is a
37
+ catalog of genuinely different kinds of machine intelligence sharing one product
38
+ surface, account system, safety boundary, and business identity.
39
+
40
+ ## The seven model theses
41
+
42
+ | Model | Durable architectural thesis | Intended character when mature |
43
+ |---|---|---|
44
+ | **BulmaX** | Adaptive multimodal autoregressive neural generation | Broad, fluent generalist |
45
+ | **Beerus** | Recurrent byte processing coupled to a growing, locally regulated swarm cortex | Continuous, adaptive streaming intelligence |
46
+ | **NYS** | Learned oscillator coupling and thermodynamic phase dynamics | Associative, resonant conceptual exploration |
47
+ | **AXIOM** | Exact causal uncertainty, interventions, and reversible machine embodiment | Investigative machine scientist |
48
+ | **Mech** | Compiled knowledge graphs, explicit rules, retrieval, and deterministic proofs | Precise librarian-logician |
49
+ | **EsoM** | A self-enciphering `.mal` organism with native memory, search, and gated mutation | Persistent executable organism |
50
+ | **WHIS** | Quality-diversity evolution over populations, niches, and lineage | Discovery collective producing diverse strategies |
51
+
52
+ These theses are durable. Their present implementations are not sacred.
53
+
54
+ A checkpoint, mechanism, trainer, representation, scale plan, or implementation
55
+ branch may be disproved, archived, or replaced. A failed route is useful when its
56
+ claim, controls, evidence, and failure are preserved. The program does not keep a
57
+ route alive merely because it consumed substantial time or money.
58
+
59
+ ## Genesis hardware thesis
60
+
61
+ Genesis begins before hardware, compilers, runtimes, and models because it is trying
62
+ to derive a compute model from first principles rather than inherit every CPU/GPU
63
+ assumption.
64
+
65
+ Its discovery track proposes candidate computational phenomena. Its falsification
66
+ track tries to eliminate artifacts, confounds, and false interpretations. Principles
67
+ that survive may become software primitives, instructions, compiler/runtime
68
+ semantics, RTL, FPGA implementations, and eventually physical hardware designed in
69
+ KiCad.
70
+
71
+ The stated staged target is a non-CUDA compute architecture capable of running and
72
+ training BulmaX through a host-CPU, reference-GPU, and Genesis-hardware system. The
73
+ first proposed integration is deliberately narrow: validate exact persistent
74
+ attention/KV-state behavior in software before FPGA offload. Other operations move
75
+ only after Genesis earns the necessary arithmetic, memory, learning, and correctness
76
+ primitives.
77
+
78
+ Genesis experiments are therefore neither side quests nor the final product. They
79
+ are the evidence pipeline for deciding what deserves embodiment in a new physical
80
+ compute substrate.
81
+
82
+ ## Non-negotiable product boundaries
83
+
84
+ ### 1. The selected model owns the answer
85
+
86
+ A shared gateway may authenticate, route, enforce budgets, resolve permissions, and
87
+ format envelopes. It may not secretly replace the selected model's cognition.
88
+
89
+ A model-specific parser or renderer may express a native result, but it must not
90
+ invent the substantive reasoning that the named substrate did not perform.
91
+
92
+ ### 2. Fallback is explicit
93
+
94
+ If one model cannot answer, it may abstain or request another named model. Automatic
95
+ fallback is allowed only with prior user consent. Every response preserves both the
96
+ requested model and the model that actually answered.
97
+
98
+ Another model's output must never be presented as though the selected model produced
99
+ it.
100
+
101
+ ### 3. Retrieval remains model-native
102
+
103
+ Repositories may share source storage, permissions, document identifiers, and
104
+ provenance infrastructure. Each model must still ingest retrieved material in a
105
+ form its own substrate can use and must own the inference that follows.
106
+
107
+ Retrieval finding the answer outside the substrate is not evidence that the model
108
+ can reason about it.
109
+
110
+ ### 4. Native state is isolated
111
+
112
+ Each model owns a separate state namespace, schema, reset rule, persistence rule,
113
+ and rollback boundary. Neural memory, swarm topology, oscillator phase, causal
114
+ quotients, fact graphs, `.mal` arenas, and evolutionary archives are not one
115
+ interchangeable database.
116
+
117
+ Cross-model transfer is allowed only through an explicit, typed, consented,
118
+ provenance-carrying artifact. It counts as learning only when the receiving model
119
+ can ingest it into native state and a before/after evaluation demonstrates a useful,
120
+ retained, reversible change.
121
+
122
+ ### 5. Capability claims require native evidence
123
+
124
+ Code presence is not capability evidence. Checkpoint existence is not intelligence
125
+ evidence. Lower loss is not automatically product progress. A bounded task does not
126
+ establish open-domain generalization.
127
+
128
+ Consequential experiments should predeclare the claim, baseline, metric, minimum
129
+ useful effect, resource ceiling, stop rule, artifact destination, and pass/fail
130
+ decision. Evaluations default to native-only execution so fallback cannot conceal a
131
+ failure.
132
+
133
+ ### 6. State changes and actions are permissioned
134
+
135
+ Persistent learning, self-modification, tool use, experiments, hardware actions,
136
+ and archive evolution require explicit permissions, resource ceilings, logs, and
137
+ rollback where possible. Outside-user access must begin narrowly and honestly.
138
+
139
+ ### 7. Backups and provenance are part of the architecture
140
+
141
+ Important source, checkpoints, experiment manifests, negative results, and lineage
142
+ must be committed or stored in their declared durable home. A result that cannot be
143
+ reconstructed from named revisions and artifacts should not guide an expensive next
144
+ step.
145
+
146
+ ## Shared product direction
147
+
148
+ All seven models should eventually be reachable through one versioned request and
149
+ response protocol, one model registry, and one outside-user-oriented chat surface.
150
+ The shared membrane is infrastructure, not another intelligence.
151
+
152
+ The first protocol implementation lives in EsoM under `src/model_family/` and is
153
+ documented at:
154
+
155
+ <https://github.com/dakuwonmoody-lab/EsoM/blob/main/docs/MODEL_PROTOCOL_V1.md>
156
+
157
+ It requires explicit requested/answering model attribution, native evidence,
158
+ state ownership, retrieval receipts, fallback receipts, resource budgets, and
159
+ truthful availability. No adapter should be marked online until it has an exact
160
+ runtime revision and passes conformance.
161
+
162
+ ## Compute direction
163
+
164
+ Progress is required across all seven model programs, but compute spending is not
165
+ equal.
166
+
167
+ - CPU-appropriate architecture, evaluation, protocol, and product work continues
168
+ even when accelerators are unavailable.
169
+ - Beerus and NYS use opportunistic accelerators only when a predeclared experiment
170
+ justifies them.
171
+ - Full-current-configuration BulmaX work uses funded B200-class campaigns. BulmaX
172
+ maintains a separate no-B200 queue for evaluation design, data audits, inference,
173
+ serving, recovery, controls, and run preparation.
174
+ - Every expensive run begins with a decision-changing run card and ends with
175
+ portable artifacts and frozen probes, not only a newer step number.
176
+ - Genesis hardware work proceeds only after software and reference-hardware gates
177
+ establish correctness and value.
178
+
179
+ ## Repository responsibilities
180
+
181
+ Every participating repository owns its native architecture, local tests, evidence,
182
+ state semantics, and failure history. Cross-repository product code must not erase
183
+ those responsibilities.
184
+
185
+ Before substantial work begins, identify:
186
+
187
+ 1. which model or Genesis thesis the work advances;
188
+ 2. the exact current route under test;
189
+ 3. the next falsifiable claim or native product rung;
190
+ 4. the baseline or control;
191
+ 5. the durable artifact destination; and
192
+ 6. what result will cause continuation, revision, pause, or archival.
193
+
194
+ Meaningful progress includes a measured capability gain, a native end-to-end rung,
195
+ a reliability improvement with evidence, a decisive ablation, or a negative result
196
+ that eliminates a plausible route. Activity alone is not progress.
197
+
198
+ ## Immediate program priorities
199
+
200
+ - **Shared product:** turn protocol v1 into a thin gateway and seven conforming
201
+ adapters with truthful availability.
202
+ - **BulmaX:** characterize the current checkpoint and make the next B200 campaign
203
+ decisive.
204
+ - **Beerus:** freeze Swarm-Cortex as the canonical identity and prove the swarm adds
205
+ value beyond the conv/GRU backbone.
206
+ - **NYS:** align sparse inference with the current sparse trainer and test ordered
207
+ language plus phase ablations.
208
+ - **AXIOM:** finish A2 acceptance and the falsification/discovery campaign before
209
+ broad language work.
210
+ - **Mech:** turn saved open-domain failures into regressions and win a curated domain
211
+ through correctness, provenance, and calibrated unknowns.
212
+ - **EsoM:** resolve or redesign the re-entrant composition route while keeping
213
+ answer-producing cognition inside `.mal`.
214
+ - **WHIS:** prove archive diversity improves a predeclared downstream consumer.
215
+ - **Genesis:** continue falsification-led principle discovery while protecting the
216
+ staged bridge from software reference to FPGA and physical compute.
217
+
218
+ ## Canonical detailed documents
219
+
220
+ - Architecture and evidence audit:
221
+ <https://github.com/dakuwonmoody-lab/EsoM/blob/main/docs/MODEL_FAMILY_ARCHITECTURE.md>
222
+ - One-year execution roadmap:
223
+ <https://github.com/dakuwonmoody-lab/EsoM/blob/main/docs/SEVEN_MODEL_EXECUTION_ROADMAP.md>
224
+ - Shared protocol v1:
225
+ <https://github.com/dakuwonmoody-lab/EsoM/blob/main/docs/MODEL_PROTOCOL_V1.md>
226
+ - Genesis hardware direction:
227
+ <https://github.com/dakuwonmoody-lab/genesis-computing/blob/main/README.md>
228
+ - Genesis staged hybrid architecture:
229
+ <https://github.com/dakuwonmoody-lab/genesis-computing/blob/main/HYBRID_ARCHITECTURE.md>
230
+
231
+ Repository-local specifications override the detailed implementation mechanics for
232
+ their substrate. They do not silently override this cross-program product direction.
233
+
234
+ ## Synchronization rule
235
+
236
+ `PROGRAM_CHARTER.md` is mirrored at the root of every active source repository and
237
+ relevant Hugging Face repository. Mirrors must remain byte-identical for a given
238
+ charter version.
239
+
240
+ Changes begin in the canonical EsoM copy, increment the charter version, and are
241
+ then synchronized across all mirrors in one documented operation. Do not make an
242
+ independent repository-local edit to this file. Propose the change against the
243
+ canonical copy and propagate it everywhere after acceptance.
244
+
245
+ If a mirror conflicts with the canonical copy at the same version, the canonical
246
+ EsoM copy wins and the mismatch must be reported.
README.md ADDED
@@ -0,0 +1,829 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: mit
3
+ library_name: nys-thermodynamic
4
+ pipeline_tag: other
5
+ language:
6
+ - en
7
+ tags:
8
+ - kuramoto-oscillator
9
+ - equilibrium-propagation
10
+ - sparse-csr
11
+ - thermodynamic-computing
12
+ - non-transformer
13
+ - zero-backprop
14
+ - hip
15
+ - mi300x
16
+ - mi325x
17
+ - cuda
18
+ - sycl
19
+ - sparse-retrieval
20
+ - long-context-memory
21
+ - experimental
22
+ - compiler-physics
23
+ - voice-crystal
24
+ - speech-vocoder
25
+ - sft
26
+ datasets:
27
+ - YNSScarSaiyan/nys-corpus
28
+ - HuggingFaceH4/ultrachat_200k
29
+ ---
30
+
31
+ # NYS — Sparse Kuramoto EqProp (public lineage)
32
+
33
+ **NYS** is not a transformer. It is a coupled-oscillator organism: phases
34
+ \(\theta_i\), amplitudes, natural frequencies \(\omega_i\), and a sparse
35
+ coupling graph \(K\). Language, compiler bits, and speech-frame IDs live on
36
+ **one** graph (\(N = 5{,}000{,}000\) nodes, \(k = 512\) edges per node).
37
+ The native graph trains by **Equilibrium Propagation** (free RK4, then
38
+ same-moment free/nudged branches, then a contrastive \(K\) update), without
39
+ backprop or a cross-entropy objective. A separate sparse context head now
40
+ provides retrieved language inputs. The GPU Qwen teacher uses
41
+ PyTorch/Transformers; that does not make the native graph a transformer.
42
+
43
+ *Dakuwon Moody (YNSScarSaiyan) · Saiyan Corp*
44
+
45
+ ---
46
+
47
+ ## Current status — 2026-10-09
48
+
49
+ **The 16,777,216-token-capacity context adapter is integrated into live
50
+ NYS language training. It is sparse indexed memory, not a 16M-token dense
51
+ attention window or proof of 16M-token language understanding.** NYS remains
52
+ an experimental research artifact, not a validated production chatbot.
53
+
54
+ Snapshot: **2026-10-09 15:29:56 UTC**. These are dated observations, not a
55
+ live dashboard or a promise that the target has been reached.
56
+
57
+ | Item | Verified state |
58
+ |---|---|
59
+ | Run / hardware | `qwen-mi325x-4b-20261009-r4`; AMD Instinct MI325X / HIP |
60
+ | Observed native step | **2,213,680**, active at this check |
61
+ | Continuation | Native + context head resumed at 2,190,210; absolute stop target **7,075,000** |
62
+ | Saves | Every **5,000 native steps**, plus first-complete-rotation checkpoint |
63
+ | Teacher | [Qwen3-4B](https://huggingface.co/Qwen/Qwen3-4B), on GPU; revision `1cfa9a7208912126459214e8b04321603b3df60c` |
64
+ | Native graph | 5,000,000 nodes, 512 CSR edges per row; unchanged by context integration |
65
+ | Memory capacity | **16,777,216 frozen NYS token IDs**, not Qwen tokenizer tokens |
66
+ | Head geometry | **1,048,576 hash features + 560 legacy features = 1,049,136 parameters**, not 16M parameters in this deployment |
67
+ | Context training | 4,954 cumulative committed head updates; **4,694 new in r4**, after restoring 260 previous updates |
68
+ | Retrieval counters | 60,958 cumulative injected retrieved IDs; 449,537 represented source IDs processed; zero recorded budget abstentions |
69
+ | Latest verified HF pair at this check | Native step **2,210,000** + matching head; recovery details below |
70
+ | Reporters | CPU-only fixed-fixture and frozen generalization reporters attached; **graph-only controls**, not context-adapter quality tests |
71
+
72
+ Cumulative source IDs processed are not the size of one retained context.
73
+ Independent corpus rows reset the previous row's memory. Most live prompts
74
+ are much shorter than capacity. Training counters and `contrast` are **not
75
+ accuracy or generalization evidence**.
76
+
77
+ Source is pinned at GitHub commit
78
+ [`df0024a30542b57d7946ea39fd453ace6180ee58`](https://github.com/dakuwonmoody-lab/nys/tree/df0024a30542b57d7946ea39fd453ace6180ee58/NYS).
79
+ The [backend workflow](https://github.com/dakuwonmoody-lab/nys/actions/runs/37933671238)
80
+ passed Linux CPU, Windows CPU, CUDA compilation, HIP compilation, and
81
+ Intel SYCL compilation. Compilation is not GPU runtime validation: real
82
+ CUDA tests ran separately on an A100, and real HIP tests on the MI325X.
83
+ **Intel GPU runtime has not been demonstrated.** Compact public provenance
84
+ and result summaries are in [the evidence snapshot](documentation/20261009-evidence.json).
85
+
86
+ ### Live context mechanism and validation
87
+
88
+ Full represented prompts are indexed on the host in an ordered SQLite
89
+ lexical-occurrence store. The shared sparse head scores candidates using
90
+ an explicitly selected CPU, CUDA, HIP, or Intel SYCL numerical backend.
91
+ The full index is not copied into GPU memory.
92
+
93
+ In the deployed language step, **up to 12 retrieved IDs with explicit word
94
+ boundaries plus the most recent 96 IDs** enter a **127-ID maximum prompt
95
+ including role framing**. The native graph and Qwen teacher see the same
96
+ represented content. Selected older evidence reaches the short native
97
+ prompt; the graph does not attend to every stored token simultaneously.
98
+ Retrieval is bounded to 32 query lexical units, 128 occurrences per cue,
99
+ and 4,096 candidate IDs. Budget exhaustion causes reported abstention
100
+ from evidence injection and the associated head update. Common/repeated
101
+ cues can exhaust these budgets.
102
+
103
+ Retrieval happens **before teacher targets exist**. Projected Qwen
104
+ probabilities update the separate head only after native acknowledgement
105
+ of the matching complete five-step curriculum rotation. The teacher cannot
106
+ run ahead. Rejected projections discard an unconsumed row without updating
107
+ either model. Row-isolated indexes prevent cross-example context leakage.
108
+
109
+ The head uses analytic free/nudged equilibria of a quadratic energy, with
110
+ sparse contrastive updates, decay, and clipping: **not** the native graph's
111
+ Kuramoto RK4 dynamics. Its parameters are separate, but save/resume requires
112
+ a logically paired native/head checkpoint. Context integration does not
113
+ rewire the graph, add a sixth curriculum step, or **replace NYS's output
114
+ with a retrieved copy**.
115
+
116
+ r3's retrieved-input formatting was corrected to preserve explicit word
117
+ boundaries. r4 restored native progress and all 260 committed head updates;
118
+ it did not silently reset the learned head. See the pinned
119
+ [implementation/build guide](https://github.com/dakuwonmoody-lab/nys/blob/df0024a30542b57d7946ea39fd453ace6180ee58/NYS/context_slot/README.md).
120
+
121
+ Verified checks, with their boundaries:
122
+
123
+ - CUDA storage/reachability probes at **1,048,576 and 16,777,216 IDs**
124
+ passed using handcrafted relation weights and punctuation distractors.
125
+ These are not learned long-document recall or broad reasoning tests.
126
+ - An isolated **4,096-node, k=32** HIP graph completed two full five-step
127
+ rotations: actual native/head HIP updates, retrieval before labels,
128
+ longer-than-old-limit inputs reaching the native stream, row isolation,
129
+ paired saving, and bitwise head restoration. This is an integration
130
+ check, not a full-size accuracy benchmark.
131
+ - The first r4 production checkpoint at **2,190,215** had
132
+ `live_deployed=true` and **712 nonzero learned head weights**. Its head
133
+ restored **bit-for-bit** in a separate bounded HIP process. Native +
134
+ head + integrity metadata were uploaded and verified together on HF.
135
+ - Production journals show continuing head updates and retrieved-input
136
+ injection. These establish actual use, **not an accuracy improvement**.
137
+
138
+ ### Current evaluations and unresolved limits
139
+
140
+ Checkpoint **2,210,000**, reported at **15:11:58 UTC**, was evaluated on
141
+ frozen September 30 cases: 512 external next-token cases, 128 arithmetic
142
+ cases, and 128 nonce-binding pairs / 256 binding cases. These are reused
143
+ regression tests, not a newly sealed uncontaminated benchmark. Exact-context
144
+ audits cannot exclude historical or semantic training overlap.
145
+
146
+ | Graph-only readout | External top-1 | Arithmetic top-1 | Binding top-1 / complete pairs |
147
+ |---|---|---|---|
148
+ | Stored phase | 0 / 512 (0%) | 0 / 128 (0%) | 0 / 256; 0 / 128 pairs |
149
+ | Weight only | 2 / 512 (0.390625%) | 0 / 128 (0%) | 0 / 256; 0 / 128 pairs |
150
+ | Canonical CPU settle | 2 / 512 (0.390625%) | 0 / 128 (0%) | 0 / 256; 0 / 128 pairs |
151
+
152
+ Binding constrained-choice was **50%**, chance, with identical predictions
153
+ for both members of all 128 pairs. Canonical CPU settle is an experimental
154
+ readout, not the live native reader. Next-token matching is not conversational
155
+ accuracy. **Broad generalization is not proven.** These reporters do not
156
+ use the context adapter, so cannot establish its quality. A matched
157
+ context-on/context-off language evaluation remains missing.
158
+
159
+ Fixed training-fixture diagnostics at this checkpoint:
160
+
161
+ - Language mix: **5 / 96 clamped top-1 (5.2083%)**; **74 / 96** targets
162
+ had direct prompt-to-gold edges. **22 / 96** still lack these sampled
163
+ connections. Stronger existing weights cannot create absent edges;
164
+ context integration did not repair native topology.
165
+ - Compiler: **80.3828% clamped free-bit accuracy** vs **61.7225%** after
166
+ 25 CPU settle steps on fixed gadgets. Different diagnostics, not
167
+ open-ended compilation or execution correctness.
168
+ - Speech: gold-edge weight lead **0.200936**; gold strongest in its wired
169
+ pool **16.8055%**, vs untrained pool reference **13.7649%**. Pool
170
+ comparison is **not speech recognition/synthesis accuracy**.
171
+
172
+ Recovered corpora and the speech codebook differ from the September
173
+ reporting series. These numbers are not a controlled before/after
174
+ comparison with historical tables below.
175
+
176
+ An independent **A100-SXM4-80GB topology-repair A/B** started from step
177
+ **2,120,000**. Each arm ran 1,440 CUDA updates on the same 288 language
178
+ records over five epochs. Repair changed 54,546 edges; 381 added edges
179
+ changed weight during native training. Sampled direct coverage rose from
180
+ 74/96 to 78/96. But canonical external accuracy stayed **3/512** for
181
+ source, control, and repair, arithmetic **1/128**, and complete binding
182
+ pairs **0/128**. Repair vs control had zero helpful and zero harmful
183
+ canonical prediction changes. **The candidate was not promoted.** This
184
+ bounded language-only trial does not evaluate the full live curriculum
185
+ or the newly integrated context adapter.
186
+
187
+ ### Paired checkpoint downloads and recovery
188
+
189
+ Known verified pair at the dated snapshot:
190
+
191
+ - [Native checkpoint at step 2,210,000](https://huggingface.co/YNSScarSaiyan/nys-sft-public/blob/9cf0a6f80a49a899d5c6583445dd1a7b9493ab14/lineages/qwen-mi325x-4b-20261009-r4/checkpoints/gpu_eqprop_2210000_20261009_150930.bin)
192
+ - Immutable HF revision: `9cf0a6f80a49a899d5c6583445dd1a7b9493ab14`
193
+ - Native bytes: **20,640,000,016**; SHA256:
194
+ `74494f292bf836582b723004d6fcee72af3b9462312079bce0d6069de28c4369`
195
+ - Required same-name `.bin.context/` directory: **`model.json`,
196
+ `weights.npy`, `pair.json`**. The native `.bin.json` upload record also
197
+ records sizes and hashes.
198
+
199
+ Download this pair using the current `hf` CLI:
200
+
201
+ ```sh
202
+ hf download YNSScarSaiyan/nys-sft-public \
203
+ --revision 9cf0a6f80a49a899d5c6583445dd1a7b9493ab14 \
204
+ --include 'lineages/qwen-mi325x-4b-20261009-r4/checkpoints/gpu_eqprop_2210000_20261009_150930.bin*' \
205
+ --local-dir ./nys-checkpoint
206
+ ```
207
+
208
+ The wildcard includes the native file, upload JSON, and three companion
209
+ files. Preserve their relative layout. Verify native size/header/hash and
210
+ companion hashes, vocabulary, geometry, basename, and exact step. A `.bin`
211
+ alone is **not** a complete context-enabled resume. `pair.json` is published
212
+ only after the head/config files are durable.
213
+
214
+ Retention commits both halves and recovery metadata together, keeps the
215
+ **two newest complete local snapshots**, and protects recovery inputs.
216
+ Local pruning requires fresh integrity plus immutable-revision **and**
217
+ current-remote verification of native and context files. Remote snapshots
218
+ are never deleted. Reviewed source/launch recipes are in the lineage's
219
+ `recovery/` directory.
220
+
221
+ Context resume requires `--context-library`, explicit `--context-backend`,
222
+ and `--context-resume CHECKPOINT` paired with the exact protected native
223
+ recovery checkpoint. Omitting context resume creates a new head, not
224
+ continued learned context state. See the pinned
225
+ [guarded supervisor](https://github.com/dakuwonmoody-lab/nys/blob/df0024a30542b57d7946ea39fd453ace6180ee58/NYS/gpu_eqprop/run_mi325x.py)
226
+ and [paired retention worker](https://github.com/dakuwonmoody-lab/nys/blob/df0024a30542b57d7946ea39fd453ace6180ee58/NYS/gpu_eqprop/verified_hf_retention.py).
227
+
228
+ The [read-only context reader](https://github.com/dakuwonmoody-lab/nys/blob/df0024a30542b57d7946ea39fd453ace6180ee58/NYS/gpu_eqprop/live_context_reader.py)
229
+ loads the same head/input adapter and a read-only graph. Its final output
230
+ is a **CPU KOPG graph readout**, not an automatic copy answer or a demonstration
231
+ of native RK4 text generation. This custom format is **not** a Transformers
232
+ `AutoModel.from_pretrained` checkpoint.
233
+
234
+ ---
235
+
236
+ ## Historical diagnosis — 2026-09-11: the training rule was measured to be broken, and why
237
+
238
+ This section preserves the diagnosis and verification status **as written
239
+ on September 11**. The corrected rule is now deployed in the October
240
+ continuation and has passed real HIP integration/numerical checks. Runtime
241
+ correctness does not establish improved accuracy; use dated results above.
242
+
243
+ Every accuracy number below step ~1,350,000 in this card was produced under
244
+ a contrastive update that we have since measured to be **dominated by noise
245
+ unrelated to the training signal**, on every slot. This is not a data
246
+ problem, a wiring problem, or a readout problem — those were real and were
247
+ fixed first (see the changelog below) — it is a problem in how this specific
248
+ kernel computed the EqProp contrast. Recording the failure honestly, and the
249
+ fix, in public:
250
+
251
+ **The mechanism.** Each training step: 5 RK4 steps free, then 5 more RK4
252
+ steps with the nudge force on, then
253
+ \(\Delta K_{ij} \propto \cos\Delta\theta_{\mathrm{nudge}} - \cos\Delta\theta_{\mathrm{free}}\)
254
+ on gated edges. That contrasts the state **5 steps later** against the state
255
+ **5 steps earlier** — two different points in time, not two conditions at
256
+ the same moment. Natural frequencies \(\omega_i\) span 50–1000 rad/s
257
+ (`host_init_graph`, uniform), so in the 0.1 s (`5 × dt=0.02`) between the two
258
+ snapshots a typical pair of nodes counter-rotates by **~36 rad** from
259
+ \(\omega\) alone. The nudge itself moves a target by **~0.003 rad** in the
260
+ same window. We measured this directly by replaying real training steps
261
+ offline against a live checkpoint, once with the nudge force on and once
262
+ with it zeroed (same amplitudes, same gated edges), and comparing:
263
+
264
+ | | value |
265
+ |---|---|
266
+ | Phase drift between snapshots from \(\omega\) alone | ~36 rad |
267
+ | Phase shift a gold target gets from the nudge | ~0.003 rad |
268
+ | Size of \(\Delta K\) that has nothing to do with the nudge (std) | ~1.0 |
269
+ | Size of the part the nudge actually contributes (mean, std) | −0.00016, 0.0035 |
270
+ | Signal-to-noise ratio on the update | **~1 : 6,000** |
271
+
272
+ At that SNR, \(K\) on any edge is a random walk that happens to be centered
273
+ near the target most of the time — not a trained value. This explains a
274
+ pattern that was otherwise puzzling: **the speech clamped-pair accuracy
275
+ went from 16.1% at 930k to 4.4% at 1,350,000** — worse, after 420,000 more
276
+ steps, on a slot that was correctly wired the whole time. A noise-dominated
277
+ update drifts; it does not reliably improve.
278
+
279
+ A second, smaller effect compounded this on speech specifically: the target
280
+ phase (\(\bar\theta\), what nudged nodes are pulled toward) was computed as
281
+ the mean over the **entire prompt**, which for a typical speech step is 32
282
+ neighbor-fill tokens spliced in for exploration plus ~2.3 real text tokens.
283
+ The fill tokens were **91.5%** of that mean. Gold frames were being pulled
284
+ toward the phase of unrelated neighbor tokens, not the phase of the text
285
+ that was supposed to teach them.
286
+
287
+ **The fix (deployed 2026-09-11, verified offline, live verification in
288
+ progress — check the changelog / latest checkpoint metadata for confirmed
289
+ results before citing numbers past this point):**
290
+
291
+ 1. **Same-moment contrast.** After the free phase, two branches run from the
292
+ identical state: one continues free, one is nudged, both for the same
293
+ number of steps. \(\Delta K\) contrasts those two — the \(\omega\)
294
+ rotation is common to both branches and cancels exactly. Costs one extra
295
+ 5-step integration per training step.
296
+ 2. **Target from the real prompt only**, not the neighbor-fill tokens.
297
+ 3. **\(\omega\) scaled by 0.01 inside the integrator** (`--omega-scale`, the
298
+ stored \(\omega\) in the checkpoint is untouched and this is reversible
299
+ by restarting with a different value). Comparing branches removes the
300
+ drift, but at the original \(\omega\) magnitude no pair can phase-lock
301
+ within a 5-step nudge window regardless — the coupling term is too slow
302
+ relative to the free rotation. Offline replay on 40 real speech records,
303
+ same edges, before vs. after:
304
+
305
+ | Rule | gold-edge \(\Delta K\) | negative-edge \(\Delta K\) |
306
+ |---|---|---|
307
+ | Old (time-shifted, \(\omega\) full) | −0.00012 (−2.1 SE) — sign noise | +0.0001 (+1.6 SE) |
308
+ | Same-moment, text-only mean, \(\omega\) full | −0.00012 (−2.1 SE) — still noise | +0.0001 |
309
+ | Same-moment, text-only mean, **\(\omega \times 0.01\)** | **+0.0044 (+23 SE)** | **−0.0048 (−22 SE)** |
310
+
311
+ Both signs correct, both far outside noise, only once \(\omega\) is
312
+ slowed. A HIP self-test confirmed the new kernel runs correctly on-GPU at
313
+ both settings before it went into the training loop.
314
+
315
+ **What this means for the numbers in the rest of this card.** They are real
316
+ measurements of a real (and, until now, undiagnosed) failure mode — useful
317
+ for exactly that reason — but they are **not** evidence about what this
318
+ graph and this coupling rule can or cannot learn. Compiler and speech being
319
+ stuck near chance was previously attributed to topology and shared-register
320
+ conflicts (both real, both fixed at 925k and 2026-09-10 respectively); it
321
+ now looks like the update itself never had the SNR to learn regardless. That
322
+ question is open again, correctly this time, and is what the current run is
323
+ testing.
324
+
325
+ ---
326
+
327
+ ## Two artifacts on this repo — do not confuse them
328
+
329
+ | Lineage | Files | Size | Graph | Trainer | Can `tlc-infer` load it? |
330
+ |---|---|---|---|---|---|
331
+ | **Current — sparse GPU EqProp** | `gpu_eqprop_<step>_<utc>.bin` | **20,640,000,016 B** exactly | \(N=5\times10^6\), CSR \(k=512\) | HIP `eqprop_gpu` (no PyTorch) | **No** |
332
+ | **Legacy — dense ASM SFT** | `nys_sft_final.bin` | ~32–34 GiB | \(N=65{,}536\) dense \(K\) | x86-64 NASM / AVX-512 | Yes (old path) |
333
+
334
+ The model card you are reading describes the **sparse GPU** lineage. The
335
+ legacy file is kept for history. `tlc-infer` / `sampler.cpp` still assume
336
+ the 65,536 dense \(K\). Dropping a `gpu_eqprop_*.bin` into that binary will
337
+ not work.
338
+ Context-enabled sparse checkpoints additionally require their learned-head
339
+ companion bundle; the native binary layout remains the same.
340
+
341
+ A complete CPU sparse base also exists on
342
+ [`YNSScarSaiyan/nys-checkpoints`](https://huggingface.co/YNSScarSaiyan/nys-checkpoints)
343
+ (`base_sparse_*`). **This public GPU run did not resume that file.** It was
344
+ `--init` (random ring + stubs) at step 0, then trained on-card. Same \(N\),
345
+ same \(k\) the whole way. No second `--init`, ever, on this lineage.
346
+
347
+ ---
348
+
349
+ ## Changelog (most recent first)
350
+
351
+ - **2026-10-09 — native context integration and paired recovery.**
352
+ 16,777,216-ID memory feeds retrieved evidence into the existing Qwen
353
+ language step; target-blind retrieval, native-acknowledged head updates,
354
+ word-boundary correction, learned-head-preserving resume, and hash-verified
355
+ native + head HF retention. CUDA capacity and actual HIP integration
356
+ checks passed; broad language generalization remains unproven.
357
+ - **2026-10-09 — A100 repair A/B.** Connectivity and native learnability
358
+ improved, canonical accuracy did not. Candidate not deployed.
359
+ - **October continuation — MI325X and GPU Qwen3-4B.** Complete verified
360
+ five-step curriculum, pinned teacher/data, 5,000-step saves, guarded
361
+ resources, and CPU-only accuracy/generalization reporters attached.
362
+ - **2026-09-11 — contrastive-update SNR fix.** Same-moment branch contrast
363
+ (cancels \(\omega\) drift exactly), target phase from real prompt tokens
364
+ only, \(\omega \times 0.01\) inside the integrator. See the section above.
365
+ Offline-verified at that date; October corrected training is live.
366
+ Runtime correctness and measured capability remain separate.
367
+ - **2026-09-10 — speech data/training fixes**, in order:
368
+ - Vocoder codebook rebuilt: the old k-means fit left **235 of 256** atoms
369
+ at their identity-DC initialization (empty-cluster failure — a centroid
370
+ with no assigned frames never moves). One code covered **79%** of every
371
+ frame in the training corpus. Refit with data-seeded k-means++ and
372
+ empty-cluster reseeding from the worst-fit frame: **1 of 256** atoms
373
+ left flat, all 256 codes in use, reconstruction SNR 2.9→9.2 dB.
374
+ - Silence frames (the code that maps to near-zero-variance PCM, ~65–75%
375
+ of real speech audio) excluded from both training targets and negative
376
+ examples, on both the HIP trainer and the Python wiring pass. They were
377
+ previously being pulled toward every record's own phase simultaneously
378
+ — the same "shared register, many writers" conflict described below for
379
+ the compiler band.
380
+ - Speech negative-example generation fixed: the old rule added `code+1`
381
+ **and** `code+127` neighbors of every gold frame as negatives appended
382
+ to the *prompt* (so they dominated the target-phase mean; see above),
383
+ and did not check whether a negative for one record was gold for
384
+ another. Measured on the live corpus: **39.9%** of positive pushes were
385
+ being fought by a negative push on the identical physical node. Fixed:
386
+ one negative per gold frame, checked against a corpus-wide gold set
387
+ (`load_global_gold_frames`) so a negative can never be a duplicate of
388
+ someone else's target.
389
+ - \(K\)-value clamp added to the kernel (\(\lvert K \rvert \le 3\)):
390
+ contested nodes had walked to \(\lvert K \rvert = 16\)–18 (vs. a
391
+ wiring-pass seed of \(K=0.2\)) before this was caught; a stale hard
392
+ reset of the speech band's edges (`--reset-speech`) was run once to
393
+ clear the accumulated damage.
394
+ - **2026-09-09 (925k) — CSR wire.** Random ring+stubs never connected vocab
395
+ tokens to the compiler band (`3,800,000+`) or the speech band
396
+ (`4,300,000+`). Wired in place on existing \(k=512\) slots (weakest
397
+ non-protected edge replaced, seed \(K=0.2\)). \(N\), \(k\) unchanged.
398
+ - Earlier: neighbor-fill (32), speech \(\beta\) floor (0.25), compiler
399
+ signed 0-bit nudge — see prior card revisions in this repo's commit
400
+ history.
401
+
402
+ ---
403
+
404
+ ## Status at 1,350,000 — under the pre-2026-09-11 rule (see caveat above)
405
+
406
+ These are the last numbers produced before the SNR fix. Treat them as a
407
+ record of the failure mode, not a capability grade.
408
+
409
+ | Slot | 930k | 1,350,000 | Trend |
410
+ |---|---|---|---|
411
+ | Compiler free bits (clamped-pair) | 49.2% | 52.8% | Flat-ish, chance ≈ 50% either way (38 free bits, 11 shared gadgets) |
412
+ | Speech codes (clamped-pair, in-pool) | 16.1% (in-pool 58%) | **4.4%** (in-pool 41%) | **Degraded** — random-walk signature |
413
+ | Language headline | 0.625% | 0.0% | Noise-level throughout |
414
+
415
+ Compiler additionally has its own, separate confound even with a correct
416
+ update: **11 gold gadgets share the same 64 bit-oscillators.** A settle-based
417
+ probe (clamp the spec, run free RK4, read where the 64 nodes land — bypassing
418
+ the old "read whatever theta a node was last left at" readout convention)
419
+ confirmed the band **is** driven by the spec (no zero-displacement edges),
420
+ but different specs produced statistically **unrelated** crystals (pairwise
421
+ Hamming ~30/64, chance is 32) — the register is being fought over, not
422
+ un-addressed. That conflict is unresolved and is a second, independent
423
+ reason compiler will need more than the SNR fix.
424
+
425
+ ### Sidecar quizzes at 930k (historical; see caveat above)
426
+
427
+ | Quiz | 930k | What it actually measures |
428
+ |---|---|---|
429
+ | Compiler Hamming **free** | 15 / 38 (chance ≈ 19) | Unconditioned 64-bit crystal vs 11 golds |
430
+ | Speech `frame_match` (max-over-shift) | 1.49%, locked=false | One global argmax tape vs many gold NYSV records — low ceiling by construction even under a working update |
431
+ | Vocoder codec (pre-rebuild) | 21/256 atoms moved | Superseded 2026-09-10; now 255/256 |
432
+ | Chat (KOPG, SFT prefix) | fragments, `looped=false` | Static 2-hop neighbor walk, **not** HIP RK4 generation |
433
+
434
+ 930k chat probe (for the record, unaffected by any of the above fixes since
435
+ language readout is a separate mechanism from the SNR issue affecting the
436
+ *trained* signal — though the same broken contrastive update was training it
437
+ the whole time):
438
+
439
+ - `hello` → `based on you provide of your was a by the there are based, we importantafter`
440
+ - `the result is` → `l0_wra in the of the`
441
+ - `nys is listening` → `"_i3gw98h`
442
+
443
+ Not a chatbot. Neighbor structure is not uniform noise; replies are still
444
+ template / salad. Whether that ceiling is topology (briefing hypothesis:
445
+ fixed ring+stubs never contain the right edge) or the same SNR problem as
446
+ speech/compiler is now an open question again — it hasn't been re-tested
447
+ under the fixed update yet.
448
+
449
+ ---
450
+
451
+ ## What the organism is
452
+
453
+ NYS is one substrate with five **slots**. Slots 1, 2, and 4 train. Slot
454
+ 3 is a crystal→beep readout. Slot 5 is a dump envelope (**not trained**).
455
+ These organism slots are **not** the five curriculum steps (`mix`, `hard`,
456
+ `qwen_direct`, `compiler`, `speech`): the first three all train language.
457
+ Context feeds `qwen_direct`; voice and dump remain non-training slots.
458
+
459
+ | Slot | Trains? | IDs / files | What it is |
460
+ |---|---|---|---|
461
+ | **1 Language** | Yes | Frozen `vocab_sparse.json` (~3,664,150 IDs). Hard next-token + SDS1 distill + live mix. | Hierarchical token IDs. |
462
+ | **2 Compiler** | Yes | **3,800,000 + bit**, 64 active bits (8 bytes × 8 spins). `compiler_physics.nysa` (412 recs). | Spec-conditioned gold x86 gadgets. 11 gadgets share one 64-bit register — unresolved conflict, independent of the SNR fix. |
463
+ | **3 Voice crystal** | No | Digit tones from the crystal integer (350 Hz + 50 Hz/digit, 8 kHz). | Beeps that report a number. **Not speech.** |
464
+ | **4 Speech** | Yes | **4,300,000 + \(t\cdot 256\) + code**. `speech_slot.nysv` (6) + `speech_align.nysv` (305). Vocoder `speech_vocoder.nyvc` (rebuilt 2026-09-10, 255/256 atoms trained). | Frame-code IDs on the same graph. |
465
+ | **5 Dump** | **No** | `dump_slot.nysd` (NYSD kinds 6/7). Inbox wrap only. | Debug dumps. Do not train NYSD. |
466
+
467
+ Compiler *execution* of open-ended x86 is a quench + `crystallize` +
468
+ `mprotect`/`CALL` path (`execute.asm`). This sidecar **does not** execute the
469
+ crystal. Demo gold (8-byte active, LSB-first spins):
470
+
471
+ - `f(x)=x+42`: `48 89 F8 48 83 C0 2A C3`
472
+ - `f(x)=x*x`: `48 89 F8 48 0F AF C7 C3`
473
+ - plus ident, inc, dec, neg, not, shl1, add_self, zero, sub1
474
+ (**11** gadgets on the **same** 64 bit nodes — this is the shared-register
475
+ conflict noted above, separate from and in addition to the SNR issue)
476
+
477
+ Spins freeze LSB-first, 8 spins/byte, bit = \(\mathrm{sign}(a_i \cos\theta_i)\).
478
+ **26 bits are tied** across all padded golds; **38 are free**. Grade free
479
+ bits.
480
+
481
+ ---
482
+
483
+ ## Physics and trainer
484
+
485
+ ### Equation and active set
486
+
487
+ The HIP fused kernel (`eqprop_gpu.hip`, `hipcc`, `--offload-arch=gfx942`)
488
+ does **not** wrap all \(N\) oscillators and does **not** explode 512
489
+ neighbors of neighbors. As of the 2026-09-11 fix, each training step:
490
+
491
+ 1. Build \(S = \mathrm{unique}(\mathrm{prompt} \cup \mathrm{nudge\ IDs})\),
492
+ hard cap **256**. Prompt is packed first; if \(|S|>256\), the tail is
493
+ dropped.
494
+ 2. Prompt nodes start at amplitude 1; others 0.
495
+ 3. **Neighbor fill:** up to 32 unseen CSR neighbors of the last prompt token
496
+ are spliced into the prompt (amp 1) for exploration. The target-phase
497
+ mean (next step) is computed from the **real prompt only**, excluding
498
+ these — fixed 2026-09-11; previously the fill dominated the mean.
499
+ 4. **Free** RK4: 5 steps, \(\mathrm{d}t = 0.02\), \(\omega\) scaled by
500
+ `--omega-scale` inside the integrator (default in the current run:
501
+ **0.01**; the checkpoint's stored \(\omega\) is never modified, so this
502
+ is a launch-time choice, reversible by restarting with a different
503
+ value).
504
+ 5. **Branch point.** From the free-phase end state:
505
+ - **Branch A (free-continued):** 5 more RK4 steps, no nudge.
506
+ - **Branch B (nudged):** 5 more RK4 steps from the *same* starting
507
+ state, force \(\beta \sin(\bar\theta - \theta_i)\) on nudge targets.
508
+ \(\bar\theta\) is the circular mean of the real prompt at the branch
509
+ point (see step 3).
510
+ 6. Nudge targets get amplitude \(\min(1, |\beta|)\) for branch B. Amp gate
511
+ for \(K\) updates is **0.1**.
512
+ 7. Contrastive \(K\): on CSR edges whose both ends have amp \(> 0.1\),
513
+ \(\Delta K_{ij} \leftarrow \eta\,(\cos\Delta\theta_{B} - \cos\Delta\theta_{A})\)
514
+ — both branches measured at the **same** elapsed time from the **same**
515
+ starting state, so \(\omega\)-driven rotation is common to both and
516
+ cancels. \(K\) is clamped to \(\lvert K \rvert \le 3\) after each update.
517
+ 8. Amplitudes written back to 0.
518
+
519
+ Before 2026-09-11, step 5 did not exist: branch B was compared directly
520
+ against the step-4 (free) snapshot, 5 steps earlier in time. See the
521
+ diagnosis at the top of this card.
522
+
523
+ Defaults: \(\eta = 0.05\), \(\beta = 1.5\). One HIP block, 256 threads.
524
+ Model state ~20.64 GB HBM. The historical cap was 77 GiB. The October
525
+ continuation caps native owned allocation at **28 GiB** and configures a
526
+ **12 GiB** GPU-teacher budget, with shared-host reserves including at least
527
+ 16 GiB free VRAM. Guards do not count every driver/runtime allocation.
528
+ No `hipDeviceReset` or control of other projects' trainer processes.
529
+
530
+ ### Contrast is the native signal
531
+
532
+ A printed `contrast=0.00000` (or a slot column of `0.00000`) can reflect no
533
+ amp-gated edges, a small rounded value, or cancellation, not a perfect model.
534
+ It is a heartbeat
535
+ (mean \(\Delta\cos\) on gated edges), **not** accuracy, and — as of
536
+ 2026-09-11 — is understood to have been dominated by an artifact for every
537
+ slot prior to the branch fix. Post-fix, it is the same statistic computed
538
+ on a rule with verified nonzero SNR; still not a substitute for the
539
+ clamped-pair / settle-based accuracy checks.
540
+
541
+ ### Live rotation
542
+
543
+ Each loop tries, in order: **mix** (tailed hard file) → **hard**
544
+ (train + SFT wrap) → **SDS1 distill** → **NYSA compiler** → **NYSV
545
+ speech**.
546
+
547
+ October training requires all five steps, using **live GPU Qwen3-4B
548
+ projected probabilities** in `qwen_direct` instead of the historical
549
+ static SDS1 file. A complete rotation advances five native updates.
550
+ Separate context-head updates do not count as extra native steps.
551
+
552
+ HIP holds NYSA/NYSV as `FILE*` for the life of the process. Slot updates
553
+ must be **atomic `mv`** onto those paths; a truncate/scp onto an open
554
+ slot file will crash the trainer.
555
+
556
+ ---
557
+
558
+ ## Sparse checkpoint layout
559
+
560
+ Little-endian. **Reject any file whose size is not 20,640,000,016.**
561
+ Current saves are append-only immutable snapshots with partial-write
562
+ handling and exclusive final publication. Resume only from a complete
563
+ file and, for context continuation, a validated companion bundle. Native
564
+ layout is unchanged by context integration or the 2026-09-11 fix — `\omega` in the file is
565
+ the same value it always was; scaling happens only inside the integrator
566
+ at load time via `--omega-scale`.
567
+
568
+ | Field | Type | Count | Notes |
569
+ |---|---|---|---|
570
+ | `N` | `uint32` | 1 | 5,000,000 |
571
+ | `k` | `uint32` | 1 | 512 |
572
+ | `theta` | `float64` | \(N\) | phase |
573
+ | `amp` | `float64` | \(N\) | 0 after a finished step |
574
+ | `omega` | `float64` | \(N\) | natural frequency, stored value unchanged by `--omega-scale` |
575
+ | `K_row_offsets` | `uint64` | \(N+1\) | CSR; row \(i\) is \([i k, (i+1)k)\) |
576
+ | `K_col_indices` | `uint32` | \(N \cdot k\) | ring + stubs; rewired at 925k (compiler/speech) and 2026-09-10 (speech reset) |
577
+ | `K_values` | `float32` | \(N \cdot k\) | learned couplings, clamped to \(\lvert K \rvert \le 3\) as of 2026-09-11 |
578
+
579
+ Offsets:
580
+
581
+ ```
582
+ theta_off = 8
583
+ amp_off = 8 + 8N
584
+ omega_off = 8 + 16N
585
+ row_off = 8 + 24N
586
+ col_off = row_off + 8(N+1)
587
+ val_off = col_off + 4 N k
588
+ end = val_off + 4 N k = 20,640,000,016
589
+ ```
590
+
591
+ Init topology (this lineage): for each row, \(k-16\) local ring
592
+ `(i - k/2 + j) mod N`, last 16 random stubs. Training updates **values**
593
+ on those edges. At **925k**, a CPU pass replaced the weakest non-protected
594
+ column in some rows so vocab ↔ compiler/speech IDs share an edge. At
595
+ **2026-09-10**, the speech band's edges were additionally force-reset to
596
+ the wiring seed value once, to clear damage accumulated before the silence/
597
+ negative-collision fixes existed. \(k\) is still 512 throughout.
598
+
599
+ Filename: `gpu_eqprop_<steps>_<YYYYMMDD>_<HHMMSS>.bin` (UTC). The trainer
600
+ parses `steps` from the name (or symlink target) for `--resume-steps`.
601
+
602
+ ---
603
+
604
+ ## Tokenizer (do not retrain)
605
+
606
+ [`vocab_sparse.json`](https://huggingface.co/datasets/YNSScarSaiyan/nys-corpus)
607
+ (~107 MB). `HierarchicalTokenizer`: bytes 0–255, then greedy 4-gram
608
+ phrases. The frozen **space-join** encode bug is part of the ID space —
609
+ do not "fix" it or IDs will not match \(K\).
610
+
611
+ The October adapter's frozen vocabulary SHA256 is
612
+ `23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e`.
613
+ Retrieved-input boundary corrections do not change the tokenizer. NYS
614
+ capacity counts these IDs, not interchangeable Qwen tokenizer tokens.
615
+
616
+ IDs above vocab and below \(N\) are reserved bands (compiler / voice
617
+ entropy / speech / dump). They are first-class oscillators, not a second
618
+ model.
619
+
620
+ ---
621
+
622
+ ## Training data
623
+
624
+ Corpora live on
625
+ [`YNSScarSaiyan/nys-corpus`](https://huggingface.co/datasets/YNSScarSaiyan/nys-corpus)
626
+ unless noted. Slot files also upload under `slots/` on this repo.
627
+
628
+ ### Slot 1 — language
629
+
630
+ | File | Layout | Records |
631
+ |---|---|---|
632
+ | `train_corpus_sparse.bin` | hard: `uint32 nrec` + (`uint16 len` + `uint32 toks[len]`), last = target | 255,435 |
633
+ | `sft_corpus_sparse.bin` | same hard layout | 50,000 |
634
+ | `mix_sft_sparse.bin` | same hard layout (Rust `sparse_convert` from JSONL) | 2,050,000 |
635
+ | `distill_corpus_sparse.bin` | SDS1: prompt + soft id/prob | 40,000 |
636
+
637
+ The static distillation row is **historical**: that fixture is absent
638
+ from the restored bundle; no accuracy on it is claimed. The current run
639
+ validates both hard corpora and a corrected 2,050,000-record mixture with
640
+ all nine sources: `agent_flan`, `glaive`, `hermes_fc`, `openhermes`, `orca`,
641
+ `toolace`, `tulu3`, `ultrachat`, `xlam`. Mix SHA256:
642
+ `425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9`.
643
+
644
+ Live teacher prompts use
645
+ [`HuggingFaceH4/ultrachat_200k`](https://huggingface.co/datasets/HuggingFaceH4/ultrachat_200k),
646
+ `train_sft`, revision `8049631c405ae6576f93f445c6b8166f76f5505a`.
647
+ Prompt repetition is enabled. Upstream model/data licenses and terms
648
+ apply separately; this repository's MIT metadata does not relicense
649
+ third-party teachers or training material.
650
+
651
+ ### Slot 2 — compiler physics (`NYSA`)
652
+
653
+ **412** records: I/O triples (kind 1: 385), gold bytes (11), structural bit
654
+ maps (5), traces (11). HIP nudges bit nodes at `3_800_000 + i` with
655
+ \(+\beta\) (spin 1) or \(-\beta\) (spin 0). All 11 gadgets share the
656
+ **same** 64 oscillators — confirmed via settle-based conditioning probe to
657
+ be a genuine unresolved conflict, not a readout artifact.
658
+
659
+ ### Slot 4 — speech (`NYSV` + `NYVC`)
660
+
661
+ | File | Records | Role |
662
+ |---|---|---|
663
+ | `speech_slot.nysv` | 6 | Seed utterances |
664
+ | `speech_align.nysv` | 305 | Aligned text ↔ frame IDs (≤64 frames / rec) |
665
+ | `speech_vocoder.nyvc` | 256 atoms × 80 samples @ 8 kHz | VQ table, rebuilt 2026-09-10 (255/256 atoms trained, was 21/256) |
666
+
667
+ Frame IDs = `4_300_000 + t*256 + code`. `ALIGN_MAX_FRAMES=64` keeps
668
+ \(|S|\) under the 256 cap. As of 2026-09-10, silence-coded frames are
669
+ excluded from both nudge targets and negative examples; negatives are
670
+ checked against a corpus-wide gold set to guarantee zero collision with
671
+ another record's target.
672
+
673
+ ### Slot 5 — dump (`NYSD`)
674
+
675
+ Reserved envelope + inbox. **The trainer does not EqProp NYSD.**
676
+
677
+ ---
678
+
679
+ ## Historical sidecar (CPU, after every complete snap)
680
+
681
+ `sidecar.py` mmaps the checkpoint. Does not run the HIP kernel. Its readouts
682
+ (chat KOPG, crystal Hamming, speech `frame_match`) score **stored theta
683
+ directly** — which, independent of the SNR issue, is the value a node was
684
+ left at by whichever record touched it last, not a response to a live
685
+ prompt. A separate read-only inference mode (`eqprop_gpu --settle`: clamp a
686
+ prompt, run free RK4, read where target nodes land, no training-state
687
+ mutation) exists and was used for the compiler conditioning check above; it
688
+ is not yet wired into the routine sidecar quiz path.
689
+
690
+ Outputs (also uploaded here):
691
+
692
+ ```
693
+ sidecar/gpu_eqprop_<step>_<utc>/
694
+ train_acc.json # clamped-pair acc (language / compiler / speech)
695
+ eval.json # train_acc + chat + free Hamming + speech lock + codec
696
+ ...
697
+ ```
698
+
699
+ This is the historical sidecar path. October CPU reporters also produce
700
+ fixed-fixture language/speech/compiler diagnostics and frozen generalization
701
+ controls, with readout/contamination caveats; they do not evaluate context.
702
+ Current retention requires size **and SHA256**, paired native/head files,
703
+ pinned/current remote verification, and **two** newest local snapshots.
704
+ A size-only remote listing is insufficient deletion authority.
705
+
706
+ ---
707
+
708
+ ## How to load (sparse)
709
+
710
+ ```python
711
+ import os, struct, mmap
712
+
713
+ FULL = 20_640_000_016
714
+ path = "gpu_eqprop_1350000_20260911_071500.bin" # use the newest on this repo
715
+ assert os.path.getsize(path) == FULL
716
+
717
+ with open(path, "rb") as f:
718
+ mm = mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ)
719
+ N, k = struct.unpack_from("<II", mm, 0)
720
+ assert N == 5_000_000 and k == 512
721
+ ```
722
+
723
+ Historical native-only train / resume example (HIP, MI300X / gfx942),
724
+ **not** the current full-curriculum GPU-teacher/context continuation.
725
+ For the current recipe use the pinned supervisor and paired recovery
726
+ metadata above. **Never `--init`
727
+ unless you intend to throw this lineage away.**
728
+
729
+ ```bash
730
+ hipcc -O3 -std=c++17 --offload-arch=gfx942 -o eqprop_gpu eqprop_gpu.hip
731
+ ./eqprop_gpu \
732
+ --data-dir ./data --live --live-mix ./data/mix_sft_sparse.bin \
733
+ --ckpt ./gpu_eqprop_1350000_20260911_071500.bin \
734
+ --save-every 5000 --save-dir ./ckpts \
735
+ --vram-limit-gb 77 --device 0 --lr 0.05 --beta 1.5 \
736
+ --omega-scale 0.01
737
+ ```
738
+
739
+ Omit `--omega-scale` (or pass `1`) to reproduce the pre-2026-09-11 dynamics
740
+ exactly — the checkpoint format and stored values are identical either way.
741
+
742
+ ---
743
+
744
+ ## Files in this repo
745
+
746
+ - `gpu_eqprop_<step>_<utc>.bin` — sparse snapshots every 5,000 steps. Each
747
+ file is the full substrate, not a delta. Reject any file that is not
748
+ 20,640,000,016 B.
749
+ - `sidecar/gpu_eqprop_*/` — CPU readouts.
750
+ - `slots/` — NYSA / NYSV / NYVC / NYSD copies used by this run.
751
+ - `nys_sft_final.bin` — **legacy** 65,536 dense SFT (see table above).
752
+ - `PROGRAM_CHARTER.md` — cross-program charter.
753
+ - `lineages/<run-id>/checkpoints/` — continuation snapshots, upload JSONs,
754
+ and required `.bin.context/` bundles for context-enabled runs.
755
+ - `lineages/<run-id>/recovery/` — reviewed source and launch/source metadata.
756
+ - `documentation/20261009-evidence.json` — compact dated evidence for this update.
757
+
758
+ Select within the documented production lineage and verify the record,
759
+ not the largest filename globally: historical and experimental lineages
760
+ coexist. Pin immutable HF revisions. More steps do not imply higher accuracy.
761
+
762
+ ---
763
+
764
+ ## Hardware and stack
765
+
766
+ | Item | This public GPU run |
767
+ |---|---|
768
+ | GPU | Current MI325X / HIP; historical MI300X; native kernel uses one block of 256 threads |
769
+ | Compile | `hipcc`, C++17, `gfx942`; native graph is not PyTorch/JAX; separate Qwen teacher uses PyTorch/Transformers |
770
+ | HBM | ~20.64 GB native state; current native allocation cap 28 GiB, teacher budget 12 GiB, shared-host reserve checks |
771
+ | Context backends | Native CPU, CUDA, HIP, Intel SYCL; production uses HIP; Intel GPU runtime unverified |
772
+ | Tokenizer | Frozen hierarchical phrase vocab (~3.66M IDs) |
773
+ | Author | Dakuwon Moody (`YNSScarSaiyan`) |
774
+
775
+ ---
776
+
777
+ ## Limitations (read this)
778
+
779
+ - **Historical accuracy caveat:** every accuracy number from before
780
+ 2026-09-11 was produced under a
781
+ contrastive update with a measured signal-to-noise ratio of roughly
782
+ 1:6,000. Treat those numbers as documentation of the failure mode, not as
783
+ evidence about the organism's ceiling in either direction.
784
+ - **Runtime correctness is not capability.** Corrected native training
785
+ and context integration have actual HIP verification; broad generalization
786
+ and a context-on/context-off language-quality gain remain unproven.
787
+ - **16M means indexed memory capacity**, not dense attention, a native
788
+ active set of that size, a 16M-parameter head in this run, or demonstrated
789
+ 16M-token language understanding. Sparse retrieval can miss evidence.
790
+ - **Native topology still limits readout.** In the fixed language sample,
791
+ 22/96 targets lack direct prompt-to-gold edges. Context does not create
792
+ native edges. The A100 topology-repair candidate was not promoted.
793
+ - **Readout/data changes matter.** Stored phase, weight only, canonical CPU
794
+ settle, clamped fixtures, and context-assisted readout differ. Restored
795
+ corpus/codebook changes prevent simple historical accuracy comparisons.
796
+ - **Compiler has a second, independent problem**: 11 gadgets share one
797
+ 64-bit register, confirmed via a settle-based conditioning probe to be a
798
+ genuine write conflict, not a readout artifact. Fixing the SNR does not
799
+ by itself fix this.
800
+ - **Not a production chat model.** No CE, no instruction-eval scores
801
+ claimed here. Contrast ≠ quality grade. KOPG chat ≠ EqProp generation.
802
+ - **This GPU lineage started from random init**, not the completed CPU
803
+ sparse base.
804
+ - **Sparse ≠ dense.** You cannot mmap these bins as a 65,536×65,536 `K`.
805
+ - **Voice (slot 3) is digit-tone crystallize**, not speech synthesis.
806
+ - **Do not retrain `vocab_sparse.json`.**
807
+ - **Do not `--init` this lineage** unless you mean to start over.
808
+ - Research artifact. Architecture-locked to this \(N,k\) and tokenizer.
809
+ - Not validated for high-stakes decisions. No comprehensive safety/bias
810
+ assessment, reliable general-purpose generation, or open-ended compiler
811
+ correctness is claimed.
812
+
813
+ ---
814
+
815
+ ## Related
816
+
817
+ - Engine / trainers: local `NYS/` (HIP `gpu_eqprop/`, CPU
818
+ `src/sparse_trainer.cpp`, ASM `src/{execute,audio,coupling,quench}.asm`)
819
+ - Public source: [deployed GitHub commit](https://github.com/dakuwonmoody-lab/nys/tree/df0024a30542b57d7946ea39fd453ace6180ee58/NYS)
820
+ - Backend CI: [October 9 passing workflow](https://github.com/dakuwonmoody-lab/nys/actions/runs/37933671238)
821
+ - Corpora: [`YNSScarSaiyan/nys-corpus`](https://huggingface.co/datasets/YNSScarSaiyan/nys-corpus)
822
+ - Private / other snaps: [`YNSScarSaiyan/nys-checkpoints`](https://huggingface.co/YNSScarSaiyan/nys-checkpoints)
823
+ - Charter: `PROGRAM_CHARTER.md` in this repo
824
+
825
+ *Updated 2026-10-09 — live MI325X / GPU Qwen3-4B continuation, 16M-capacity
826
+ sparse context integration, preserved learned head, paired HF recovery,
827
+ backend validation boundaries, current graph-only metrics, and unpromoted
828
+ A100 repair A/B. September diagnosis/results remain historical records.
829
+ No broad generalization or 16M-token language-understanding claim is made.*
documentation/20261009-evidence.json ADDED
@@ -0,0 +1,173 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": 1,
3
+ "snapshot_utc": "2026-10-09T15:29:56.677849+00:00",
4
+ "scope": "Dated model-card evidence; not a live dashboard or proof of broad generalization",
5
+ "source": {
6
+ "github_repo": "dakuwonmoody-lab/nys",
7
+ "deployed_commit": "df0024a30542b57d7946ea39fd453ace6180ee58",
8
+ "ci_url": "https://github.com/dakuwonmoody-lab/nys/actions/runs/37933671238",
9
+ "ci_passed": ["Linux CPU", "Windows CPU", "CUDA compile", "HIP compile", "Intel SYCL compile"],
10
+ "intel_gpu_runtime_verified": false,
11
+ "native_hip_binary_sha256": "4b9dacb4592e56d29647c240094ed298ca93a8b3fbb71e6ced970eaf5b0f6572",
12
+ "context_hip_library_sha256": "285034c2de4bb898b5cb6a4112c3e3e61e1a8cf3e761eec7e362186bdf7a703c"
13
+ },
14
+ "run": {
15
+ "run_id": "qwen-mi325x-4b-20261009-r4",
16
+ "running_at_snapshot": true,
17
+ "hardware": "AMD Instinct MI325X",
18
+ "native_backend": "HIP",
19
+ "start_step": 2190210,
20
+ "observed_step": 2213680,
21
+ "target_absolute_step": 7075000,
22
+ "additional_steps_at_launch": 4884790,
23
+ "save_every_native_steps": 5000,
24
+ "curriculum": ["mix", "hard", "qwen_direct", "compiler", "speech"],
25
+ "teacher_model": "Qwen/Qwen3-4B",
26
+ "teacher_revision": "1cfa9a7208912126459214e8b04321603b3df60c",
27
+ "teacher_on_gpu": true,
28
+ "teacher_dataset": "HuggingFaceH4/ultrachat_200k",
29
+ "teacher_split": "train_sft",
30
+ "teacher_dataset_revision": "8049631c405ae6576f93f445c6b8166f76f5505a",
31
+ "vocab_sha256": "23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e",
32
+ "corrected_mix_sha256": "425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9"
33
+ },
34
+ "context": {
35
+ "capacity_nys_token_ids": 16777216,
36
+ "hash_features": 1048576,
37
+ "legacy_features": 560,
38
+ "parameters": 1049136,
39
+ "backend": "HIP",
40
+ "records": 4954,
41
+ "committed_head_updates": 4954,
42
+ "positive_updates": 4897,
43
+ "restored_previous_updates": 260,
44
+ "new_head_updates_in_r4": 4694,
45
+ "retrieved_records": 4954,
46
+ "retrieved_token_ids_injected": 60958,
47
+ "represented_source_ids_processed": 449537,
48
+ "budget_abstentions": 0,
49
+ "retrieved_ids_per_prompt_max": 12,
50
+ "recent_ids_per_prompt_max": 96,
51
+ "framed_native_prompt_max_ids": 127,
52
+ "cross_row_memory": false,
53
+ "retrieval_before_teacher_targets": true,
54
+ "head_commit_requires_native_rotation_ack": true,
55
+ "native_graph_topology_modified_by_context": false,
56
+ "automatic_copy_answer_override": false,
57
+ "broad_long_context_quality_demonstrated": false
58
+ },
59
+ "latest_verified_backup_at_snapshot": {
60
+ "repo": "YNSScarSaiyan/nys-sft-public",
61
+ "step": 2210000,
62
+ "checkpoint": "lineages/qwen-mi325x-4b-20261009-r4/checkpoints/gpu_eqprop_2210000_20261009_150930.bin",
63
+ "immutable_revision": "9cf0a6f80a49a899d5c6583445dd1a7b9493ab14",
64
+ "native_bytes": 20640000016,
65
+ "native_sha256": "74494f292bf836582b723004d6fcee72af3b9462312079bce0d6069de28c4369",
66
+ "native_and_context_verified_together": true,
67
+ "companion_files": {
68
+ "model.json": {"bytes": 1051, "sha256": "3765bece9bc97761d0ecf02f84a1e157e6ee143eadc8480e0f8db5a7a3364894"},
69
+ "weights.npy": {"bytes": 4196672, "sha256": "ce9ea63e3827c7f9792e84e02f28b7e67f4c7299ca516fa67a15e3256c8a5143"},
70
+ "pair.json": {"bytes": 1023, "sha256": "35a2667fc2ed67899f07cb5b959ce0304842e0eb866cd2756638e616aafe4827"}
71
+ }
72
+ },
73
+ "production_head_restore_check": {
74
+ "native_step": 2190215,
75
+ "context_updates_saved": 261,
76
+ "live_deployed": true,
77
+ "backend": "HIP",
78
+ "nonzero_weights": 712,
79
+ "max_abs_weight": 0.006360475905239582,
80
+ "bitwise_head_restore": true,
81
+ "hf_pair_revision": "42408ae032f6d4050dae477c026a53ed917a4ab2"
82
+ },
83
+ "isolated_hip_integration_check": {
84
+ "native_nodes": 4096,
85
+ "native_k": 32,
86
+ "native_steps": 10,
87
+ "complete_five_step_rotations": 2,
88
+ "actual_native_hip": true,
89
+ "actual_context_hip": true,
90
+ "longer_than_old_limit_input_entered_native_stream": true,
91
+ "query_before_targets": true,
92
+ "paired_save_and_bitwise_head_resume": true,
93
+ "independent_rows_isolated": true,
94
+ "original_seed_unchanged": true
95
+ },
96
+ "graph_only_report": {
97
+ "step": 2210000,
98
+ "checked_utc": "2026-10-09T15:11:58.576523+00:00",
99
+ "cases_sha256": "71c34c27405a498278622d2f24562c19020db7d01c03aad86061f43978448fc4",
100
+ "cases": 896,
101
+ "cpu_only": true,
102
+ "training_updates": 0,
103
+ "uses_context_adapter": false,
104
+ "metrics": {
105
+ "stored_phase": {
106
+ "external": {"records": 512, "correct": 0, "accuracy": 0.0},
107
+ "arithmetic": {"records": 128, "correct": 0, "accuracy": 0.0},
108
+ "binding": {"records": 256, "correct": 0, "pairs": 128, "complete_correct_pairs": 0, "constrained_choice_accuracy": 0.5, "identical_prediction_pairs": 128}
109
+ },
110
+ "weight_only": {
111
+ "external": {"records": 512, "correct": 2, "accuracy": 0.00390625},
112
+ "arithmetic": {"records": 128, "correct": 0, "accuracy": 0.0},
113
+ "binding": {"records": 256, "correct": 0, "pairs": 128, "complete_correct_pairs": 0, "constrained_choice_accuracy": 0.5, "identical_prediction_pairs": 128}
114
+ },
115
+ "canonical_settle": {
116
+ "external": {"records": 512, "correct": 2, "accuracy": 0.00390625},
117
+ "arithmetic": {"records": 128, "correct": 0, "accuracy": 0.0},
118
+ "binding": {"records": 256, "correct": 0, "pairs": 128, "complete_correct_pairs": 0, "constrained_choice_accuracy": 0.5, "identical_prediction_pairs": 128}
119
+ }
120
+ },
121
+ "fixed_training_fixtures": {
122
+ "language_mix_records": 96,
123
+ "language_mix_clamped_correct": 5,
124
+ "language_mix_clamped_top1": 0.052083333333333336,
125
+ "language_mix_directly_reachable_gold": 74,
126
+ "language_mix_missing_direct_gold": 22,
127
+ "compiler_clamped_free_bit_accuracy": 0.8038277511961722,
128
+ "compiler_cpu_settled_free_bit_accuracy": 0.6172248803827751,
129
+ "compiler_cpu_settle_steps": 25,
130
+ "speech_gold_edge_weight_lead": 0.20093635986587288,
131
+ "speech_gold_strongest_wired_pool_fraction": 0.1680550525172039,
132
+ "speech_untrained_pool_reference": 0.1376490080665108
133
+ },
134
+ "broad_generalization_proven": false,
135
+ "limitations": [
136
+ "Reused frozen September 30 regression cases, not a fresh sealed benchmark",
137
+ "Historical and semantic contamination cannot be excluded",
138
+ "Canonical settle is an experimental CPU readout, not the live native reader",
139
+ "Next-token matching is not conversational accuracy",
140
+ "These reporters do not measure the context adapter",
141
+ "Recovered corpus/codebook changes prevent simple September comparisons",
142
+ "Compiler fixed-bit and speech wired-pool diagnostics are not broad capabilities"
143
+ ]
144
+ },
145
+ "a100_topology_ab": {
146
+ "hardware": "A100-SXM4-80GB",
147
+ "source_native_step": 2120000,
148
+ "source_native_sha256": "5f922bf29d82e13cccb37eb642216f5e7112fb0869276739d70ff2a1162ed4ba",
149
+ "source_hf_revision": "be84d1ea03b422721e956b336b5e00c6715ff96d",
150
+ "cuda_updates_per_arm": 1440,
151
+ "matched_language_records": 288,
152
+ "epochs": 5,
153
+ "changed_edges": 54546,
154
+ "added_edges_changed_by_native_training": 381,
155
+ "direct_sampled_coverage_before": 74,
156
+ "direct_sampled_coverage_after": 78,
157
+ "direct_sampled_records": 96,
158
+ "canonical_external": {"records": 512, "source_correct": 3, "control_correct": 3, "repaired_correct": 3},
159
+ "canonical_arithmetic": {"records": 128, "source_correct": 1, "control_correct": 1, "repaired_correct": 1},
160
+ "canonical_binding": {"pairs": 128, "source_complete_pairs": 0, "control_complete_pairs": 0, "repaired_complete_pairs": 0},
161
+ "repaired_vs_control_helpful_predictions": 0,
162
+ "repaired_vs_control_harmful_predictions": 0,
163
+ "candidate_deployed": false,
164
+ "limitation": "Bounded language-only trial, not the full live curriculum or the integrated context adapter"
165
+ },
166
+ "cuda_capacity_checks": {
167
+ "context_tests_passed": 28,
168
+ "tested_capacity_nys_ids": [1048576, 16777216],
169
+ "hash_features_in_probes": 1048576,
170
+ "limitation": "Handcrafted relation weights and punctuation distractors; not learned language recall, broad generalization, or a controlled speed comparison"
171
+ },
172
+ "documentation_change_scope": "README.md and this compact public evidence file only; no trainer, checkpoint, billing, or credential changes"
173
+ }
lineages/corrected-20260930-v2/LATEST.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"path": "lineages/corrected-20260930-v2/gpu_eqprop_2075000_20260930_214548.bin", "step": 2075000, "bytes": 20640000016, "sha256": "96673f09132d5a0bb2adf5664f4e3839567bbb6371a07b16b46714f3967e9b33"}
lineages/corrected-20260930-v2/curriculum.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "verified": true,
3
+ "rotation": [
4
+ "mix",
5
+ "hard",
6
+ "qwen_direct",
7
+ "compiler",
8
+ "speech"
9
+ ],
10
+ "records": {
11
+ "mix": 2050000,
12
+ "train_corpus_sparse.bin": 255435,
13
+ "sft_corpus_sparse.bin": 50000,
14
+ "speech_align.nysv": 305,
15
+ "speech_slot.nysv": 6,
16
+ "compiler_physics.nysa": 412
17
+ },
18
+ "mix_sha256": "425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9",
19
+ "mix_sources": [
20
+ "agent_flan",
21
+ "glaive",
22
+ "hermes_fc",
23
+ "openhermes",
24
+ "orca",
25
+ "toolace",
26
+ "tulu3",
27
+ "ultrachat",
28
+ "xlam"
29
+ ],
30
+ "vocab_sha256": "23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e"
31
+ }
lineages/corrected-20260930-v2/lineage.json ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "lineage": "corrected-20260930-v2",
3
+ "parent_root": "/data/nys-l40s",
4
+ "quarantined_parent_runs": true,
5
+ "source_step": 1915000,
6
+ "source_revision": "6f1f1e314e6d680f0e1541ddc6b1f6590b46cf40",
7
+ "status": "VALIDATED",
8
+ "full_curriculum_sha256": "e1e79d842ec60f3f646a563850a683e49519c6892807ee3a70a3a60047c3de81",
9
+ "mix_sha256": "425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9",
10
+ "superseded_lineage": "corrected-20260930"
11
+ }
lineages/corrected-20260930-v2/mix_manifest.json ADDED
@@ -0,0 +1,187 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "verified": true,
3
+ "records": 2050000,
4
+ "bytes": 810830240,
5
+ "sha256": "425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9",
6
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649",
7
+ "vocab_sha256": "23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e",
8
+ "sources": [
9
+ {
10
+ "name": "tulu3",
11
+ "repo": "allenai/tulu-3-sft-mixture",
12
+ "revision": "b14afda60f1bbebe55d5d2fa1e4df5042f97f8be",
13
+ "config": null,
14
+ "records": 273333,
15
+ "target_records": 273333,
16
+ "bytes": 74691934,
17
+ "sha256": "a6791c7f5ebfbf73982f16200c57da1f986ec82835e26d310337723811488efe",
18
+ "fed_conversations": {
19
+ "train": 14280
20
+ },
21
+ "normalized_input_sha256": "43a1ceebba233ccc1295c0e6e4c38df03ca107aeca88ba76aa8ebdff3db05739",
22
+ "max_per_conversation": 25,
23
+ "skipped_no_assistant_targets": {},
24
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
25
+ },
26
+ {
27
+ "name": "openhermes",
28
+ "repo": "teknium/OpenHermes-2.5",
29
+ "revision": "b82037821055c377bed0d495e72e46de3bc72e84",
30
+ "config": null,
31
+ "records": 205000,
32
+ "target_records": 205000,
33
+ "bytes": 37265984,
34
+ "sha256": "a14995d3982265093dade0e5456bf4e807adc1606502a008e010bf1cae91a86e",
35
+ "fed_conversations": {
36
+ "train": 9472
37
+ },
38
+ "normalized_input_sha256": "0c8290012cd577e5fa0e719b3e1a109d0be0e7352a180fcc87d558123f96b55d",
39
+ "max_per_conversation": 25,
40
+ "skipped_no_assistant_targets": {},
41
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
42
+ },
43
+ {
44
+ "name": "ultrachat",
45
+ "repo": "HuggingFaceH4/ultrachat_200k",
46
+ "revision": "8049631c405ae6576f93f445c6b8166f76f5505a",
47
+ "config": null,
48
+ "records": 205000,
49
+ "target_records": 205000,
50
+ "bytes": 54089028,
51
+ "sha256": "ca6fe7a2c74cf1612a51d6c8b425c4869a5147f56647581a318b148a1660fca0",
52
+ "fed_conversations": {
53
+ "train_sft": 8211
54
+ },
55
+ "normalized_input_sha256": "c5e3c60cfe787c0ee6dc15a934e2d9f0de46a866faafb8d871464f69972301de",
56
+ "max_per_conversation": 25,
57
+ "skipped_no_assistant_targets": {},
58
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
59
+ },
60
+ {
61
+ "name": "xlam",
62
+ "repo": "Salesforce/xlam-function-calling-60k",
63
+ "revision": "26d14ebfe18b1f7b524bd39b404b50af5dc97866",
64
+ "config": "dataset",
65
+ "records": 273333,
66
+ "target_records": 273333,
67
+ "bytes": 133978174,
68
+ "sha256": "35fa9266bf1f5080dd52798145a68fcfaae141e5ae7060e9049d8b5518dc9f83",
69
+ "fed_conversations": {
70
+ "train": 11216
71
+ },
72
+ "normalized_input_sha256": "4d384db283209ff38aecbd1fc07498e9d2e9158aa5c151e488a6e3bcfa7dba73",
73
+ "max_per_conversation": 25,
74
+ "skipped_no_assistant_targets": {},
75
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
76
+ },
77
+ {
78
+ "name": "toolace",
79
+ "repo": "Team-ACE/ToolACE",
80
+ "revision": "6bda777c88d21e5a204703c1ee45597a8fa4f734",
81
+ "config": null,
82
+ "records": 205000,
83
+ "target_records": 205000,
84
+ "bytes": 105319644,
85
+ "sha256": "4e2351be623bf5d2cb4e6c72fe5454289dbb467055d8ab5c4ef9f9a50cca8b36",
86
+ "fed_conversations": {
87
+ "train": 9014
88
+ },
89
+ "normalized_input_sha256": "129d36d3541f946be6d6199eaed8ab7a796cdd54ffc45abe735f9837c79e7e3d",
90
+ "max_per_conversation": 25,
91
+ "skipped_no_assistant_targets": {},
92
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
93
+ },
94
+ {
95
+ "name": "glaive",
96
+ "repo": "glaiveai/glaive-function-calling-v2",
97
+ "revision": "e7f4b6456019f5d8bcb991ef0dd67d8ff23221ac",
98
+ "config": null,
99
+ "records": 205000,
100
+ "target_records": 205000,
101
+ "bytes": 80593548,
102
+ "sha256": "c1e0f9b1b3e57bea3d726ee282748959aaeefdce5a01dea4d10e980a152ee1fa",
103
+ "fed_conversations": {
104
+ "train": 8320
105
+ },
106
+ "normalized_input_sha256": "9a8a47aca6a49e0054690a267bdc610e9f9ffc2b67ae3782f00aae34921d33b7",
107
+ "max_per_conversation": 25,
108
+ "skipped_no_assistant_targets": {},
109
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
110
+ },
111
+ {
112
+ "name": "hermes_fc",
113
+ "repo": "NousResearch/hermes-function-calling-v1",
114
+ "revision": "dae3e1d28cfbcf4b915c04ea1e072030529b4bda",
115
+ "config": "func_calling_singleturn",
116
+ "records": 47325,
117
+ "target_records": 136666,
118
+ "bytes": 24325054,
119
+ "sha256": "b83105e2de0c9317cbdf064b8a022c0e2e6994b20ee95a9e2aee28fa7a4ea2b3",
120
+ "fed_conversations": {
121
+ "train": 1893
122
+ },
123
+ "normalized_input_sha256": "f9faf972ab53f865c517574dd83540982ec94b3d19024acf941bd627117e09d5",
124
+ "max_per_conversation": 25,
125
+ "skipped_no_assistant_targets": {},
126
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
127
+ },
128
+ {
129
+ "name": "orca",
130
+ "repo": "microsoft/orca-agentinstruct-1M-v1",
131
+ "revision": "86d609183249ff8037eae33d76ebca3af9390ea8",
132
+ "config": null,
133
+ "records": 273333,
134
+ "target_records": 273333,
135
+ "bytes": 115045826,
136
+ "sha256": "b9194dd3c45ac353809a19c851ef5935ef06d568c46943ab021963a4a8adf677",
137
+ "fed_conversations": {
138
+ "analytical_reasoning": 1873,
139
+ "code_": 1873,
140
+ "rag": 1872,
141
+ "follow_up": 1872,
142
+ "fs_cot_flow": 1872,
143
+ "open_domain_qa": 1872
144
+ },
145
+ "normalized_input_sha256": "c4b165c28a5e4ba92304e38d23f2da93cf691eb9eb137b847aa9f6f5592b7180",
146
+ "max_per_conversation": 25,
147
+ "skipped_no_assistant_targets": {},
148
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
149
+ },
150
+ {
151
+ "name": "agent_flan",
152
+ "repo": "internlm/Agent-FLAN",
153
+ "revision": "8b25999e795a58b264fcb51e8746edb2faee9161",
154
+ "config": null,
155
+ "records": 362676,
156
+ "target_records": 362676,
157
+ "bytes": 185521080,
158
+ "sha256": "b760628a29e8f7e50237379d2ba4f4aaf4400b771706347032a80bc032524856",
159
+ "fed_conversations": {
160
+ "agent_instruct_tflan": 1730,
161
+ "agent_instruct_react": 1730,
162
+ "toolbench_tflan_60p_r10r5u7": 5530,
163
+ "toolbench_instruct_j1s1_3k": 5530
164
+ },
165
+ "normalized_input_sha256": "e67470488beec5b7dbc132d818185f9dc2c3fa5d39859dc59e27573e90582fca",
166
+ "max_per_conversation": 25,
167
+ "skipped_no_assistant_targets": {
168
+ "agent_instruct_react": 1,
169
+ "agent_instruct_tflan": 1
170
+ },
171
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
172
+ }
173
+ ],
174
+ "shuffle_seed": 7,
175
+ "first_32000_source_counts": {
176
+ "tulu3": 4287,
177
+ "orca": 4204,
178
+ "agent_flan": 5622,
179
+ "xlam": 4393,
180
+ "ultrachat": 3193,
181
+ "openhermes": 3171,
182
+ "toolace": 3164,
183
+ "glaive": 3219,
184
+ "hermes_fc": 747
185
+ },
186
+ "note": "Rebuilt original nine-source recipe; not the unavailable historical byte-for-byte mix"
187
+ }
lineages/corrected-20260930-v2/recovery/backup-manifest.json ADDED
@@ -0,0 +1,119 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "checkpoint": "lineages/corrected-20260930-v2/gpu_eqprop_2075000_20260930_214548.bin",
3
+ "checkpoint_sha256": "96673f09132d5a0bb2adf5664f4e3839567bbb6371a07b16b46714f3967e9b33",
4
+ "captured_utc": "2026-09-30T21:48:18.852227+00:00",
5
+ "files": {
6
+ "data/corrected-mix/mix_sft_sparse.bin": {
7
+ "bytes": 810830240,
8
+ "sha256": "425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9"
9
+ },
10
+ "data/corrected-mix/mix_manifest.json": {
11
+ "bytes": 7067,
12
+ "sha256": "bf22f42a0fd2fdcbdc23e60821e0614cd1bc5ab69b7c1589efc762c5adc4c624"
13
+ },
14
+ "data/language_baseline/vocab_sparse.json": {
15
+ "bytes": 106816698,
16
+ "sha256": "23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e"
17
+ },
18
+ "data/language_baseline/train_corpus_sparse.bin": {
19
+ "bytes": 66816546,
20
+ "sha256": "a9b901c8c02b35eebe638abc0bc8ae551e359f51370890733ac917f0c8a97916"
21
+ },
22
+ "data/language_baseline/sft_corpus_sparse.bin": {
23
+ "bytes": 13200476,
24
+ "sha256": "91cc73e689d8e8e457221d3a20bb4d552cd9d2e5df1c1ed2f7ad2ce5c796cd60"
25
+ },
26
+ "data/language_baseline/recovery_manifest.json": {
27
+ "bytes": 703,
28
+ "sha256": "aac6d1a44e1037fcac4de6b3a9cc8de1e714c357f7495e9aff4fe6309dd561cb"
29
+ },
30
+ "data/slots/compiler_physics.nysa": {
31
+ "bytes": 48815,
32
+ "sha256": "052fc7dee4b1081392051b9fde2fac6533aa77e82dbea56ac32c27a67ced6cf8"
33
+ },
34
+ "data/slots/speech_align.nysv": {
35
+ "bytes": 1490468,
36
+ "sha256": "fea716002f75cde8a4567d64e28c14e353c3cdb349cd9ec8ecf3c991f52fbf8c"
37
+ },
38
+ "data/slots/speech_slot.nysv": {
39
+ "bytes": 15284,
40
+ "sha256": "2988cc801e2b95d6ebba35d335ec656486bc0fe0b5938d071a536c63879aca0c"
41
+ },
42
+ "data/slots/speech_vocoder.nyvc": {
43
+ "bytes": 20500,
44
+ "sha256": "028d0e759f9721b2110d8527028e7de3d80d5b611401d33db895d48e2f2df4d1"
45
+ },
46
+ "data/slots/recovery_manifest.json": {
47
+ "bytes": 3007,
48
+ "sha256": "b74dc14a623e7a4e1b9b1726d1c162a2e23cc19d373e9f53e1fd9064eb46dc35"
49
+ },
50
+ "gpu_eqprop/build_cuda.sh": {
51
+ "bytes": 287,
52
+ "sha256": "8c2148c696f5f4520b9cfa13bc6e80accc3de5630e449de0197391877db13173"
53
+ },
54
+ "gpu_eqprop/run_cuda_l40s.sh": {
55
+ "bytes": 640,
56
+ "sha256": "7f0360130f5d5f28c22e40310b9f6cec532ee0d8afd2cd1cf3afdd319375f96c"
57
+ },
58
+ "gpu_eqprop/run_qwen_live.py": {
59
+ "bytes": 6718,
60
+ "sha256": "deac34a492afb95687a97200955debb38d8a1f4e2c4bf39e8e6a64315e815ffb"
61
+ },
62
+ "gpu_eqprop/start_qwen_l40s.sh": {
63
+ "bytes": 3840,
64
+ "sha256": "b018e78b46355d022614352b362ee8bdc3bf8e701758cf97b60a7419ef33b0f8"
65
+ },
66
+ "gpu_eqprop/qwen_live.py": {
67
+ "bytes": 12912,
68
+ "sha256": "40ee3d2f4ba916f3d93db3f515a21af1feec3a1572b9d782798711ae7d787cf7"
69
+ },
70
+ "gpu_eqprop/validate_curriculum.py": {
71
+ "bytes": 3744,
72
+ "sha256": "d433f10fc417e2a87154c2c0e9721271d084ccb12f6bcda335c2a995cf79af02"
73
+ },
74
+ "gpu_eqprop/eqprop_gpu.cu": {
75
+ "bytes": 85834,
76
+ "sha256": "79f4648ab7b8e7964fea5d65fd4b8812ffff986daee648c2c182ddc09e58a5c0"
77
+ },
78
+ "gpu_eqprop/distill_stdin.h": {
79
+ "bytes": 4590,
80
+ "sha256": "6ad021d70316d74f29a43359f3436669d57827c84bdd47919e9c907f75c42eb3"
81
+ },
82
+ "gpu_eqprop/cuda_self_test.h": {
83
+ "bytes": 8128,
84
+ "sha256": "96e95af580d1d133cf72c7636f2da52ef0358f4d7e7978c5050029f8f61e0ff2"
85
+ },
86
+ "gpu_eqprop/hf_checkpoint_service.py": {
87
+ "bytes": 22182,
88
+ "sha256": "267d5da011641fb30aeba219afd8a835c1b892420fcac28c0c11557635db294a"
89
+ },
90
+ "gpu_eqprop/start_backup_services.py": {
91
+ "bytes": 1845,
92
+ "sha256": "de96145ac8e0ac90b48a72f9e199b5c7879e29967f9a52d5b5bc28493cd4c56f"
93
+ },
94
+ "gpu_eqprop/live_status.py": {
95
+ "bytes": 4514,
96
+ "sha256": "e9ae43c50ea0ae723f4ba0162c7a951bffccf3317abdd7adae7aa2faadf82662"
97
+ },
98
+ "gpu_eqprop/build_corrected_mix.py": {
99
+ "bytes": 10711,
100
+ "sha256": "a2bcd06abe180432c009d84b6a298b9c6d893747a60f6a16f4eea5977136b27f"
101
+ },
102
+ "hierarchical_tokenizer.py": {
103
+ "bytes": 4963,
104
+ "sha256": "e7bacce12c795901359a57dbe2fc53915aa79f9a9c446ec5725c3e6e3f4a809f"
105
+ },
106
+ "source_checkpoint.json": {
107
+ "bytes": 422,
108
+ "sha256": "bda715ec4a981fd23d2164206b37956f639102fa20cd7c621f93bf81c018a8cc"
109
+ },
110
+ "lineage.json": {
111
+ "bytes": 452,
112
+ "sha256": "1182cc66a6e857ca7ad956a24e3b8e9f922af73ad7c679bf815c870cdfd6ba37"
113
+ },
114
+ "recovery_policy.json": {
115
+ "bytes": 3800,
116
+ "sha256": "551addf5606189ac9d3c0e133773d477971729e092af82c1d6e90a2b3a277cb0"
117
+ }
118
+ }
119
+ }
lineages/corrected-20260930-v2/recovery/data/corrected-mix/mix_manifest.json ADDED
@@ -0,0 +1,187 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "verified": true,
3
+ "records": 2050000,
4
+ "bytes": 810830240,
5
+ "sha256": "425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9",
6
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649",
7
+ "vocab_sha256": "23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e",
8
+ "sources": [
9
+ {
10
+ "name": "tulu3",
11
+ "repo": "allenai/tulu-3-sft-mixture",
12
+ "revision": "b14afda60f1bbebe55d5d2fa1e4df5042f97f8be",
13
+ "config": null,
14
+ "records": 273333,
15
+ "target_records": 273333,
16
+ "bytes": 74691934,
17
+ "sha256": "a6791c7f5ebfbf73982f16200c57da1f986ec82835e26d310337723811488efe",
18
+ "fed_conversations": {
19
+ "train": 14280
20
+ },
21
+ "normalized_input_sha256": "43a1ceebba233ccc1295c0e6e4c38df03ca107aeca88ba76aa8ebdff3db05739",
22
+ "max_per_conversation": 25,
23
+ "skipped_no_assistant_targets": {},
24
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
25
+ },
26
+ {
27
+ "name": "openhermes",
28
+ "repo": "teknium/OpenHermes-2.5",
29
+ "revision": "b82037821055c377bed0d495e72e46de3bc72e84",
30
+ "config": null,
31
+ "records": 205000,
32
+ "target_records": 205000,
33
+ "bytes": 37265984,
34
+ "sha256": "a14995d3982265093dade0e5456bf4e807adc1606502a008e010bf1cae91a86e",
35
+ "fed_conversations": {
36
+ "train": 9472
37
+ },
38
+ "normalized_input_sha256": "0c8290012cd577e5fa0e719b3e1a109d0be0e7352a180fcc87d558123f96b55d",
39
+ "max_per_conversation": 25,
40
+ "skipped_no_assistant_targets": {},
41
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
42
+ },
43
+ {
44
+ "name": "ultrachat",
45
+ "repo": "HuggingFaceH4/ultrachat_200k",
46
+ "revision": "8049631c405ae6576f93f445c6b8166f76f5505a",
47
+ "config": null,
48
+ "records": 205000,
49
+ "target_records": 205000,
50
+ "bytes": 54089028,
51
+ "sha256": "ca6fe7a2c74cf1612a51d6c8b425c4869a5147f56647581a318b148a1660fca0",
52
+ "fed_conversations": {
53
+ "train_sft": 8211
54
+ },
55
+ "normalized_input_sha256": "c5e3c60cfe787c0ee6dc15a934e2d9f0de46a866faafb8d871464f69972301de",
56
+ "max_per_conversation": 25,
57
+ "skipped_no_assistant_targets": {},
58
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
59
+ },
60
+ {
61
+ "name": "xlam",
62
+ "repo": "Salesforce/xlam-function-calling-60k",
63
+ "revision": "26d14ebfe18b1f7b524bd39b404b50af5dc97866",
64
+ "config": "dataset",
65
+ "records": 273333,
66
+ "target_records": 273333,
67
+ "bytes": 133978174,
68
+ "sha256": "35fa9266bf1f5080dd52798145a68fcfaae141e5ae7060e9049d8b5518dc9f83",
69
+ "fed_conversations": {
70
+ "train": 11216
71
+ },
72
+ "normalized_input_sha256": "4d384db283209ff38aecbd1fc07498e9d2e9158aa5c151e488a6e3bcfa7dba73",
73
+ "max_per_conversation": 25,
74
+ "skipped_no_assistant_targets": {},
75
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
76
+ },
77
+ {
78
+ "name": "toolace",
79
+ "repo": "Team-ACE/ToolACE",
80
+ "revision": "6bda777c88d21e5a204703c1ee45597a8fa4f734",
81
+ "config": null,
82
+ "records": 205000,
83
+ "target_records": 205000,
84
+ "bytes": 105319644,
85
+ "sha256": "4e2351be623bf5d2cb4e6c72fe5454289dbb467055d8ab5c4ef9f9a50cca8b36",
86
+ "fed_conversations": {
87
+ "train": 9014
88
+ },
89
+ "normalized_input_sha256": "129d36d3541f946be6d6199eaed8ab7a796cdd54ffc45abe735f9837c79e7e3d",
90
+ "max_per_conversation": 25,
91
+ "skipped_no_assistant_targets": {},
92
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
93
+ },
94
+ {
95
+ "name": "glaive",
96
+ "repo": "glaiveai/glaive-function-calling-v2",
97
+ "revision": "e7f4b6456019f5d8bcb991ef0dd67d8ff23221ac",
98
+ "config": null,
99
+ "records": 205000,
100
+ "target_records": 205000,
101
+ "bytes": 80593548,
102
+ "sha256": "c1e0f9b1b3e57bea3d726ee282748959aaeefdce5a01dea4d10e980a152ee1fa",
103
+ "fed_conversations": {
104
+ "train": 8320
105
+ },
106
+ "normalized_input_sha256": "9a8a47aca6a49e0054690a267bdc610e9f9ffc2b67ae3782f00aae34921d33b7",
107
+ "max_per_conversation": 25,
108
+ "skipped_no_assistant_targets": {},
109
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
110
+ },
111
+ {
112
+ "name": "hermes_fc",
113
+ "repo": "NousResearch/hermes-function-calling-v1",
114
+ "revision": "dae3e1d28cfbcf4b915c04ea1e072030529b4bda",
115
+ "config": "func_calling_singleturn",
116
+ "records": 47325,
117
+ "target_records": 136666,
118
+ "bytes": 24325054,
119
+ "sha256": "b83105e2de0c9317cbdf064b8a022c0e2e6994b20ee95a9e2aee28fa7a4ea2b3",
120
+ "fed_conversations": {
121
+ "train": 1893
122
+ },
123
+ "normalized_input_sha256": "f9faf972ab53f865c517574dd83540982ec94b3d19024acf941bd627117e09d5",
124
+ "max_per_conversation": 25,
125
+ "skipped_no_assistant_targets": {},
126
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
127
+ },
128
+ {
129
+ "name": "orca",
130
+ "repo": "microsoft/orca-agentinstruct-1M-v1",
131
+ "revision": "86d609183249ff8037eae33d76ebca3af9390ea8",
132
+ "config": null,
133
+ "records": 273333,
134
+ "target_records": 273333,
135
+ "bytes": 115045826,
136
+ "sha256": "b9194dd3c45ac353809a19c851ef5935ef06d568c46943ab021963a4a8adf677",
137
+ "fed_conversations": {
138
+ "analytical_reasoning": 1873,
139
+ "code_": 1873,
140
+ "rag": 1872,
141
+ "follow_up": 1872,
142
+ "fs_cot_flow": 1872,
143
+ "open_domain_qa": 1872
144
+ },
145
+ "normalized_input_sha256": "c4b165c28a5e4ba92304e38d23f2da93cf691eb9eb137b847aa9f6f5592b7180",
146
+ "max_per_conversation": 25,
147
+ "skipped_no_assistant_targets": {},
148
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
149
+ },
150
+ {
151
+ "name": "agent_flan",
152
+ "repo": "internlm/Agent-FLAN",
153
+ "revision": "8b25999e795a58b264fcb51e8746edb2faee9161",
154
+ "config": null,
155
+ "records": 362676,
156
+ "target_records": 362676,
157
+ "bytes": 185521080,
158
+ "sha256": "b760628a29e8f7e50237379d2ba4f4aaf4400b771706347032a80bc032524856",
159
+ "fed_conversations": {
160
+ "agent_instruct_tflan": 1730,
161
+ "agent_instruct_react": 1730,
162
+ "toolbench_tflan_60p_r10r5u7": 5530,
163
+ "toolbench_instruct_j1s1_3k": 5530
164
+ },
165
+ "normalized_input_sha256": "e67470488beec5b7dbc132d818185f9dc2c3fa5d39859dc59e27573e90582fca",
166
+ "max_per_conversation": 25,
167
+ "skipped_no_assistant_targets": {
168
+ "agent_instruct_react": 1,
169
+ "agent_instruct_tflan": 1
170
+ },
171
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
172
+ }
173
+ ],
174
+ "shuffle_seed": 7,
175
+ "first_32000_source_counts": {
176
+ "tulu3": 4287,
177
+ "orca": 4204,
178
+ "agent_flan": 5622,
179
+ "xlam": 4393,
180
+ "ultrachat": 3193,
181
+ "openhermes": 3171,
182
+ "toolace": 3164,
183
+ "glaive": 3219,
184
+ "hermes_fc": 747
185
+ },
186
+ "note": "Rebuilt original nine-source recipe; not the unavailable historical byte-for-byte mix"
187
+ }
lineages/corrected-20260930-v2/recovery/data/language_baseline/recovery_manifest.json ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "repo": "YNSScarSaiyan/nys-corpus",
3
+ "revision": "7eb0b0dd1f3122a4189d33ae6cc13c21d0a25774",
4
+ "files": {
5
+ "vocab_sparse.json": {
6
+ "bytes": 106816698,
7
+ "sha256": "23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e"
8
+ },
9
+ "train_corpus_sparse.bin": {
10
+ "bytes": 66816546,
11
+ "sha256": "a9b901c8c02b35eebe638abc0bc8ae551e359f51370890733ac917f0c8a97916"
12
+ },
13
+ "sft_corpus_sparse.bin": {
14
+ "bytes": 13200476,
15
+ "sha256": "91cc73e689d8e8e457221d3a20bb4d552cd9d2e5df1c1ed2f7ad2ce5c796cd60"
16
+ },
17
+ "distill_corpus_sparse.bin": {
18
+ "bytes": 15363524,
19
+ "sha256": "e1e27ee163ef7288fa6d68230fe167ef368e553ac8f19922019d80f87eb7e4e4"
20
+ }
21
+ }
22
+ }
lineages/corrected-20260930-v2/recovery/data/language_baseline/vocab_sparse.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e
3
+ size 106816698
lineages/corrected-20260930-v2/recovery/data/slots/compiler_physics.nysa ADDED
Binary file (48.8 kB). View file
 
lineages/corrected-20260930-v2/recovery/data/slots/recovery_manifest.json ADDED
@@ -0,0 +1,108 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "verified": true,
3
+ "checks": {
4
+ "2_compiler": true,
5
+ "4_speech": true,
6
+ "hard": true,
7
+ "sft": true,
8
+ "distill": true,
9
+ "source_checkpoint_sha256_unchanged": true
10
+ },
11
+ "record_counts": {
12
+ "speech_align.nysv": 305,
13
+ "speech_slot.nysv": 6,
14
+ "compiler_physics.nysa": 412
15
+ },
16
+ "checkpoint": "gpu_eqprop_1915000_20260913_095122.bin",
17
+ "checkpoint_sha256": "406a691b1ac42112705c5193814f76ff064f922161d48ab0e3f67452403bbeec",
18
+ "published_revision": "6f1f1e314e6d680f0e1541ddc6b1f6590b46cf40",
19
+ "vocoder_sha256": "028d0e759f9721b2110d8527028e7de3d80d5b611401d33db895d48e2f2df4d1",
20
+ "original_slot_sha256": {
21
+ "compiler_physics.nysa": "052fc7dee4b1081392051b9fde2fac6533aa77e82dbea56ac32c27a67ced6cf8",
22
+ "speech_slot.nysv": "2fd8be22d9408bb5d88a11b1c27c6c0ac80f42d7808f93a8d1ed31734a5f5cee",
23
+ "speech_align.nysv": "274f4d2e283a6b1de14eb73d4ea4f6c2e10fc18aff8bd7f1d1aca29b5e1430c5"
24
+ },
25
+ "actual": {
26
+ "slots": {
27
+ "1_language": {
28
+ "mix": {
29
+ "acc": null,
30
+ "hits": 0,
31
+ "n": 0,
32
+ "in_pool": 0,
33
+ "in_pool_frac": null
34
+ },
35
+ "hard": {
36
+ "acc": 0.0,
37
+ "hits": 0,
38
+ "n": 32,
39
+ "in_pool": 4,
40
+ "in_pool_frac": 0.125
41
+ },
42
+ "sft": {
43
+ "acc": 0.0,
44
+ "hits": 0,
45
+ "n": 32,
46
+ "in_pool": 30,
47
+ "in_pool_frac": 0.9375
48
+ },
49
+ "distill": {
50
+ "acc": 0.0,
51
+ "hits": 0,
52
+ "n": 32,
53
+ "in_pool": 18,
54
+ "in_pool_frac": 0.5625
55
+ }
56
+ },
57
+ "2_compiler": {
58
+ "clamped_free": {
59
+ "acc": 0.6385690789473685,
60
+ "hits": 1553,
61
+ "n": 2432,
62
+ "in_pool": 2432,
63
+ "in_pool_frac": 1.0
64
+ },
65
+ "spin_free": {
66
+ "acc": 0.4004934210526316,
67
+ "hits": 974,
68
+ "n": 2432,
69
+ "in_pool": 2432,
70
+ "in_pool_frac": 1.0
71
+ },
72
+ "n_recs": 64,
73
+ "n_free": 38
74
+ },
75
+ "4_speech": {
76
+ "acc": 0.10825043885313049,
77
+ "hits": 185,
78
+ "n": 1709,
79
+ "in_pool": 700,
80
+ "in_pool_frac": 0.4095962551199532
81
+ }
82
+ },
83
+ "headline": {
84
+ "language": 0.0,
85
+ "compiler": 0.6385690789473685,
86
+ "speech": 0.10825043885313049
87
+ }
88
+ },
89
+ "limitation": "Original live mix is unavailable; its score and aggregate language headline cannot be reproduced.",
90
+ "staged_files": {
91
+ "compiler_physics.nysa": {
92
+ "bytes": 48815,
93
+ "sha256": "052fc7dee4b1081392051b9fde2fac6533aa77e82dbea56ac32c27a67ced6cf8"
94
+ },
95
+ "speech_vocoder.nyvc": {
96
+ "bytes": 20500,
97
+ "sha256": "028d0e759f9721b2110d8527028e7de3d80d5b611401d33db895d48e2f2df4d1"
98
+ },
99
+ "speech_align.nysv": {
100
+ "bytes": 1490468,
101
+ "sha256": "fea716002f75cde8a4567d64e28c14e353c3cdb349cd9ec8ecf3c991f52fbf8c"
102
+ },
103
+ "speech_slot.nysv": {
104
+ "bytes": 15284,
105
+ "sha256": "2988cc801e2b95d6ebba35d335ec656486bc0fe0b5938d071a536c63879aca0c"
106
+ }
107
+ }
108
+ }
lineages/corrected-20260930-v2/recovery/data/slots/speech_align.nysv ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fea716002f75cde8a4567d64e28c14e353c3cdb349cd9ec8ecf3c991f52fbf8c
3
+ size 1490468
lineages/corrected-20260930-v2/recovery/data/slots/speech_slot.nysv ADDED
Binary file (15.3 kB). View file
 
lineages/corrected-20260930-v2/recovery/data/slots/speech_vocoder.nyvc ADDED
Binary file (20.5 kB). View file
 
lineages/corrected-20260930-v2/recovery/gpu_eqprop/build_corrected_mix.py ADDED
@@ -0,0 +1,217 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Rebuild all nine original SFT sources using frozen NYS IDs. No training/upload.
2
+
3
+ Pinned schemas are normalized explicitly. Every source and every requested split
4
+ must contribute; failed or empty sources fail the build, never shrink the diet.
5
+ """
6
+ import argparse
7
+ from array import array
8
+ from collections import Counter
9
+ import hashlib
10
+ import json
11
+ import mmap
12
+ import os
13
+ from pathlib import Path
14
+ import random
15
+ import re
16
+ import struct
17
+ import subprocess
18
+
19
+ VOCAB_SHA = '23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e'
20
+ SPECS = [
21
+ ('tulu3',4,['train'],None), ('openhermes',3,['train'],None),
22
+ ('ultrachat',3,['train_sft'],None), ('xlam',4,['train'],'dataset'),
23
+ ('toolace',3,['train'],None), ('glaive',3,['train'],None),
24
+ ('hermes_fc',2,['train'],'func_calling_singleturn'),
25
+ ('orca',4,['analytical_reasoning','code_','rag','follow_up','fs_cot_flow','open_domain_qa'],None),
26
+ ('agent_flan',4,['agent_instruct_tflan','agent_instruct_react','toolbench_tflan_60p_r10r5u7','toolbench_instruct_j1s1_3k'],None),
27
+ ]
28
+
29
+ def sha(path):
30
+ with Path(path).open('rb') as f:
31
+ return hashlib.file_digest(f, 'sha256').hexdigest()
32
+
33
+ def text(v):
34
+ if v is None: return ''
35
+ return v if isinstance(v,str) else json.dumps(v,ensure_ascii=False,separators=(',',':'))
36
+
37
+ def decode(v):
38
+ return json.loads(v) if isinstance(v,str) else v
39
+
40
+ class NoTargets(ValueError):
41
+ pass
42
+
43
+ def messages(row):
44
+ out=[]
45
+ conv=row.get('messages') or row.get('conversations') or row.get('conversation')
46
+ if conv:
47
+ conv=decode(conv)
48
+ for m in conv:
49
+ m=decode(m)
50
+ role=(m.get('role') or m.get('from') or '').lower()
51
+ role={'human':'user','gpt':'assistant','function':'tool','tool_response':'tool'}.get(role,role)
52
+ if role not in ('user','assistant','system','tool'):
53
+ raise ValueError(f'Unknown conversation role {role!r}')
54
+ content=text(m.get('content',m.get('value')))
55
+ if m.get('tool_calls'):
56
+ content += '\n'+text(m['tool_calls'])
57
+ if m.get('loss') is False:
58
+ role='tool' # context only; not an assistant target
59
+ if content.strip(): out.append(dict(role=role,content=content))
60
+ system=row.get('system') or row.get('system_prompt')
61
+ if system and not any(m['role']=='system' for m in out):
62
+ out.insert(0,dict(role='system',content=text(system)))
63
+ elif row.get('query') is not None and row.get('answers') is not None:
64
+ out=[dict(role='user',content=text(row['query'])+'\n tools '+text(row.get('tools'))),
65
+ dict(role='assistant',content=text(row['answers']))]
66
+ elif isinstance(row.get('chat'),str):
67
+ if row.get('system'): out.append(dict(role='system',content=text(row['system'])))
68
+ parts=re.split(r'(?:^|\n)(USER|ASSISTANT|FUNCTION RESPONSE):\s*', row['chat'])
69
+ if len(parts)<3 or parts[0].strip(): raise ValueError('Unsupported Glaive chat framing')
70
+ for role,body in zip(parts[1::2],parts[2::2]):
71
+ out.append(dict(role={'USER':'user','ASSISTANT':'assistant','FUNCTION RESPONSE':'tool'}[role],
72
+ content=body.replace('<|endoftext|>','').strip()))
73
+ else:
74
+ raise ValueError(f'Unsupported source schema: {sorted(row)}')
75
+ if not any(m['role']=='assistant' and m['content'].strip() for m in out):
76
+ raise NoTargets('No assistant targets in conversation')
77
+ return out
78
+
79
+ def offsets(path, expected):
80
+ off=array('Q')
81
+ with open(path,'rb') as f:
82
+ with mmap.mmap(f.fileno(),0,access=mmap.ACCESS_READ) as mm:
83
+ count,=struct.unpack_from('<I',mm,0)
84
+ if count!=expected: raise ValueError(f'Incomplete source {path}: {count} != {expected}')
85
+ pos=4
86
+ for _ in range(count):
87
+ off.append(pos)
88
+ n,=struct.unpack_from('<H',mm,pos)
89
+ if not 2<=n<=128: raise ValueError('Invalid sparse record length')
90
+ end=pos+2+4*n
91
+ if end>len(mm): raise ValueError('Truncated sparse record')
92
+ ids=struct.unpack_from('<'+'I'*n,mm,pos+2)
93
+ if max(ids)>=5_000_000: raise ValueError('Invalid sparse token ID')
94
+ pos=end
95
+ if pos!=len(mm): raise ValueError('Trailing corpus bytes')
96
+ return off
97
+
98
+ def build(args):
99
+ if args.token_file:
100
+ os.environ['HF_TOKEN']=args.token_file.read_text().strip()
101
+ from datasets import load_dataset
102
+ from huggingface_hub import hf_hub_download
103
+ def open_stream(name, info, config, sp):
104
+ if name=='agent_flan':
105
+ # Its published Arrow feature declaration omits the optional `type`
106
+ # field in some splits. Read the pinned JSONL rows without that cast.
107
+ p=hf_hub_download(info['repo'],'data/'+sp+'.jsonl',repo_type='dataset',revision=info['revision'])
108
+ def rows():
109
+ with open(p,encoding='utf-8') as f:
110
+ for line in f:
111
+ if line.strip(): yield json.loads(line)
112
+ return iter(rows())
113
+ return iter(load_dataset(info['repo'],name=config,split=sp,revision=info['revision'],streaming=True))
114
+ out=args.output
115
+ out.mkdir(parents=True,exist_ok=True)
116
+ if sha(args.vocab)!=VOCAB_SHA: raise ValueError('Wrong frozen vocabulary')
117
+ discovered=json.loads(args.discovery.read_text())
118
+ converter_sha=sha(args.converter)
119
+ reports=[]
120
+ used=0
121
+ for idx,(spec,info) in enumerate(zip(SPECS,discovered)):
122
+ name,weight,splits,config=spec
123
+ quota=args.records-used if idx==len(SPECS)-1 else args.records*weight//30
124
+ path=out/(name+'.bin')
125
+ report_path=out/(name+'.json')
126
+ if report_path.exists():
127
+ report=json.loads(report_path.read_text())
128
+ if report.get('converter_sha256')!=converter_sha or report['sha256']!=sha(path) or report.get('target_records',report['records'])!=quota or report['revision']!=info['revision']:
129
+ raise ValueError('Prior build identity differs')
130
+ offsets(path,report['records'])
131
+ reports.append(report)
132
+ used+=report['records']
133
+ continue
134
+ streams={sp:open_stream(name,info,config,sp) for sp in splits}
135
+ empty=Counter()
136
+ def next_messages(sp,it):
137
+ while True:
138
+ row=next(it)
139
+ try: return messages(row)
140
+ except NoTargets: empty[sp]+=1
141
+ # Check each schema before starting the converter; preserve the first row.
142
+ first={sp:next_messages(sp,it) for sp,it in streams.items()}
143
+ counts=Counter()
144
+ source_hash=hashlib.sha256()
145
+ with (out/(name+'.convert.log')).open('w') as log:
146
+ proc=subprocess.Popen([str(args.converter),'--vocab',str(args.vocab),'--output',str(path),
147
+ '--limit',str(quota),'--max-per-convo','25','--flush-every','1000'],
148
+ stdin=subprocess.PIPE,stderr=log,text=True)
149
+ try:
150
+ while proc.poll() is None and streams:
151
+ for sp,it in list(streams.items()):
152
+ try:
153
+ ms=first.pop(sp) if sp in first else next_messages(sp,it)
154
+ except StopIteration:
155
+ del streams[sp]
156
+ continue
157
+ line=json.dumps({'messages':ms},ensure_ascii=False)+'\n'
158
+ proc.stdin.write(line)
159
+ counts[sp]+=1
160
+ source_hash.update(line.encode())
161
+ except BrokenPipeError:
162
+ pass
163
+ except BaseException:
164
+ proc.kill(); proc.wait(); raise
165
+ finally:
166
+ try: proc.stdin.close()
167
+ except BrokenPipeError: pass
168
+ if proc.wait()!=0: raise RuntimeError(f'Converter failed: {name}')
169
+ actual=struct.unpack('<I',path.read_bytes()[:4])[0]
170
+ if actual<=0 or actual>quota: raise ValueError('Empty source or quota overrun')
171
+ offsets(path,actual)
172
+ if not all(counts[sp]>0 for sp in splits): raise ValueError('Missing split')
173
+ report=dict(name=name,repo=info['repo'],revision=info['revision'],config=config,
174
+ records=actual,target_records=quota,bytes=path.stat().st_size,sha256=sha(path),fed_conversations=dict(counts),
175
+ normalized_input_sha256=source_hash.hexdigest(),max_per_conversation=25,
176
+ skipped_no_assistant_targets=dict(empty),converter_sha256=converter_sha)
177
+ report_path.write_text(json.dumps(report,indent=2))
178
+ reports.append(report)
179
+ used+=actual
180
+ print(json.dumps(report),flush=True)
181
+ # Shuffle record order globally so the first live rotations include all sources.
182
+ handles=[open(out/(r['name']+'.bin'),'rb') for r in reports]
183
+ maps=[mmap.mmap(f.fileno(),0,access=mmap.ACCESS_READ) for f in handles]
184
+ record_offsets=[offsets(out/(r['name']+'.bin'),r['records']) for r in reports]
185
+ order=array('Q', ((i<<32)|j for i,r in enumerate(reports) for j in range(r['records'])))
186
+ random.Random(7).shuffle(order)
187
+ temp=out/'mix_sft_sparse.bin.writing'
188
+ first_counts=Counter()
189
+ with temp.open('wb') as f:
190
+ f.write(struct.pack('<I',len(order)))
191
+ for at,key in enumerate(order):
192
+ i,j=key>>32,key&0xffffffff
193
+ pos=record_offsets[i][j]
194
+ n,=struct.unpack_from('<H',maps[i],pos)
195
+ f.write(maps[i][pos:pos+2+4*n])
196
+ if at<32000: first_counts[reports[i]['name']]+=1
197
+ f.flush(); os.fsync(f.fileno())
198
+ offsets(temp,args.records)
199
+ final=out/'mix_sft_sparse.bin'
200
+ temp.replace(final)
201
+ for m,f in zip(maps,handles): m.close(); f.close()
202
+ manifest=dict(verified=True,records=args.records,bytes=final.stat().st_size,sha256=sha(final),
203
+ converter_sha256=converter_sha,
204
+ vocab_sha256=VOCAB_SHA,sources=reports,shuffle_seed=7,first_32000_source_counts=dict(first_counts),
205
+ note='Rebuilt original nine-source recipe; not the unavailable historical byte-for-byte mix')
206
+ (out/'mix_manifest.json').write_text(json.dumps(manifest,indent=2))
207
+ print(json.dumps(manifest),flush=True)
208
+
209
+ if __name__=='__main__':
210
+ ap=argparse.ArgumentParser(description=__doc__)
211
+ ap.add_argument('--discovery',type=Path,required=True)
212
+ ap.add_argument('--vocab',type=Path,required=True)
213
+ ap.add_argument('--converter',type=Path,required=True)
214
+ ap.add_argument('--output',type=Path,required=True)
215
+ ap.add_argument('--records',type=int,default=2_050_000)
216
+ ap.add_argument('--token-file',type=Path)
217
+ build(ap.parse_args())
lineages/corrected-20260930-v2/recovery/gpu_eqprop/build_cuda.sh ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env bash
2
+ set -euo pipefail
3
+ cd -- "$(dirname -- "${BASH_SOURCE[0]}")"
4
+ NVCC="${NVCC:-/data/yue-cuda/cuda/toolkit/bin/nvcc}"
5
+ if [[ ! -x "$NVCC" ]]; then NVCC="$(command -v nvcc)"; fi
6
+ "$NVCC" -O3 -std=c++17 -gencode=arch=compute_89,code=sm_89 \
7
+ -o eqprop_gpu_cuda eqprop_gpu.cu
lineages/corrected-20260930-v2/recovery/gpu_eqprop/cuda_self_test.h ADDED
@@ -0,0 +1,177 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ // Independent serial RK4 reference for the CUDA port. Synthetic data only.
2
+ struct CpuFixture {
3
+ std::vector<double> theta, amp, omega;
4
+ std::vector<uint64_t> off;
5
+ std::vector<uint32_t> col;
6
+ std::vector<float> K;
7
+ };
8
+
9
+ static double reference_step(CpuFixture& h, const StepPack& p, int n_mean,
10
+ double lr, bool settle, bool neutral,
11
+ std::vector<double>& settled) {
12
+ const size_t m = p.S.size();
13
+ std::vector<int> loc(h.theta.size(), -1);
14
+ std::vector<double> th(m), amps(m), start, free_phase;
15
+ for (size_t i = 0; i < m; ++i) {
16
+ loc[p.S[i]] = (int)i;
17
+ th[i] = settle && neutral && i >= (size_t)p.n_prompt ? 0.0 : h.theta[p.S[i]];
18
+ amps[i] = i < (size_t)p.n_prompt ? 1.0 : 0.0;
19
+ }
20
+ auto wrap = [](double x) {
21
+ if (x > M_PI) x -= 2 * M_PI;
22
+ else if (x < -M_PI) x += 2 * M_PI;
23
+ return x;
24
+ };
25
+ double mean = 0;
26
+ auto derivative = [&](const std::vector<double>& x, bool nudge) {
27
+ std::vector<double> d(m);
28
+ for (size_t i = 0; i < m; ++i) {
29
+ auto id = p.S[i];
30
+ double sum = 0;
31
+ for (uint64_t e = h.off[id]; e < h.off[id + 1]; ++e) {
32
+ int j = loc[h.col[e]];
33
+ if (j >= 0 && amps[j] > 1e-5)
34
+ sum += (double)h.K[e] * amps[j] * std::sin(x[j] - x[i]);
35
+ }
36
+ d[i] = g_omega_scale * h.omega[id] + sum;
37
+ if (nudge) {
38
+ for (size_t j = 0; j < p.nudge_id.size(); ++j)
39
+ if (p.nudge_id[j] == id)
40
+ d[i] += p.nudge_beta[j] * std::sin(mean - x[i]);
41
+ }
42
+ }
43
+ return d;
44
+ };
45
+ auto advance = [&](bool nudge, int steps) {
46
+ for (int step = 0; step < steps; ++step) {
47
+ auto x = th;
48
+ auto a = derivative(x, nudge);
49
+ for (size_t i = 0; i < m; ++i) th[i] = wrap(x[i] + .5 * kDt * a[i]);
50
+ auto b = derivative(th, nudge);
51
+ for (size_t i = 0; i < m; ++i) th[i] = wrap(x[i] + .5 * kDt * b[i]);
52
+ auto c = derivative(th, nudge);
53
+ for (size_t i = 0; i < m; ++i) th[i] = wrap(x[i] + kDt * c[i]);
54
+ auto d = derivative(th, nudge);
55
+ for (size_t i = 0; i < m; ++i)
56
+ th[i] = wrap(x[i] + kDt / 6 * (a[i] + 2*b[i] + 2*c[i] + d[i]));
57
+ }
58
+ };
59
+ advance(false, kRk4Free);
60
+ double X = 0, Y = 0;
61
+ for (int i = 0; i < n_mean; ++i) { X += std::cos(th[i]); Y += std::sin(th[i]); }
62
+ mean = std::atan2(Y, X);
63
+ if (settle) { settled = th; return mean; }
64
+ start = th;
65
+ for (size_t i = 0; i < m; ++i)
66
+ for (size_t j = 0; j < p.nudge_id.size(); ++j)
67
+ if (p.nudge_id[j] == p.S[i])
68
+ amps[i] = std::max(amps[i], std::min(1.0, std::abs(p.nudge_beta[j])));
69
+ advance(false, kRk4Nudge);
70
+ free_phase = th;
71
+ th = start;
72
+ advance(true, kRk4Nudge);
73
+ double contrast = 0;
74
+ int edges = 0;
75
+ for (size_t i = 0; i < m; ++i) {
76
+ auto id = p.S[i];
77
+ h.theta[id] = th[i];
78
+ h.amp[id] = 0;
79
+ if (amps[i] <= .1) continue;
80
+ for (uint64_t e = h.off[id]; e < h.off[id + 1]; ++e) {
81
+ int j = loc[h.col[e]];
82
+ if (j < 0 || amps[j] <= .1) continue;
83
+ double dc = std::cos(th[i] - th[j]) - std::cos(free_phase[i] - free_phase[j]);
84
+ contrast += dc;
85
+ ++edges;
86
+ h.K[e] = std::max(-kKMax, std::min(kKMax, h.K[e] + (float)(lr * dc)));
87
+ }
88
+ }
89
+ return edges ? contrast / edges : 0;
90
+ }
91
+
92
+ template<class T> static double max_error(const std::vector<T>& a, const std::vector<T>& b) {
93
+ if (a.size() != b.size()) return INFINITY;
94
+ double result = 0;
95
+ for (size_t i = 0; i < a.size(); ++i) {
96
+ if (!std::isfinite((double)a[i]) || !std::isfinite((double)b[i])) return INFINITY;
97
+ result = std::max(result, std::abs((double)a[i] - (double)b[i]));
98
+ }
99
+ return result;
100
+ }
101
+
102
+ static int run_self_test(size_t cap) {
103
+ const uint32_t N = 4096, k = 32;
104
+ GpuState g = alloc_state(N, k, cap);
105
+ CpuFixture h;
106
+ host_init_graph(N, k, h.theta, h.amp, h.omega, h.off, h.col, h.K);
107
+ auto read_state = [&]() {
108
+ CpuFixture out = h;
109
+ CUDA_CHECK(cudaMemcpy(out.theta.data(), g.theta, N * sizeof(double), cudaMemcpyDeviceToHost));
110
+ CUDA_CHECK(cudaMemcpy(out.amp.data(), g.amp, N * sizeof(double), cudaMemcpyDeviceToHost));
111
+ CUDA_CHECK(cudaMemcpy(out.omega.data(), g.omega, N * sizeof(double), cudaMemcpyDeviceToHost));
112
+ CUDA_CHECK(cudaMemcpy(out.K.data(), g.K, out.K.size() * sizeof(float), cudaMemcpyDeviceToHost));
113
+ return out;
114
+ };
115
+ for (int test = 0; test < 3; ++test) {
116
+ upload(g, h.theta, h.amp, h.omega, h.off, h.col, h.K);
117
+ std::vector<uint32_t> prompt = {1, 2, 3, 4, 5};
118
+ std::vector<std::pair<uint32_t, double>> nudges = {{6, 1.5}, {7, -1.5}, {8, .05}};
119
+ if (test == 2) {
120
+ prompt.clear(); nudges.clear();
121
+ for (uint32_t i = 1; i <= 128; ++i) prompt.push_back(i);
122
+ for (uint32_t i = 129; i <= 256; ++i) nudges.emplace_back(i, i % 2 ? 1.5 : -1.5);
123
+ }
124
+ auto p0 = pack_step(prompt, nudges, N);
125
+ auto p = p0;
126
+ fill_prompt_neighbors(g, p, kFillNeighbors);
127
+ g_omega_scale = test == 1 ? 1.0 : .01;
128
+ double lr = test == 2 ? 250 : .05; // exercise clipping as well as normal learning rate
129
+ CpuFixture expected = h;
130
+ std::vector<double> unused;
131
+ double ref = reference_step(expected, p, p0.n_prompt, lr, false, false, unused);
132
+ double got = launch_step(g, p0, lr);
133
+ auto actual = read_state();
134
+ double th = max_error(actual.theta, expected.theta), kw = max_error(actual.K, expected.K);
135
+ double dc = std::abs(got - ref);
136
+ std::printf("[self-test] case=%d active=%zu theta=%.3g K=%.3g contrast=%.3g\n", test, p.S.size(), th, kw, dc);
137
+ if (th > 1e-10 || kw > 1e-6 || !std::isfinite(got) || dc > 1e-10 ||
138
+ max_error(actual.amp, expected.amp) != 0 || max_error(actual.omega, h.omega) != 0) return 1;
139
+ }
140
+ // Verify neutral and stored-phase inference against the serial reference,
141
+ // including that inference leaves every model value unchanged.
142
+ upload(g, h.theta, h.amp, h.omega, h.off, h.col, h.K);
143
+ g_omega_scale = .01;
144
+ for (int neutral = 0; neutral <= 1; ++neutral) {
145
+ auto p = pack_step({1, 2, 3}, {{4, 0}, {5, 0}, {6, 0}}, N);
146
+ CpuFixture expected = h;
147
+ std::vector<double> th;
148
+ double ref = reference_step(expected, p, 3, 0, true, neutral, th);
149
+ std::vector<SettleRow> rows;
150
+ double mean = launch_settle(g, {1,2,3}, 0, {4,5,6}, kRk4Free, neutral, rows);
151
+ if (rows.size() != 3 || !std::isfinite(mean) || std::abs(mean-ref) > 1e-10) return 1;
152
+ for (size_t i = 0; i < rows.size(); ++i)
153
+ if (!std::isfinite(rows[i].theta) || std::abs(rows[i].theta-th[i+3]) > 1e-10) return 1;
154
+ auto actual = read_state();
155
+ if (max_error(actual.theta,h.theta) || max_error(actual.K,h.K) ||
156
+ max_error(actual.amp,h.amp) || max_error(actual.omega,h.omega)) return 1;
157
+ }
158
+ char temp[] = "/tmp/nys-cuda-test-XXXXXX";
159
+ if (!mkdtemp(temp)) return 1;
160
+ Args a;
161
+ a.save_dir = std::string(temp) + "/snapshots";
162
+ a.save = std::string(temp) + "/latest.bin";
163
+ save_snapshot(g, a, 123);
164
+ load_ckpt(g, a.save);
165
+ auto roundtrip = read_state();
166
+ bool ok = !max_error(roundtrip.theta,h.theta) && !max_error(roundtrip.K,h.K) &&
167
+ !max_error(roundtrip.amp,h.amp) && !max_error(roundtrip.omega,h.omega);
168
+ // A write failure must be reported, and an existing checkpoint must survive it.
169
+ ok = ok && !save_ckpt(g, std::string(temp) + "/missing/fail.bin");
170
+ const auto snapshot = std::filesystem::read_symlink(a.save);
171
+ std::filesystem::remove(a.save);
172
+ std::filesystem::remove(snapshot);
173
+ std::filesystem::remove(a.save_dir);
174
+ std::filesystem::remove(temp);
175
+ std::puts(ok ? "[self-test] PASS: CPU parity, read-only settle, checkpoint roundtrip/failure" : "[self-test] FAIL");
176
+ return ok ? 0 : 1;
177
+ }
lineages/corrected-20260930-v2/recovery/gpu_eqprop/distill_stdin.h ADDED
@@ -0,0 +1,98 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ // Streaming SDS1: UINT32_MAX record count means consume once, until EOF.
2
+ // Nonblocking parsing handles fragmented pipes; the caller paces rotations.
3
+ #include <fcntl.h>
4
+ #include <stdexcept>
5
+
6
+ struct DistillStdin {
7
+ std::vector<unsigned char> buf;
8
+ bool header = false, eof = false, done = false;
9
+ uint64_t emitted = 0;
10
+ uint32_t N;
11
+ explicit DistillStdin(uint32_t nodes) : N(nodes) {
12
+ int flags = fcntl(STDIN_FILENO, F_GETFL);
13
+ if (flags < 0 || fcntl(STDIN_FILENO, F_SETFL, flags | O_NONBLOCK) < 0)
14
+ throw std::runtime_error("cannot make distill stdin nonblocking");
15
+ }
16
+ static uint32_t u32(const unsigned char* p) {
17
+ return uint32_t(p[0]) | (uint32_t(p[1]) << 8) | (uint32_t(p[2]) << 16) | (uint32_t(p[3]) << 24);
18
+ }
19
+ static uint16_t u16(const unsigned char* p) { return p[0] | (uint16_t(p[1]) << 8); }
20
+ // 1 = record, 0 = temporarily empty, -1 = complete EOF. Malformed input throws.
21
+ int next(std::vector<uint32_t>& prompt, std::vector<uint32_t>& ids, std::vector<float>& pr) {
22
+ if (!eof && buf.size() < 8192) {
23
+ unsigned char chunk[4096];
24
+ ssize_t n = read(STDIN_FILENO, chunk, sizeof(chunk));
25
+ if (n > 0) buf.insert(buf.end(), chunk, chunk + n);
26
+ else if (!n) eof = true;
27
+ else if (errno != EAGAIN && errno != EWOULDBLOCK && errno != EINTR)
28
+ throw std::runtime_error("distill stdin read failed");
29
+ }
30
+ auto incomplete = [&]() -> int {
31
+ if (eof) throw std::runtime_error("truncated distill stdin record/header");
32
+ return 0;
33
+ };
34
+ if (!header) {
35
+ if (buf.size() < 8) return incomplete();
36
+ if (std::memcmp(buf.data(), "SDS1", 4) || u32(buf.data()+4) != UINT32_MAX)
37
+ throw std::runtime_error("stdin needs streaming SDS1 header (count UINT32_MAX)");
38
+ buf.erase(buf.begin(), buf.begin()+8);
39
+ header = true;
40
+ }
41
+ if (done) {
42
+ if (!buf.empty()) throw std::runtime_error("data after distill DONE footer");
43
+ return eof ? -1 : 0;
44
+ }
45
+ if (buf.size() < 2) return incomplete();
46
+ uint16_t plen = u16(buf.data());
47
+ if (!plen) {
48
+ if (buf.size() < 6) return incomplete();
49
+ if (std::memcmp(buf.data()+2, "DONE", 4))
50
+ throw std::runtime_error("invalid distill DONE footer");
51
+ buf.erase(buf.begin(), buf.begin()+6);
52
+ done = true;
53
+ if (!buf.empty()) throw std::runtime_error("data after distill DONE footer");
54
+ return eof ? -1 : 0;
55
+ }
56
+ if (!plen || plen > 127) throw std::runtime_error("distill prompt length outside 1..127");
57
+ size_t soft_at = 2 + 4 * plen;
58
+ if (buf.size() < soft_at + 2) return incomplete();
59
+ uint16_t nsoft = u16(buf.data()+soft_at);
60
+ if (!nsoft || nsoft > 64) throw std::runtime_error("distill target count outside 1..64");
61
+ size_t bytes = soft_at + 2 + 8*nsoft;
62
+ if (buf.size() < bytes) return incomplete();
63
+ prompt.resize(plen); ids.resize(nsoft); pr.resize(nsoft);
64
+ for (size_t i=0; i<plen; ++i) {
65
+ prompt[i] = u32(buf.data()+2+4*i);
66
+ if (prompt[i] >= N) throw std::runtime_error("distill prompt ID outside graph");
67
+ }
68
+ double mass = 0;
69
+ std::unordered_set<uint32_t> seen;
70
+ for (size_t i=0; i<nsoft; ++i) {
71
+ ids[i] = u32(buf.data()+soft_at+2+4*i);
72
+ uint32_t bits = u32(buf.data()+soft_at+2+4*nsoft+4*i);
73
+ std::memcpy(&pr[i], &bits, 4);
74
+ if (ids[i] >= N || !seen.insert(ids[i]).second || !std::isfinite(pr[i]) || pr[i] <= 0 || pr[i] > 1)
75
+ throw std::runtime_error("invalid distill target ID/probability");
76
+ mass += pr[i];
77
+ }
78
+ if (mass > 1.00001) throw std::runtime_error("distill probability mass exceeds 1");
79
+ buf.erase(buf.begin(), buf.begin()+bytes);
80
+ ++emitted;
81
+ return 1;
82
+ }
83
+ };
84
+
85
+ static int validate_distill_stdin(uint32_t N) {
86
+ DistillStdin source(N);
87
+ std::vector<uint32_t> prompt, ids;
88
+ std::vector<float> probs;
89
+ while (true) {
90
+ int status = source.next(prompt, ids, probs);
91
+ if (status < 0) break;
92
+ if (!status) std::this_thread::sleep_for(std::chrono::milliseconds(5));
93
+ }
94
+ if (!source.emitted) throw std::runtime_error("distill stream contained no records");
95
+ std::printf("[stream] validated %llu records, no CUDA allocation or training\n",
96
+ (unsigned long long)source.emitted);
97
+ return 0;
98
+ }
lineages/corrected-20260930-v2/recovery/gpu_eqprop/eqprop_gpu.cu ADDED
@@ -0,0 +1,2157 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ // NYS GPU EqProp — from-scratch CUDA fused kernel.
2
+ // No PyTorch. Compile with nvcc. Caps VRAM. Does not reset the device.
3
+ //
4
+ // nvcc -O3 -std=c++17 -arch=sm_89 -o eqprop_gpu_cuda eqprop_gpu.cu
5
+ // ./eqprop_gpu --probe
6
+ // ./eqprop_gpu --corpus train_corpus_sparse.bin --ckpt complete.bin --save out.bin \
7
+ // --vram-limit-gb 28 --device 0
8
+ //
9
+ // Physics vs the CPU trainer:
10
+ // - Integrate unique(prompt ∪ targets) only (S). No 512-neighbor explosion.
11
+ // - No length-N RK4 temps and no all-N omega wrap.
12
+ // - Same RK4 counts, dt, amp-gate, contrastive K rule, checkpoint layout.
13
+
14
+ #include <cuda_runtime.h>
15
+ #include <filesystem>
16
+ #include <memory>
17
+
18
+ #include <algorithm>
19
+ #include <cctype>
20
+ #include <cerrno>
21
+ #include <chrono>
22
+ #include <cmath>
23
+ #include <cstdint>
24
+ #include <cstdio>
25
+ #include <cstdlib>
26
+ #include <cstring>
27
+ #include <ctime>
28
+ #include <fstream>
29
+ #include <iostream>
30
+ #include <random>
31
+ #include <string>
32
+ #include <sys/stat.h>
33
+ #include <thread>
34
+ #include <unistd.h>
35
+ #include <unordered_set>
36
+ #include <utility>
37
+ #include <vector>
38
+
39
+ #ifndef M_PI
40
+ #define M_PI 3.14159265358979323846
41
+ #endif
42
+
43
+ static constexpr uint32_t kDefaultN = 5000000u;
44
+ static constexpr uint32_t kDefaultK = 512u;
45
+ static constexpr int kMaxS = 256;
46
+ static constexpr int kRk4Free = 5;
47
+ static constexpr int kRk4Nudge = 5;
48
+ static constexpr double kDt = 0.02;
49
+ // K is initialized in [-0.5, 0.5] and wired at k0=0.2. A single contested
50
+ // node -- one that many records push in conflicting directions every time
51
+ // it appears -- can walk an edge into the double digits over enough steps,
52
+ // since nothing bounds the accumulation (measured: text->silence-code
53
+ // edges reached |K|=16-18 with lr=0.05 over ~1e5 steps). Clip after every
54
+ // update so a shared node saturates instead of diverging; kKMax is well
55
+ // outside the normal learned range (empirically -0.2 to +0.7).
56
+ static constexpr float kKMax = 3.0f;
57
+ // Natural frequencies are multiplied by this inside the integrator only; the
58
+ // omega stored in the checkpoint is never modified, so any value is
59
+ // reversible by restarting with another. 1.0 = the original dynamics.
60
+ // Measured offline (nudge_signal_probe.py, 40 speech records, same-moment
61
+ // branches, target from text): gold dc -0.00012 (-2.1 SE) at 1.0,
62
+ // +0.0044 (+23 SE) at 0.01, +0.0049 (+24.5 SE) at 0. With the full spread
63
+ // no pair can phase-lock in a 0.1 s window, so nothing is learnable.
64
+ static double g_omega_scale = 1.0;
65
+ static constexpr double kVramFraction = 0.40; // share with SImi 0.50 / Uni 0.35
66
+ static constexpr size_t kVramLimitMax = 77ull << 30; // 77 GiB
67
+ static constexpr size_t kLeaveFreeBytes = 16ull << 30; // do not eat their headroom
68
+
69
+
70
+ #define CUDA_CHECK(expr) \
71
+ do { \
72
+ cudaError_t _e = (expr); \
73
+ if (_e != cudaSuccess) { \
74
+ std::cerr << "CUDA error " << cudaGetErrorString(_e) << " at " << __FILE__ << ":" \
75
+ << __LINE__ << " :: " << #expr << std::endl; \
76
+ std::exit(1); \
77
+ } \
78
+ } while (0)
79
+
80
+ static uint64_t ckpt_bytes(uint32_t N, uint32_t k) {
81
+ return 16ull + 32ull * (uint64_t)N + 8ull * (uint64_t)N * (uint64_t)k;
82
+ }
83
+
84
+ static void gb_print(const char* tag, size_t bytes) {
85
+ std::printf("%s %.3f GB (%zu B)\n", tag, (double)bytes / 1e9, bytes);
86
+ }
87
+
88
+ struct Args {
89
+ std::vector<std::string> corpora;
90
+ std::string data_dir;
91
+ std::string slots_dir;
92
+ std::string ckpt;
93
+ std::string save = "nys_gpu_checkpoint.bin";
94
+ std::string save_dir;
95
+ double lr = 0.05;
96
+ double beta = 1.5;
97
+ int save_every = 0;
98
+ int max_steps = -1;
99
+ int log_every = 50;
100
+ int device = 0;
101
+ double vram_limit_gb = 0.0; // 0 = 0.40 of device, max 77 GiB
102
+ bool probe = false;
103
+ bool check_ckpt = false;
104
+ bool check_inputs = false;
105
+ bool self_test = false;
106
+ bool settle = false;
107
+ int settle_rk4 = kRk4Free;
108
+ double omega_scale = 1.0;
109
+ int settle_fill = 0;
110
+ int fill_neighbors = 32;
111
+ bool init = false;
112
+ bool live = false;
113
+ bool distill_stdin = false;
114
+ bool require_full_curriculum = false;
115
+ bool save_first_rotation = false;
116
+ bool validate_distill_stdin = false;
117
+ std::string live_mix;
118
+ int live_idle_sec = 180;
119
+ int resume_steps = 0;
120
+ uint32_t N = kDefaultN;
121
+ uint32_t k = kDefaultK;
122
+ };
123
+
124
+ static void resolve_slots(Args& a) {
125
+ std::string dir = a.slots_dir;
126
+ if (dir.empty() && !a.data_dir.empty()) {
127
+ dir = a.data_dir;
128
+ if (!dir.empty() && dir.back() != '/') dir.push_back('/');
129
+ dir += "slots";
130
+ }
131
+ if (dir.empty()) return;
132
+ // Native slot files only. SDS1/hard projections stay off the live
133
+ // rotation so compiler/speech do not triple-count against language.
134
+ const char* names[] = {
135
+ "compiler_physics.nysa",
136
+ "speech_slot.nysv",
137
+ "speech_align.nysv",
138
+ };
139
+ for (const char* n : names) {
140
+ std::string p = dir;
141
+ if (!p.empty() && p.back() != '/') p.push_back('/');
142
+ p += n;
143
+ std::ifstream f(p, std::ios::binary);
144
+ if (f) {
145
+ if (std::find(a.corpora.begin(), a.corpora.end(), p) != a.corpora.end()) continue;
146
+ a.corpora.push_back(p);
147
+ std::printf("[data] + slot %s\n", p.c_str());
148
+ } else {
149
+ std::printf("[data] missing slot %s (skip)\n", p.c_str());
150
+ }
151
+ }
152
+ }
153
+
154
+ static void resolve_corpora(Args& a) {
155
+ const char* names[] = {
156
+ "train_corpus_sparse.bin",
157
+ "sft_corpus_sparse.bin",
158
+ "mix_sft_sparse.bin",
159
+ "distill_corpus_sparse.bin",
160
+ };
161
+ if (!a.data_dir.empty()) {
162
+ for (const char* n : names) {
163
+ std::string p = a.data_dir;
164
+ if (!p.empty() && p.back() != '/') p.push_back('/');
165
+ p += n;
166
+ if (a.live && std::strcmp(n, "mix_sft_sparse.bin") == 0) {
167
+ if (a.live_mix.empty()) a.live_mix = p;
168
+ std::printf("[data] live-tail %s (not preloaded)\n", a.live_mix.c_str());
169
+ continue;
170
+ }
171
+ std::ifstream f(p, std::ios::binary);
172
+ if (f) {
173
+ a.corpora.push_back(p);
174
+ std::printf("[data] + %s\n", p.c_str());
175
+ } else {
176
+ std::printf("[data] missing %s (skip)\n", p.c_str());
177
+ }
178
+ }
179
+ }
180
+ resolve_slots(a);
181
+ }
182
+
183
+ static bool flag(const char* a, const char* n) { return std::strcmp(a, n) == 0; }
184
+
185
+ static Args parse_args(int argc, char** argv) {
186
+ Args a;
187
+ for (int i = 1; i < argc; ++i) {
188
+ auto need = [&](const char* n) -> const char* {
189
+ if (i + 1 >= argc) {
190
+ std::cerr << "missing value for " << n << std::endl;
191
+ std::exit(2);
192
+ }
193
+ return argv[++i];
194
+ };
195
+ if (flag(argv[i], "--probe")) a.probe = true;
196
+ else if (flag(argv[i], "--check-ckpt")) a.check_ckpt = true;
197
+ else if (flag(argv[i], "--settle")) a.settle = true;
198
+ else if (flag(argv[i], "--settle-rk4")) a.settle_rk4 = std::atoi(need("--settle-rk4"));
199
+ else if (flag(argv[i], "--settle-fill")) a.settle_fill = std::atoi(need("--settle-fill"));
200
+ else if (flag(argv[i], "--omega-scale")) a.omega_scale = std::atof(need("--omega-scale"));
201
+ else if (flag(argv[i], "--fill-neighbors")) a.fill_neighbors = std::atoi(need("--fill-neighbors"));
202
+ else if (flag(argv[i], "--self-test")) a.self_test = true;
203
+ else if (flag(argv[i], "--init")) a.init = true;
204
+ else if (flag(argv[i], "--distill-stdin")) a.distill_stdin = true;
205
+ else if (flag(argv[i], "--require-full-curriculum")) a.require_full_curriculum = true;
206
+ else if (flag(argv[i], "--save-first-rotation")) a.save_first_rotation = true;
207
+ else if (flag(argv[i], "--validate-distill-stdin")) a.validate_distill_stdin = true;
208
+ else if (flag(argv[i], "--check-inputs")) a.check_inputs = true;
209
+ else if (flag(argv[i], "--live")) a.live = true;
210
+ else if (flag(argv[i], "--live-mix")) a.live_mix = need("--live-mix");
211
+ else if (flag(argv[i], "--live-idle-sec")) a.live_idle_sec = std::atoi(need("--live-idle-sec"));
212
+ else if (flag(argv[i], "--corpus")) a.corpora.push_back(need("--corpus"));
213
+ else if (flag(argv[i], "--data-dir")) a.data_dir = need("--data-dir");
214
+ else if (flag(argv[i], "--slots-dir")) a.slots_dir = need("--slots-dir");
215
+ else if (flag(argv[i], "--ckpt")) a.ckpt = need("--ckpt");
216
+ else if (flag(argv[i], "--resume-steps")) a.resume_steps = std::atoi(need("--resume-steps"));
217
+ else if (flag(argv[i], "--save")) a.save = need("--save");
218
+ else if (flag(argv[i], "--save-dir")) a.save_dir = need("--save-dir");
219
+ else if (flag(argv[i], "--lr")) a.lr = std::atof(need("--lr"));
220
+ else if (flag(argv[i], "--beta")) a.beta = std::atof(need("--beta"));
221
+ else if (flag(argv[i], "--save-every")) a.save_every = std::atoi(need("--save-every"));
222
+ else if (flag(argv[i], "--max-steps")) a.max_steps = std::atoi(need("--max-steps"));
223
+ else if (flag(argv[i], "--log-every")) a.log_every = std::atoi(need("--log-every"));
224
+ else if (flag(argv[i], "--device")) a.device = std::atoi(need("--device"));
225
+ else if (flag(argv[i], "--vram-limit-gb")) a.vram_limit_gb = std::atof(need("--vram-limit-gb"));
226
+ else if (flag(argv[i], "--N")) a.N = (uint32_t)std::strtoul(need("--N"), nullptr, 10);
227
+ else if (flag(argv[i], "--k")) a.k = (uint32_t)std::strtoul(need("--k"), nullptr, 10);
228
+ else if (flag(argv[i], "-h") || flag(argv[i], "--help")) {
229
+ std::puts(
230
+ "eqprop_gpu — CUDA fused EqProp (no PyTorch)\n"
231
+ " --check-ckpt validate checkpoint size/header, without CUDA\n"
232
+ " --probe print GPU memory and exit (no alloc)\n"
233
+ " --self-test synthetic numerical parity and checkpoint I/O checks\n"
234
+ " --check-inputs inspect corpus/slot wiring on CPU; no training\n"
235
+ " --settle read-only inference REPL on stdin (no K update)\n"
236
+ " --settle-rk4 N free-phase RK4 steps per settle (default 5)\n"
237
+ " --settle-fill N prompt neighbours clamped, as in training\n"
238
+ " --omega-scale F multiply natural frequencies in the integrator (default 1)\n"
239
+ " --fill-neighbors N prompt-neighbour fill for training (default 32)\n"
240
+ " --corpus PATH hard next-token or SDS1 (repeatable)\n"
241
+ " --data-dir DIR train + sft + distill sparse bins, in that order\n"
242
+ " --slots-dir DIR NYSA/NYSV compiler+speech (default DIR/slots)\n"
243
+ " --distill-stdin live SDS1 probabilities from stdin (requires --live)\n"
244
+ " --validate-distill-stdin validate a stream on CPU, without training\n"
245
+ " --live round-robin step now; tail mix as it grows\n"
246
+ " --live-mix PATH growing mix bin from sparse_convert\n"
247
+ " --ckpt PATH complete sparse checkpoint\n"
248
+ " --save PATH latest symlink / fallback dump\n"
249
+ " --save-dir DIR timestamped complete snapshots\n"
250
+ " --save-every N snapshot every N steps (0 = end only)\n"
251
+ " --vram-limit-gb F hard cap (default 0.40 of card, max 77 GiB)\n"
252
+ " --device I CUDA device index (default 0)\n"
253
+ " --init --N --k random init instead of load\n");
254
+ std::exit(0);
255
+ } else {
256
+ std::cerr << "unknown arg: " << argv[i] << std::endl;
257
+ std::exit(2);
258
+ }
259
+ }
260
+ return a;
261
+ }
262
+
263
+ struct VramCap {
264
+ size_t cap = kVramLimitMax;
265
+ size_t used = 0;
266
+ size_t leave_free = kLeaveFreeBytes;
267
+
268
+ void* alloc(size_t bytes, const char* name) {
269
+ size_t need = (bytes + 255ull) & ~255ull;
270
+ if (used + need > cap) {
271
+ std::cerr << "[vram] refuse " << name << " " << need << " B; used " << used
272
+ << " cap " << cap << std::endl;
273
+ std::exit(3);
274
+ }
275
+ size_t free_b = 0, total_b = 0;
276
+ CUDA_CHECK(cudaMemGetInfo(&free_b, &total_b));
277
+ if (free_b < need + leave_free) {
278
+ std::cerr << "[vram] refuse " << name << ": free=" << free_b << " need=" << need
279
+ << " plus leave_free=" << leave_free
280
+ << " (will not squeeze other jobs)" << std::endl;
281
+ std::exit(3);
282
+ }
283
+ void* p = nullptr;
284
+ CUDA_CHECK(cudaMalloc(&p, need));
285
+ used += need;
286
+ std::printf("[vram] +%-18s %8.3f GB running_total=%8.3f / %8.3f GB\n", name,
287
+ need / 1e9, used / 1e9, cap / 1e9);
288
+ return p;
289
+ }
290
+ };
291
+
292
+ __device__ __forceinline__ double wrap_pi(double x) {
293
+ if (x > M_PI) x -= 2.0 * M_PI;
294
+ else if (x < -M_PI) x += 2.0 * M_PI;
295
+ return x;
296
+ }
297
+
298
+ __device__ int loc_of(uint32_t id, const uint32_t* S, int m) {
299
+ for (int t = 0; t < m; ++t) {
300
+ if (S[t] == id) return t;
301
+ }
302
+ return -1;
303
+ }
304
+
305
+ // One persistent-ish block: free RK4, snapshot, nudge RK4, contrastive K.
306
+ __global__ void eqprop_fused(
307
+ uint32_t N,
308
+ uint32_t k,
309
+ double* theta,
310
+ const double* omega,
311
+ double* amp,
312
+ const uint64_t* off,
313
+ const uint32_t* col,
314
+ float* K,
315
+ const uint32_t* S_g,
316
+ int m,
317
+ int n_prompt,
318
+ const uint32_t* nudge_id,
319
+ const double* nudge_beta,
320
+ int n_nudge,
321
+ double lr,
322
+ double dt,
323
+ double* contrast_out,
324
+ int settle_mode,
325
+ int n_rk4,
326
+ int neutral_read,
327
+ double* settle_out,
328
+ int n_mean,
329
+ double omega_scale
330
+ ) {
331
+ (void)N;
332
+ if (blockIdx.x != 0 || m <= 0 || m > kMaxS) return;
333
+
334
+ __shared__ uint32_t sh_S[kMaxS];
335
+ __shared__ double sh_th[kMaxS];
336
+ __shared__ double sh_th0[kMaxS];
337
+ __shared__ double sh_free[kMaxS];
338
+ __shared__ double sh_start[kMaxS];
339
+ __shared__ double sh_amp[kMaxS];
340
+ __shared__ double sh_k1[kMaxS];
341
+ __shared__ double sh_k2[kMaxS];
342
+ __shared__ double sh_k3[kMaxS];
343
+ __shared__ double sh_k4[kMaxS];
344
+ __shared__ double sh_mean;
345
+ __shared__ double sh_dc[kMaxS];
346
+ __shared__ int sh_ne[kMaxS];
347
+
348
+ // The target phase comes from the real prompt only. fill_prompt_neighbors
349
+ // appends up to 32 neighbours of the last prompt token to the clamped set;
350
+ // for a speech step (2.3 text ids on average) they were 91.5% of the mean,
351
+ // so gold was pulled toward the neighbours' phase, not the text's.
352
+ if (n_mean <= 0 || n_mean > n_prompt) n_mean = n_prompt;
353
+
354
+ int li = (int)threadIdx.x;
355
+ if (li < m) {
356
+ uint32_t i = S_g[li];
357
+ sh_S[li] = i;
358
+ // Read nodes normally start from stored theta -- which on a shared
359
+ // slot is the nudged phase the last record left behind, so two
360
+ // nodes feel different coupling from the same prompt. A neutral
361
+ // start puts them on a common origin so disp reflects K alone.
362
+ sh_th[li] = (settle_mode && neutral_read && li >= n_prompt) ? 0.0 : theta[i];
363
+ sh_amp[li] = (li < n_prompt) ? 1.0 : 0.0;
364
+ // Settle is a read-only probe: it must not touch the graph at all.
365
+ if (settle_mode) {
366
+ settle_out[kMaxS + li] = sh_th[li]; // theta0, before integration
367
+ settle_out[2 * kMaxS + li] = omega_scale * omega[i]; // effective omega
368
+ } else {
369
+ amp[i] = sh_amp[li];
370
+ }
371
+ }
372
+ __syncthreads();
373
+
374
+ // Training contrasts two branches that start from the SAME state at the
375
+ // same moment: phase 1 continues free, phase 2 is nudged. The old rule
376
+ // contrasted the free snapshot with a nudged one taken 5 steps later, so
377
+ // every pair's natural counter-rotation (omega spread 50-1000 rad/s, ~36
378
+ // rad in 0.1 s) landed in dc -- the nudge was ~1/6000 of each update.
379
+ // Branching cancels that drift exactly; it costs kRk4Nudge extra steps.
380
+ int n_phase = settle_mode ? 1 : 3;
381
+ for (int phase = 0; phase < n_phase; ++phase) {
382
+ int nsteps = (phase == 0) ? (settle_mode ? n_rk4 : kRk4Free) : kRk4Nudge;
383
+ int use_nudge = (phase == 2);
384
+ for (int step = 0; step < nsteps; ++step) {
385
+ if (li < m) sh_th0[li] = sh_th[li];
386
+ __syncthreads();
387
+ for (int stg = 0; stg < 4; ++stg) {
388
+ if (li < m) {
389
+ uint32_t i = sh_S[li];
390
+ double thi = sh_th[li];
391
+ double sum = 0.0;
392
+ uint64_t a = off[i];
393
+ uint64_t b = off[i + 1];
394
+ if (b > a + k) b = a + k;
395
+ for (uint64_t e = a; e < b; ++e) {
396
+ uint32_t j = col[e];
397
+ int lj = loc_of(j, sh_S, m);
398
+ if (lj < 0) continue;
399
+ double aj = sh_amp[lj];
400
+ if (aj > 1e-5) {
401
+ sum += (double)K[e] * aj * sin(sh_th[lj] - thi);
402
+ }
403
+ }
404
+ double d = omega_scale * omega[i] + sum;
405
+ if (use_nudge) {
406
+ for (int n = 0; n < n_nudge; ++n) {
407
+ if (nudge_id[n] == i) {
408
+ d += nudge_beta[n] * sin(sh_mean - thi);
409
+ }
410
+ }
411
+ }
412
+ if (stg == 0) sh_k1[li] = d;
413
+ else if (stg == 1) sh_k2[li] = d;
414
+ else if (stg == 2) sh_k3[li] = d;
415
+ else sh_k4[li] = d;
416
+ }
417
+ __syncthreads();
418
+ if (li < m) {
419
+ if (stg == 0) sh_th[li] = wrap_pi(sh_th0[li] + 0.5 * dt * sh_k1[li]);
420
+ else if (stg == 1) sh_th[li] = wrap_pi(sh_th0[li] + 0.5 * dt * sh_k2[li]);
421
+ else if (stg == 2) sh_th[li] = wrap_pi(sh_th0[li] + dt * sh_k3[li]);
422
+ else {
423
+ double upd =
424
+ (dt / 6.0) * (sh_k1[li] + 2.0 * sh_k2[li] + 2.0 * sh_k3[li] + sh_k4[li]);
425
+ sh_th[li] = wrap_pi(sh_th0[li] + upd);
426
+ }
427
+ }
428
+ __syncthreads();
429
+ }
430
+ }
431
+ if (phase == 0) {
432
+ // Branch point: both training branches restart from here.
433
+ if (li < m) {
434
+ sh_start[li] = sh_th[li];
435
+ sh_free[li] = sh_th[li];
436
+ }
437
+ __syncthreads();
438
+ if (li == 0) {
439
+ double X = 0.0, Y = 0.0;
440
+ for (int t = 0; t < n_mean; ++t) {
441
+ X += cos(sh_start[t]);
442
+ Y += sin(sh_start[t]);
443
+ }
444
+ sh_mean = atan2(Y, X);
445
+ }
446
+ __syncthreads();
447
+ if (li < m && !settle_mode) {
448
+ uint32_t i = sh_S[li];
449
+ for (int n = 0; n < n_nudge; ++n) {
450
+ if (nudge_id[n] == i) {
451
+ double v = nudge_beta[n];
452
+ if (v < 0.0) v = -v;
453
+ if (v > 1.0) v = 1.0;
454
+ if (v > sh_amp[li]) sh_amp[li] = v;
455
+ }
456
+ }
457
+ amp[i] = sh_amp[li];
458
+ }
459
+ __syncthreads();
460
+ } else if (phase == 1) {
461
+ // Free-continued reference at the same moment the nudged branch
462
+ // will end; then rewind to the branch point for phase 2.
463
+ if (li < m) {
464
+ sh_free[li] = sh_th[li];
465
+ sh_th[li] = sh_start[li];
466
+ }
467
+ __syncthreads();
468
+ }
469
+ }
470
+
471
+ if (settle_mode) {
472
+ // Settled free phase + the prompt mean the nudge would have pulled
473
+ // targets toward. No theta writeback, no amp zeroing, no K update.
474
+ if (li < m) settle_out[li] = sh_th[li];
475
+ __syncthreads();
476
+ if (li == 0) settle_out[3 * kMaxS] = sh_mean;
477
+ return;
478
+ }
479
+
480
+ if (li < m) {
481
+ uint32_t i = sh_S[li];
482
+ theta[i] = sh_th[li];
483
+ uint64_t a = off[i];
484
+ uint64_t b = off[i + 1];
485
+ if (b > a + k) b = a + k;
486
+ double local = 0.0;
487
+ int nedge = 0;
488
+ if (sh_amp[li] > 0.1) {
489
+ for (uint64_t e = a; e < b; ++e) {
490
+ uint32_t j = col[e];
491
+ int lj = loc_of(j, sh_S, m);
492
+ if (lj < 0) continue;
493
+ if (sh_amp[lj] > 0.1) {
494
+ double cfree = cos(sh_free[li] - sh_free[lj]);
495
+ double cnudge = cos(sh_th[li] - sh_th[lj]);
496
+ double dc = cnudge - cfree;
497
+ local += dc;
498
+ nedge++;
499
+ K[e] += (float)(lr * dc);
500
+ if (K[e] > kKMax) K[e] = kKMax;
501
+ else if (K[e] < -kKMax) K[e] = -kKMax;
502
+ }
503
+ }
504
+ }
505
+ sh_dc[li] = local;
506
+ sh_ne[li] = nedge;
507
+ amp[i] = 0.0;
508
+ } else {
509
+ sh_dc[li] = 0.0;
510
+ sh_ne[li] = 0;
511
+ }
512
+ __syncthreads();
513
+ if (li == 0 && contrast_out) {
514
+ double s = 0.0;
515
+ int c = 0;
516
+ for (int t = 0; t < m; ++t) {
517
+ s += sh_dc[t];
518
+ c += sh_ne[t];
519
+ }
520
+ *contrast_out = (c > 0) ? (s / (double)c) : 0.0;
521
+ }
522
+ }
523
+
524
+ struct GpuState {
525
+ uint32_t N = 0;
526
+ uint32_t k = 0;
527
+ double* theta = nullptr;
528
+ double* amp = nullptr;
529
+ double* omega = nullptr;
530
+ uint64_t* off = nullptr;
531
+ uint32_t* col = nullptr;
532
+ float* K = nullptr;
533
+ uint32_t* S = nullptr;
534
+ uint32_t* nudge_id = nullptr;
535
+ double* nudge_beta = nullptr;
536
+ double* contrast = nullptr;
537
+ double* settle_out = nullptr;
538
+ VramCap cap;
539
+ };
540
+
541
+ static void probe_device(int device, double limit_gb) {
542
+ int ndev = 0;
543
+ CUDA_CHECK(cudaGetDeviceCount(&ndev));
544
+ std::printf("[gpu] cudaGetDeviceCount=%d\n", ndev);
545
+ if (ndev <= 0) {
546
+ std::cerr << "[gpu] no CUDA devices\n";
547
+ std::exit(1);
548
+ }
549
+ CUDA_CHECK(cudaSetDevice(device));
550
+ cudaDeviceProp p{};
551
+ CUDA_CHECK(cudaGetDeviceProperties(&p, device));
552
+ size_t free_b = 0, total_b = 0;
553
+ CUDA_CHECK(cudaMemGetInfo(&free_b, &total_b));
554
+ std::printf("[gpu] device=%d name=%s\n", device, p.name);
555
+ gb_print("[gpu] total", total_b);
556
+ gb_print("[gpu] used ", total_b - free_b);
557
+ gb_print("[gpu] free ", free_b);
558
+ size_t cap = limit_gb > 0.0 ? (size_t)(limit_gb * (1ull << 30))
559
+ : (size_t)(kVramFraction * total_b);
560
+ gb_print("[gpu] cap ", std::min(cap, kVramLimitMax));
561
+ std::printf("[gpu] fraction=%.2f will not cudaDeviceReset; will not touch other PIDs\n",
562
+ kVramFraction);
563
+ }
564
+
565
+ static void host_init_graph(uint32_t N, uint32_t k,
566
+ std::vector<double>& theta,
567
+ std::vector<double>& amp,
568
+ std::vector<double>& omega,
569
+ std::vector<uint64_t>& off,
570
+ std::vector<uint32_t>& col,
571
+ std::vector<float>& K) {
572
+ theta.assign(N, 0.0);
573
+ amp.assign(N, 0.0);
574
+ omega.assign(N, 0.0);
575
+ off.assign((size_t)N + 1, 0);
576
+ col.assign((size_t)N * k, 0);
577
+ K.assign((size_t)N * k, 0.0f);
578
+ std::mt19937 gen(0xC0FFEE);
579
+ std::uniform_real_distribution<double> phase(-M_PI, M_PI);
580
+ std::uniform_real_distribution<double> freq(50.0, 1000.0);
581
+ std::uniform_real_distribution<float> w(-0.5f, 0.5f);
582
+ for (uint32_t i = 0; i < N; ++i) {
583
+ theta[i] = phase(gen);
584
+ omega[i] = freq(gen);
585
+ off[i] = (uint64_t)i * k;
586
+ for (uint32_t j = 0; j < k; ++j) {
587
+ uint64_t e = (uint64_t)i * k + j;
588
+ if (j < k - 16u) {
589
+ col[e] = (uint32_t)((i + N - k / 2u + j) % N);
590
+ } else {
591
+ col[e] = gen() % N;
592
+ }
593
+ K[e] = w(gen);
594
+ }
595
+ }
596
+ off[N] = (uint64_t)N * k;
597
+ }
598
+
599
+ static void upload(GpuState& g, const std::vector<double>& theta, const std::vector<double>& amp,
600
+ const std::vector<double>& omega, const std::vector<uint64_t>& off,
601
+ const std::vector<uint32_t>& col, const std::vector<float>& K) {
602
+ CUDA_CHECK(cudaMemcpy(g.theta, theta.data(), theta.size() * sizeof(double), cudaMemcpyHostToDevice));
603
+ CUDA_CHECK(cudaMemcpy(g.amp, amp.data(), amp.size() * sizeof(double), cudaMemcpyHostToDevice));
604
+ CUDA_CHECK(cudaMemcpy(g.omega, omega.data(), omega.size() * sizeof(double), cudaMemcpyHostToDevice));
605
+ CUDA_CHECK(cudaMemcpy(g.off, off.data(), off.size() * sizeof(uint64_t), cudaMemcpyHostToDevice));
606
+ CUDA_CHECK(cudaMemcpy(g.col, col.data(), col.size() * sizeof(uint32_t), cudaMemcpyHostToDevice));
607
+ CUDA_CHECK(cudaMemcpy(g.K, K.data(), K.size() * sizeof(float), cudaMemcpyHostToDevice));
608
+ }
609
+
610
+ static GpuState alloc_state(uint32_t N, uint32_t k, size_t cap_bytes) {
611
+ GpuState g;
612
+ g.N = N;
613
+ g.k = k;
614
+ g.cap.cap = cap_bytes;
615
+ size_t nnz = (size_t)N * (size_t)k;
616
+ g.theta = (double*)g.cap.alloc(N * sizeof(double), "theta");
617
+ g.amp = (double*)g.cap.alloc(N * sizeof(double), "amp");
618
+ g.omega = (double*)g.cap.alloc(N * sizeof(double), "omega");
619
+ g.off = (uint64_t*)g.cap.alloc(((size_t)N + 1) * sizeof(uint64_t), "K_offsets");
620
+ g.col = (uint32_t*)g.cap.alloc(nnz * sizeof(uint32_t), "K_cols");
621
+ g.K = (float*)g.cap.alloc(nnz * sizeof(float), "K_vals");
622
+ g.S = (uint32_t*)g.cap.alloc(kMaxS * sizeof(uint32_t), "S");
623
+ g.nudge_id = (uint32_t*)g.cap.alloc(kMaxS * sizeof(uint32_t), "nudge_id");
624
+ g.nudge_beta = (double*)g.cap.alloc(kMaxS * sizeof(double), "nudge_beta");
625
+ g.contrast = (double*)g.cap.alloc(sizeof(double), "contrast");
626
+ g.settle_out = (double*)g.cap.alloc((3 * kMaxS + 1) * sizeof(double), "settle_out");
627
+ CUDA_CHECK(cudaMemset(g.amp, 0, N * sizeof(double)));
628
+ return g;
629
+ }
630
+
631
+ static void load_ckpt(GpuState& g, const std::string& path) {
632
+ std::ifstream f(path, std::ios::binary);
633
+ if (!f) {
634
+ std::cerr << "cannot open ckpt " << path << std::endl;
635
+ std::exit(1);
636
+ }
637
+ f.seekg(0, std::ios::end);
638
+ uint64_t sz = (uint64_t)f.tellg();
639
+ f.seekg(0, std::ios::beg);
640
+ uint64_t need = ckpt_bytes(g.N, g.k);
641
+ if (sz != need) {
642
+ std::cerr << "[ckpt] size " << sz << " != " << need << " (truncated or wrong N/k). abort\n";
643
+ std::exit(1);
644
+ }
645
+ uint32_t file_N = 0, file_k = 0;
646
+ f.read((char*)&file_N, 4);
647
+ f.read((char*)&file_k, 4);
648
+ if (file_N != g.N || file_k != g.k) {
649
+ std::cerr << "[ckpt] N/k mismatch file=" << file_N << "x" << file_k << " expect " << g.N
650
+ << "x" << g.k << std::endl;
651
+ std::exit(1);
652
+ }
653
+ std::vector<char> buf;
654
+ auto copy_field = [&](void* dst, size_t bytes, const char* name) {
655
+ const size_t chunk = 256ull * 1024ull * 1024ull;
656
+ size_t off = 0;
657
+ while (off < bytes) {
658
+ size_t n = std::min(chunk, bytes - off);
659
+ buf.resize(n);
660
+ f.read(buf.data(), (std::streamsize)n);
661
+ if ((size_t)f.gcount() != n) {
662
+ std::cerr << "[ckpt] short read on " << name << std::endl;
663
+ std::exit(1);
664
+ }
665
+ CUDA_CHECK(cudaMemcpy((char*)dst + off, buf.data(), n, cudaMemcpyHostToDevice));
666
+ off += n;
667
+ }
668
+ std::printf("[ckpt] loaded %s\n", name);
669
+ };
670
+ copy_field(g.theta, g.N * sizeof(double), "theta");
671
+ copy_field(g.amp, g.N * sizeof(double), "amp");
672
+ copy_field(g.omega, g.N * sizeof(double), "omega");
673
+ copy_field(g.off, ((size_t)g.N + 1) * sizeof(uint64_t), "offsets");
674
+ copy_field(g.col, (size_t)g.N * g.k * sizeof(uint32_t), "cols");
675
+ copy_field(g.K, (size_t)g.N * g.k * sizeof(float), "K");
676
+ std::puts("[ckpt] complete");
677
+ }
678
+
679
+ static bool save_ckpt(const GpuState& g, const std::string& path) {
680
+ std::string tmp = path + ".writing";
681
+ std::printf("[ckpt] saving %s (device -> host, other jobs left running)\n", path.c_str());
682
+ std::ofstream f(tmp, std::ios::binary | std::ios::trunc);
683
+ if (!f) {
684
+ std::cerr << "cannot write " << tmp << std::endl;
685
+ return false;
686
+ }
687
+ f.write((const char*)&g.N, 4);
688
+ f.write((const char*)&g.k, 4);
689
+ std::vector<char> buf;
690
+ auto dump = [&](const void* src, size_t bytes) {
691
+ const size_t chunk = 256ull * 1024ull * 1024ull;
692
+ size_t off = 0;
693
+ while (off < bytes) {
694
+ size_t n = std::min(chunk, bytes - off);
695
+ buf.resize(n);
696
+ CUDA_CHECK(cudaMemcpy(buf.data(), (const char*)src + off, n, cudaMemcpyDeviceToHost));
697
+ f.write(buf.data(), (std::streamsize)n);
698
+ off += n;
699
+ }
700
+ };
701
+ dump(g.theta, g.N * sizeof(double));
702
+ dump(g.amp, g.N * sizeof(double));
703
+ dump(g.omega, g.N * sizeof(double));
704
+ dump(g.off, ((size_t)g.N + 1) * sizeof(uint64_t));
705
+ dump(g.col, (size_t)g.N * g.k * sizeof(uint32_t));
706
+ dump(g.K, (size_t)g.N * g.k * sizeof(float));
707
+ f.flush();
708
+ if (!f) {
709
+ std::cerr << "[ckpt] write failed " << tmp << std::endl;
710
+ return false;
711
+ }
712
+ f.close();
713
+ if (!f) {
714
+ std::cerr << "[ckpt] close failed " << tmp << std::endl;
715
+ return false;
716
+ }
717
+ uint64_t sz = 0;
718
+ {
719
+ std::ifstream chk(tmp, std::ios::binary | std::ios::ate);
720
+ sz = chk ? (uint64_t)chk.tellg() : 0;
721
+ }
722
+ uint64_t need = ckpt_bytes(g.N, g.k);
723
+ if (sz != need) {
724
+ std::cerr << "[ckpt] abort rename: size " << sz << " != " << need << "\n";
725
+ std::remove(tmp.c_str());
726
+ return false;
727
+ }
728
+ if (std::rename(tmp.c_str(), path.c_str()) != 0) {
729
+ std::cerr << "[ckpt] rename failed " << tmp << " -> " << path << std::endl;
730
+ return false;
731
+ }
732
+ std::printf("[ckpt] saved %s (%llu B)\n", path.c_str(), (unsigned long long)sz);
733
+ std::fflush(stdout);
734
+ return true;
735
+ }
736
+
737
+ static std::string utc_stamp() {
738
+ std::time_t t = std::time(nullptr);
739
+ std::tm tm{};
740
+ gmtime_r(&t, &tm);
741
+ char buf[32];
742
+ std::strftime(buf, sizeof(buf), "%Y%m%d_%H%M%S", &tm);
743
+ return buf;
744
+ }
745
+
746
+ static void ensure_dir(const std::string& dir) {
747
+ if (dir.empty() || dir == ".") return;
748
+ if (mkdir(dir.c_str(), 0755) != 0 && errno != EEXIST) {
749
+ std::cerr << "[ckpt] mkdir " << dir << " failed\n";
750
+ }
751
+ }
752
+
753
+ static void save_snapshot(const GpuState& g, const Args& a, int steps) {
754
+ std::string dir = a.save_dir.empty() ? "." : a.save_dir;
755
+ ensure_dir(dir);
756
+ if (a.require_full_curriculum && std::filesystem::space(dir).available < 30'640'000'016ULL)
757
+ throw std::runtime_error("Insufficient disk headroom for complete checkpoint; stopping before write");
758
+ std::string name = "gpu_eqprop_" + std::to_string(steps) + "_" + utc_stamp() + ".bin";
759
+ std::string path = dir;
760
+ if (!path.empty() && path.back() != '/') path.push_back('/');
761
+ path += name;
762
+ if (!save_ckpt(g, path)) std::exit(1);
763
+ if (!a.save.empty() && a.save != path) {
764
+ // An absolute target works even when latest and snapshots have different parents.
765
+ const std::string target = std::filesystem::absolute(path).string();
766
+ const std::string link_tmp = a.save + ".link." + std::to_string(getpid());
767
+ if (symlink(target.c_str(), link_tmp.c_str()) != 0) {
768
+ std::perror("[ckpt] create temporary latest link");
769
+ std::exit(1);
770
+ }
771
+ if (std::rename(link_tmp.c_str(), a.save.c_str()) != 0) {
772
+ std::perror("[ckpt] replace latest link");
773
+ std::remove(link_tmp.c_str());
774
+ std::exit(1);
775
+ }
776
+ std::printf("[ckpt] latest -> %s\n", target.c_str());
777
+ }
778
+ }
779
+
780
+ struct StepPack {
781
+ std::vector<uint32_t> S;
782
+ int n_prompt = 0;
783
+ std::vector<uint32_t> nudge_id;
784
+ std::vector<double> nudge_beta;
785
+ };
786
+
787
+ static int kFillNeighbors = 32;
788
+
789
+ static StepPack pack_step(const std::vector<uint32_t>& prompt,
790
+ const std::vector<std::pair<uint32_t, double>>& nudges,
791
+ uint32_t N) {
792
+ StepPack p;
793
+ std::unordered_set<uint32_t> seen;
794
+ for (uint32_t t : prompt) {
795
+ if (t >= N) continue;
796
+ if (seen.insert(t).second) p.S.push_back(t);
797
+ }
798
+ p.n_prompt = (int)p.S.size();
799
+ for (auto& nb : nudges) {
800
+ if (nb.first >= N) continue;
801
+ p.nudge_id.push_back(nb.first);
802
+ p.nudge_beta.push_back(nb.second);
803
+ if (seen.insert(nb.first).second) p.S.push_back(nb.first);
804
+ }
805
+ if ((int)p.S.size() > kMaxS) {
806
+ p.S.resize(kMaxS);
807
+ if (p.n_prompt > kMaxS) p.n_prompt = kMaxS;
808
+ }
809
+ return p;
810
+ }
811
+
812
+ static void fill_prompt_neighbors(GpuState& g, StepPack& p, int n_fill) {
813
+ if (n_fill <= 0 || p.n_prompt <= 0 || (int)p.S.size() >= kMaxS) return;
814
+ uint32_t last = p.S[(size_t)p.n_prompt - 1];
815
+ if (last >= g.N) return;
816
+ uint64_t start = 0, end = 0;
817
+ CUDA_CHECK(cudaMemcpy(&start, g.off + last, sizeof(uint64_t), cudaMemcpyDeviceToHost));
818
+ CUDA_CHECK(cudaMemcpy(&end, g.off + last + 1, sizeof(uint64_t), cudaMemcpyDeviceToHost));
819
+ uint64_t n = end > start ? end - start : 0;
820
+ if (n > (uint64_t)g.k) n = g.k;
821
+ if (n == 0) return;
822
+ std::vector<uint32_t> cols((size_t)n);
823
+ CUDA_CHECK(cudaMemcpy(cols.data(), g.col + start, (size_t)n * sizeof(uint32_t), cudaMemcpyDeviceToHost));
824
+ std::unordered_set<uint32_t> seen(p.S.begin(), p.S.end());
825
+ std::vector<uint32_t> extras;
826
+ extras.reserve((size_t)n_fill);
827
+ for (uint32_t j : cols) {
828
+ if ((int)p.S.size() + (int)extras.size() >= kMaxS) break;
829
+ if ((int)extras.size() >= n_fill) break;
830
+ if (j >= g.N || seen.count(j)) continue;
831
+ seen.insert(j);
832
+ extras.push_back(j);
833
+ }
834
+ if (extras.empty()) return;
835
+ p.S.insert(p.S.begin() + p.n_prompt, extras.begin(), extras.end());
836
+ p.n_prompt += (int)extras.size();
837
+ if ((int)p.S.size() > kMaxS) {
838
+ p.S.resize(kMaxS);
839
+ if (p.n_prompt > kMaxS) p.n_prompt = kMaxS;
840
+ }
841
+ }
842
+
843
+ static double launch_step(GpuState& g, const StepPack& p0, double lr) {
844
+ StepPack p = p0;
845
+ const int n_mean = p0.n_prompt; // real prompt, before the fill neighbours
846
+ fill_prompt_neighbors(g, p, kFillNeighbors);
847
+ if (p.S.empty() || p.nudge_id.empty()) return 0.0;
848
+ CUDA_CHECK(cudaMemcpy(g.S, p.S.data(), p.S.size() * sizeof(uint32_t), cudaMemcpyHostToDevice));
849
+ CUDA_CHECK(cudaMemcpy(g.nudge_id, p.nudge_id.data(), p.nudge_id.size() * sizeof(uint32_t),
850
+ cudaMemcpyHostToDevice));
851
+ CUDA_CHECK(cudaMemcpy(g.nudge_beta, p.nudge_beta.data(), p.nudge_beta.size() * sizeof(double),
852
+ cudaMemcpyHostToDevice));
853
+ eqprop_fused<<<1, kMaxS>>>(g.N, g.k, g.theta, g.omega, g.amp,
854
+ g.off, g.col, g.K, g.S, (int)p.S.size(), p.n_prompt, g.nudge_id,
855
+ g.nudge_beta, (int)p.nudge_id.size(), lr, kDt, g.contrast, 0, kRk4Free, 0,
856
+ nullptr, n_mean, g_omega_scale);
857
+ CUDA_CHECK(cudaGetLastError());
858
+ double contrast = 0.0;
859
+ CUDA_CHECK(cudaMemcpy(&contrast, g.contrast, sizeof(double), cudaMemcpyDeviceToHost));
860
+ return contrast;
861
+ }
862
+
863
+ // ---------------------------------------------------------------------------
864
+ // Settle: read-only inference. Same integrator as the training free phase,
865
+ // same clamp (prompt amp 1, everything else 0), no nudge, no K update, no
866
+ // writeback. This is the operator every readout was missing: the sidecar
867
+ // scores stored theta, which is the *nudged* phase left behind by whichever
868
+ // record touched a node last. Here the prompt drives the read nodes and we
869
+ // report where they land.
870
+ // ---------------------------------------------------------------------------
871
+
872
+ // Host wrap into [-pi, pi]. The device wrap_pi is __device__ only and
873
+ // nudges by a single period, which is not enough here: omega*T reaches
874
+ // ~100 rad for the fastest oscillators.
875
+ static double wrap_pi_host(double x) { return std::remainder(x, 2.0 * M_PI); }
876
+
877
+ struct SettleRow {
878
+ uint32_t id;
879
+ double theta;
880
+ double theta0;
881
+ double omega;
882
+ };
883
+
884
+ // One launch. S = prompt (+ clamped neighbours) followed by `read`.
885
+ // Returns the prompt mean phase; fills `rows` for the read nodes.
886
+ static double launch_settle(GpuState& g,
887
+ const std::vector<uint32_t>& prompt,
888
+ int n_fill,
889
+ const std::vector<uint32_t>& read,
890
+ int n_rk4,
891
+ int neutral,
892
+ std::vector<SettleRow>& rows) {
893
+ rows.clear();
894
+ StepPack p;
895
+ std::unordered_set<uint32_t> seen;
896
+ for (uint32_t t : prompt) {
897
+ if (t >= g.N) continue;
898
+ if (seen.insert(t).second) p.S.push_back(t);
899
+ }
900
+ p.n_prompt = (int)p.S.size();
901
+ if (p.n_prompt <= 0) return 0.0;
902
+ if (p.n_prompt > kMaxS) {
903
+ p.S.resize(kMaxS);
904
+ p.n_prompt = kMaxS;
905
+ }
906
+ if (n_fill > 0) fill_prompt_neighbors(g, p, n_fill);
907
+ seen.clear();
908
+ seen.insert(p.S.begin(), p.S.end());
909
+
910
+ // Read nodes ride along at amp 0: they receive coupling from the clamped
911
+ // prompt but do not drive it, so the prompt trajectory (and the mean) is
912
+ // identical no matter which read chunk we are on.
913
+ int first_read = (int)p.S.size();
914
+ for (uint32_t r : read) {
915
+ if ((int)p.S.size() >= kMaxS) break;
916
+ if (r >= g.N || seen.count(r)) continue;
917
+ seen.insert(r);
918
+ p.S.push_back(r);
919
+ }
920
+ int m = (int)p.S.size();
921
+
922
+ CUDA_CHECK(cudaMemcpy(g.S, p.S.data(), (size_t)m * sizeof(uint32_t), cudaMemcpyHostToDevice));
923
+ eqprop_fused<<<1, kMaxS>>>(g.N, g.k, g.theta, g.omega, g.amp,
924
+ g.off, g.col, g.K, g.S, m, p.n_prompt, g.nudge_id, g.nudge_beta, 0, 0.0,
925
+ kDt, nullptr, 1, n_rk4, neutral, g.settle_out, p.n_prompt, g_omega_scale);
926
+ CUDA_CHECK(cudaGetLastError());
927
+ CUDA_CHECK(cudaDeviceSynchronize());
928
+
929
+ std::vector<double> out((size_t)(3 * kMaxS + 1), 0.0);
930
+ CUDA_CHECK(cudaMemcpy(out.data(), g.settle_out, out.size() * sizeof(double),
931
+ cudaMemcpyDeviceToHost));
932
+
933
+ for (int li = first_read; li < m; ++li) {
934
+ SettleRow r;
935
+ r.id = p.S[li];
936
+ r.theta = out[(size_t)li];
937
+ r.theta0 = out[(size_t)kMaxS + li];
938
+ r.omega = out[(size_t)2 * kMaxS + li];
939
+ rows.push_back(r);
940
+ }
941
+ return out[(size_t)3 * kMaxS];
942
+ }
943
+
944
+ static std::vector<uint32_t> parse_id_list(const std::string& v) {
945
+ std::vector<uint32_t> ids;
946
+ size_t i = 0;
947
+ while (i < v.size()) {
948
+ size_t j = v.find(',', i);
949
+ if (j == std::string::npos) j = v.size();
950
+ if (j > i) {
951
+ const std::string tok = v.substr(i, j - i);
952
+ char* endp = nullptr;
953
+ unsigned long long x = std::strtoull(tok.c_str(), &endp, 10);
954
+ if (endp && endp != tok.c_str()) ids.push_back((uint32_t)x);
955
+ }
956
+ i = j + 1;
957
+ }
958
+ return ids;
959
+ }
960
+
961
+ // Line protocol on stdin, one request per line:
962
+ // settle prompt=<ids> [read=<ids>] [steps=N] [fill=N]
963
+ // ping | quit
964
+ // Response:
965
+ // mean <theta_bar>
966
+ // n <rows>
967
+ // <id> <theta> <theta0> <omega> <disp> <align>
968
+ // ...
969
+ // end
970
+ // disp is the coupling-only phase shift (omega*T removed); align is
971
+ // cos(theta - mean), the quantity the nudge drives targets toward in training.
972
+ static int run_settle(GpuState& g, const Args& a) {
973
+ std::printf("[settle] ready N=%u k=%u rk4=%d fill=%d kMaxS=%d\n", g.N, g.k, a.settle_rk4,
974
+ a.settle_fill, kMaxS);
975
+ std::puts("[settle] read-only: no theta writeback, no amp write, no K update");
976
+ std::fflush(stdout);
977
+
978
+ std::string line;
979
+ while (std::getline(std::cin, line)) {
980
+ while (!line.empty() && (line.back() == 13 || line.back() == 10)) line.pop_back();
981
+ if (line.empty()) continue;
982
+ if (line == "quit" || line == "exit") break;
983
+ if (line == "ping") {
984
+ std::puts("pong");
985
+ std::puts("end");
986
+ std::fflush(stdout);
987
+ continue;
988
+ }
989
+
990
+ std::vector<uint32_t> prompt, read;
991
+ int steps = a.settle_rk4;
992
+ int fill = a.settle_fill;
993
+ int neutral = 0;
994
+ bool bad = false;
995
+ size_t i = 0;
996
+ while (i < line.size()) {
997
+ while (i < line.size() && std::isspace((unsigned char)line[i])) ++i;
998
+ size_t j = i;
999
+ while (j < line.size() && !std::isspace((unsigned char)line[j])) ++j;
1000
+ if (j <= i) break;
1001
+ const std::string tok = line.substr(i, j - i);
1002
+ i = j;
1003
+ if (tok == "settle") continue;
1004
+ size_t eq = tok.find('=');
1005
+ if (eq == std::string::npos) {
1006
+ bad = true;
1007
+ break;
1008
+ }
1009
+ const std::string key = tok.substr(0, eq);
1010
+ const std::string val = tok.substr(eq + 1);
1011
+ if (key == "prompt") prompt = parse_id_list(val);
1012
+ else if (key == "read") read = parse_id_list(val);
1013
+ else if (key == "steps") steps = std::atoi(val.c_str());
1014
+ else if (key == "fill") fill = std::atoi(val.c_str());
1015
+ else if (key == "neutral") neutral = std::atoi(val.c_str()) ? 1 : 0;
1016
+ else {
1017
+ bad = true;
1018
+ break;
1019
+ }
1020
+ }
1021
+ if (bad || prompt.empty()) {
1022
+ std::puts(bad ? "err bad request" : "err empty prompt");
1023
+ std::puts("end");
1024
+ std::fflush(stdout);
1025
+ continue;
1026
+ }
1027
+ if (steps < 1) steps = 1;
1028
+ if (steps > 200) steps = 200;
1029
+ if (fill < 0) fill = 0;
1030
+
1031
+ // Chunk the read set so every launch stays inside kMaxS.
1032
+ int n_prompt_est = 0;
1033
+ {
1034
+ std::unordered_set<uint32_t> u;
1035
+ for (uint32_t t : prompt) {
1036
+ if (t < g.N && u.insert(t).second) ++n_prompt_est;
1037
+ }
1038
+ }
1039
+ if (n_prompt_est > kMaxS) n_prompt_est = kMaxS;
1040
+ int used = n_prompt_est + fill;
1041
+ if (used > kMaxS - 1) used = kMaxS - 1;
1042
+ int budget = kMaxS - used;
1043
+ if (budget < 1) budget = 1;
1044
+
1045
+ std::vector<SettleRow> all;
1046
+ double mean = 0.0;
1047
+ if (read.empty()) {
1048
+ std::vector<SettleRow> rows;
1049
+ mean = launch_settle(g, prompt, fill, read, steps, neutral, rows);
1050
+ } else {
1051
+ for (size_t off = 0; off < read.size(); off += (size_t)budget) {
1052
+ size_t n = std::min((size_t)budget, read.size() - off);
1053
+ std::vector<uint32_t> chunk(read.begin() + off, read.begin() + off + n);
1054
+ std::vector<SettleRow> rows;
1055
+ mean = launch_settle(g, prompt, fill, chunk, steps, neutral, rows);
1056
+ all.insert(all.end(), rows.begin(), rows.end());
1057
+ }
1058
+ }
1059
+
1060
+ const double T = (double)steps * kDt;
1061
+ std::printf("mean %.17g\n", mean);
1062
+ std::printf("n %zu\n", all.size());
1063
+ for (const SettleRow& r : all) {
1064
+ double disp = wrap_pi_host(r.theta - r.theta0 - r.omega * T);
1065
+ double align = std::cos(r.theta - mean);
1066
+ std::printf("%u %.17g %.17g %.17g %.17g %.17g\n", r.id, r.theta, r.theta0, r.omega,
1067
+ disp, align);
1068
+ }
1069
+ std::puts("end");
1070
+ std::fflush(stdout);
1071
+ }
1072
+ std::puts("[settle] bye");
1073
+ return 0;
1074
+ }
1075
+
1076
+ static int train_hard(GpuState& g, const Args& a, const std::string& path) {
1077
+ std::ifstream file(path, std::ios::binary);
1078
+ if (!file) {
1079
+ std::cerr << "corpus not found: " << path << "\n";
1080
+ return 1;
1081
+ }
1082
+ uint32_t nrec = 0;
1083
+ file.read((char*)&nrec, 4);
1084
+ std::printf("[train] hard %s records=%u\n", path.c_str(), nrec);
1085
+ int steps = 0;
1086
+ for (uint32_t r = 0; r < nrec; ++r) {
1087
+ if (a.max_steps > 0 && steps >= a.max_steps) break;
1088
+ uint16_t len = 0;
1089
+ file.read((char*)&len, 2);
1090
+ if (len < 2) {
1091
+ file.seekg(len * 4, std::ios::cur);
1092
+ continue;
1093
+ }
1094
+ std::vector<uint32_t> toks(len);
1095
+ file.read((char*)toks.data(), len * 4);
1096
+ uint32_t target = toks.back();
1097
+ std::vector<uint32_t> prompt(toks.begin(), toks.end() - 1);
1098
+ std::vector<std::pair<uint32_t, double>> nudges{{target, a.beta}};
1099
+ auto pack = pack_step(prompt, nudges, g.N);
1100
+ auto t0 = std::chrono::high_resolution_clock::now();
1101
+ launch_step(g, pack, a.lr);
1102
+ CUDA_CHECK(cudaDeviceSynchronize());
1103
+ auto t1 = std::chrono::high_resolution_clock::now();
1104
+ double us = std::chrono::duration<double, std::micro>(t1 - t0).count();
1105
+ steps++;
1106
+ if (a.log_every > 0 && steps % a.log_every == 0) {
1107
+ std::printf(" [step %d] target=%u |S|=%zu n_prompt=%d %.1f us\n", steps, target,
1108
+ pack.S.size(), pack.n_prompt, us);
1109
+ }
1110
+ if (a.save_every > 0 && steps % a.save_every == 0) save_snapshot(g, a, steps);
1111
+ }
1112
+ save_snapshot(g, a, steps);
1113
+ std::printf("[train] done steps=%d\n", steps);
1114
+ return 0;
1115
+ }
1116
+
1117
+ static int train_sds1(GpuState& g, const Args& a, std::ifstream& file, const std::string& path) {
1118
+ uint32_t nrec = 0;
1119
+ file.read((char*)&nrec, 4);
1120
+ std::printf("[train] SDS1 %s records=%u\n", path.c_str(), nrec);
1121
+ int steps = 0;
1122
+ for (uint32_t r = 0; r < nrec; ++r) {
1123
+ if (a.max_steps > 0 && steps >= a.max_steps) break;
1124
+ uint16_t plen = 0, nsoft = 0;
1125
+ file.read((char*)&plen, 2);
1126
+ std::vector<uint32_t> prompt(plen);
1127
+ if (plen) file.read((char*)prompt.data(), plen * 4);
1128
+ file.read((char*)&nsoft, 2);
1129
+ std::vector<uint32_t> ids(nsoft);
1130
+ std::vector<float> pr(nsoft);
1131
+ if (nsoft) {
1132
+ file.read((char*)ids.data(), nsoft * 4);
1133
+ file.read((char*)pr.data(), nsoft * 4);
1134
+ }
1135
+ if (plen < 1 || nsoft < 1) continue;
1136
+ std::vector<std::pair<uint32_t, double>> nudges;
1137
+ for (uint16_t i = 0; i < nsoft; ++i) {
1138
+ if (pr[i] > 0.0f) nudges.emplace_back(ids[i], a.beta * (double)pr[i]);
1139
+ }
1140
+ if (nudges.empty()) continue;
1141
+ auto pack = pack_step(prompt, nudges, g.N);
1142
+ auto t0 = std::chrono::high_resolution_clock::now();
1143
+ launch_step(g, pack, a.lr);
1144
+ CUDA_CHECK(cudaDeviceSynchronize());
1145
+ auto t1 = std::chrono::high_resolution_clock::now();
1146
+ double us = std::chrono::duration<double, std::micro>(t1 - t0).count();
1147
+ steps++;
1148
+ if (a.log_every > 0 && steps % a.log_every == 0) {
1149
+ std::printf(" [distill %d] |S|=%zu soft=%zu %.1f us\n", steps, pack.S.size(),
1150
+ nudges.size(), us);
1151
+ }
1152
+ if (a.save_every > 0 && steps % a.save_every == 0) save_snapshot(g, a, steps);
1153
+ }
1154
+ save_snapshot(g, a, steps);
1155
+ std::printf("[train] distill done steps=%d\n", steps);
1156
+ return 0;
1157
+ }
1158
+
1159
+ struct HardTail {
1160
+ std::string path;
1161
+ FILE* f = nullptr;
1162
+ bool live = false;
1163
+ bool wrap = false;
1164
+ uint64_t emitted = 0;
1165
+
1166
+ bool ensure() {
1167
+ if (f) return true;
1168
+ f = std::fopen(path.c_str(), "rb");
1169
+ if (!f) return false;
1170
+ uint32_t nrec = 0;
1171
+ if (std::fread(&nrec, 4, 1, f) != 1) {
1172
+ std::fclose(f);
1173
+ f = nullptr;
1174
+ return false;
1175
+ }
1176
+ std::printf("[data] open hard %s header_nrec=%u live=%d wrap=%d\n", path.c_str(), nrec,
1177
+ (int)live, (int)wrap);
1178
+ return true;
1179
+ }
1180
+
1181
+ bool try_read(std::vector<uint32_t>& toks) {
1182
+ if (!ensure()) return false;
1183
+ long start = std::ftell(f);
1184
+ uint16_t len = 0;
1185
+ if (std::fread(&len, 2, 1, f) != 1) {
1186
+ std::clearerr(f);
1187
+ if (wrap) {
1188
+ std::fseek(f, 4, SEEK_SET);
1189
+ start = std::ftell(f);
1190
+ if (std::fread(&len, 2, 1, f) != 1) return false;
1191
+ } else {
1192
+ std::fseek(f, start, SEEK_SET);
1193
+ return false;
1194
+ }
1195
+ }
1196
+ if (len < 2 || len > 2048) {
1197
+ std::clearerr(f);
1198
+ std::fseek(f, start, SEEK_SET);
1199
+ return false;
1200
+ }
1201
+ toks.resize(len);
1202
+ if (std::fread(toks.data(), 4, (size_t)len, f) != (size_t)len) {
1203
+ std::clearerr(f);
1204
+ std::fseek(f, start, SEEK_SET);
1205
+ toks.clear();
1206
+ return false;
1207
+ }
1208
+ emitted++;
1209
+ return true;
1210
+ }
1211
+ };
1212
+
1213
+ struct Sds1Tail {
1214
+ std::string path;
1215
+ FILE* f = nullptr;
1216
+ bool wrap = true;
1217
+ uint64_t emitted = 0;
1218
+ static constexpr long kHdr = 8;
1219
+
1220
+ bool ensure() {
1221
+ if (f) return true;
1222
+ f = std::fopen(path.c_str(), "rb");
1223
+ if (!f) return false;
1224
+ char mag[4] = {};
1225
+ uint32_t nrec = 0;
1226
+ if (std::fread(mag, 1, 4, f) != 4 || std::memcmp(mag, "SDS1", 4) != 0 ||
1227
+ std::fread(&nrec, 4, 1, f) != 1) {
1228
+ std::fclose(f);
1229
+ f = nullptr;
1230
+ return false;
1231
+ }
1232
+ std::printf("[data] open SDS1 %s header_nrec=%u\n", path.c_str(), nrec);
1233
+ return true;
1234
+ }
1235
+
1236
+ bool try_read(std::vector<uint32_t>& prompt, std::vector<uint32_t>& ids, std::vector<float>& pr) {
1237
+ if (!ensure()) return false;
1238
+ long start = std::ftell(f);
1239
+ uint16_t plen = 0, nsoft = 0;
1240
+ auto rewind_incomplete = [&]() {
1241
+ std::clearerr(f);
1242
+ std::fseek(f, start, SEEK_SET);
1243
+ };
1244
+ if (std::fread(&plen, 2, 1, f) != 1) {
1245
+ std::clearerr(f);
1246
+ if (wrap) {
1247
+ std::fseek(f, kHdr, SEEK_SET);
1248
+ start = std::ftell(f);
1249
+ if (std::fread(&plen, 2, 1, f) != 1) return false;
1250
+ } else {
1251
+ std::fseek(f, start, SEEK_SET);
1252
+ return false;
1253
+ }
1254
+ }
1255
+ prompt.assign(plen, 0);
1256
+ if (plen && std::fread(prompt.data(), 4, plen, f) != (size_t)plen) {
1257
+ rewind_incomplete();
1258
+ return false;
1259
+ }
1260
+ if (std::fread(&nsoft, 2, 1, f) != 1) {
1261
+ rewind_incomplete();
1262
+ return false;
1263
+ }
1264
+ ids.assign(nsoft, 0);
1265
+ pr.assign(nsoft, 0.0f);
1266
+ if (nsoft) {
1267
+ if (std::fread(ids.data(), 4, nsoft, f) != (size_t)nsoft ||
1268
+ std::fread(pr.data(), 4, nsoft, f) != (size_t)nsoft) {
1269
+ rewind_incomplete();
1270
+ return false;
1271
+ }
1272
+ }
1273
+ if (plen < 1 || nsoft < 1) return false;
1274
+ emitted++;
1275
+ return true;
1276
+ }
1277
+ };
1278
+
1279
+ struct NysaRec {
1280
+ std::vector<uint32_t> spec;
1281
+ std::vector<uint8_t> bits;
1282
+ };
1283
+
1284
+ static bool read_nysa_rec(FILE* f, NysaRec& r) {
1285
+ uint16_t kind = 0, slen = 0, nbytes = 0, nbits = 0, ntrace = 0;
1286
+ int64_t x = 0, y = 0;
1287
+ uint8_t flags = 0;
1288
+ if (std::fread(&kind, 2, 1, f) != 1) return false;
1289
+ if (std::fread(&slen, 2, 1, f) != 1) return false;
1290
+ r.spec.assign(slen, 0);
1291
+ if (slen && std::fread(r.spec.data(), 4, slen, f) != (size_t)slen) return false;
1292
+ if (std::fread(&x, 8, 1, f) != 1 || std::fread(&y, 8, 1, f) != 1) return false;
1293
+ if (std::fread(&nbytes, 2, 1, f) != 1) return false;
1294
+ if (nbytes) std::fseek(f, nbytes, SEEK_CUR);
1295
+ if (std::fread(&nbits, 2, 1, f) != 1) return false;
1296
+ r.bits.assign(nbits, 0);
1297
+ if (nbits && std::fread(r.bits.data(), 1, nbits, f) != (size_t)nbits) return false;
1298
+ if (std::fread(&flags, 1, 1, f) != 1) return false;
1299
+ if (std::fread(&ntrace, 2, 1, f) != 1) return false;
1300
+ if (ntrace) std::fseek(f, ntrace, SEEK_CUR);
1301
+ (void)kind;
1302
+ (void)x;
1303
+ (void)y;
1304
+ (void)flags;
1305
+ return true;
1306
+ }
1307
+
1308
+ struct NysaTail {
1309
+ std::string path;
1310
+ FILE* f = nullptr;
1311
+ bool wrap = true;
1312
+ uint64_t emitted = 0;
1313
+ static constexpr uint32_t kCompilerBase = 3800000u;
1314
+ static constexpr int kActiveBits = 64;
1315
+ static constexpr long kHdr = 12;
1316
+
1317
+ bool ensure() {
1318
+ if (f) return true;
1319
+ f = std::fopen(path.c_str(), "rb");
1320
+ if (!f) return false;
1321
+ char mag[4] = {};
1322
+ uint32_t ver = 0, nrec = 0;
1323
+ if (std::fread(mag, 1, 4, f) != 4 || std::memcmp(mag, "NYSA", 4) != 0 ||
1324
+ std::fread(&ver, 4, 1, f) != 1 || std::fread(&nrec, 4, 1, f) != 1) {
1325
+ std::fclose(f);
1326
+ f = nullptr;
1327
+ return false;
1328
+ }
1329
+ std::printf("[data] open NYSA %s header_nrec=%u\n", path.c_str(), nrec);
1330
+ return true;
1331
+ }
1332
+
1333
+ bool try_read(NysaRec& r) {
1334
+ if (!ensure()) return false;
1335
+ long start = std::ftell(f);
1336
+ if (!read_nysa_rec(f, r)) {
1337
+ std::clearerr(f);
1338
+ if (wrap) {
1339
+ std::fseek(f, kHdr, SEEK_SET);
1340
+ if (!read_nysa_rec(f, r)) return false;
1341
+ } else {
1342
+ std::fseek(f, start, SEEK_SET);
1343
+ return false;
1344
+ }
1345
+ }
1346
+ if (r.spec.empty() && r.bits.empty()) return false;
1347
+ emitted++;
1348
+ return true;
1349
+ }
1350
+ };
1351
+
1352
+ struct NysvRec {
1353
+ std::vector<uint32_t> text;
1354
+ std::vector<uint32_t> frames;
1355
+ };
1356
+
1357
+ static bool read_nysv_rec(FILE* f, NysvRec& r) {
1358
+ uint16_t tlen = 0, nfl = 0;
1359
+ uint32_t rate = 0, plen = 0;
1360
+ if (std::fread(&tlen, 2, 1, f) != 1) return false;
1361
+ r.text.assign(tlen, 0);
1362
+ if (tlen && std::fread(r.text.data(), 4, tlen, f) != (size_t)tlen) return false;
1363
+ if (std::fread(&rate, 4, 1, f) != 1) return false;
1364
+ if (std::fread(&nfl, 2, 1, f) != 1) return false;
1365
+ r.frames.assign(nfl, 0);
1366
+ if (nfl && std::fread(r.frames.data(), 4, nfl, f) != (size_t)nfl) return false;
1367
+ if (std::fread(&plen, 4, 1, f) != 1) return false;
1368
+ if (plen) std::fseek(f, plen, SEEK_CUR);
1369
+ (void)rate;
1370
+ return true;
1371
+ }
1372
+
1373
+ struct NysvTail {
1374
+ std::string path;
1375
+ FILE* f = nullptr;
1376
+ bool wrap = true;
1377
+ uint64_t emitted = 0;
1378
+ uint32_t record_count = 0;
1379
+ static constexpr long kHdr = 12;
1380
+
1381
+ bool ensure() {
1382
+ if (f) return true;
1383
+ f = std::fopen(path.c_str(), "rb");
1384
+ if (!f) return false;
1385
+ char mag[4] = {};
1386
+ uint32_t ver = 0, nrec = 0;
1387
+ if (std::fread(mag, 1, 4, f) != 4 || std::memcmp(mag, "NYSV", 4) != 0 ||
1388
+ std::fread(&ver, 4, 1, f) != 1 || std::fread(&nrec, 4, 1, f) != 1) {
1389
+ std::fclose(f);
1390
+ f = nullptr;
1391
+ return false;
1392
+ }
1393
+ std::printf("[data] open NYSV %s header_nrec=%u\n", path.c_str(), nrec);
1394
+ record_count = nrec;
1395
+ return true;
1396
+ }
1397
+
1398
+ bool try_read(NysvRec& r) {
1399
+ if (!ensure()) return false;
1400
+ long start = std::ftell(f);
1401
+ if (!read_nysv_rec(f, r)) {
1402
+ std::clearerr(f);
1403
+ if (wrap) {
1404
+ std::fseek(f, kHdr, SEEK_SET);
1405
+ if (!read_nysv_rec(f, r)) return false;
1406
+ } else {
1407
+ std::fseek(f, start, SEEK_SET);
1408
+ return false;
1409
+ }
1410
+ }
1411
+ if (r.text.empty() || r.frames.empty()) return false;
1412
+ emitted++;
1413
+ return true;
1414
+ }
1415
+ };
1416
+
1417
+ static uint32_t compiler_base_for_n(uint32_t N) {
1418
+ const uint32_t base = 3800000u;
1419
+ const uint32_t bits = 64u;
1420
+ if (N > base + bits) return base;
1421
+ return N > bits + 256u ? N - bits - 1u : 256u;
1422
+ }
1423
+
1424
+ static uint32_t speech_base_for_n(uint32_t N) {
1425
+ const uint32_t base = 4300000u;
1426
+ const uint32_t need = 400u * 256u;
1427
+ if (N > base + need) return base;
1428
+ return N > need + 256u ? N - need - 1u : 256u;
1429
+ }
1430
+
1431
+ // Vocoder codes whose atom is flat (digital silence). Frames on these codes
1432
+ // stay in the records -- t indexing is untouched -- but are never nudged and
1433
+ // never used as negatives. About two-thirds of the speech corpus is literal
1434
+ // silence (frame stdev 0), and the silence node (t, code) is gold for most
1435
+ // records at every t, so each text pulled that one node toward its own phase:
1436
+ // +w against +w, the shared-register conflict the compiler band also has.
1437
+ //
1438
+ // Derived from the installed codebook at startup, so it follows whatever
1439
+ // vocoder is in the slots dir. An untrained codebook is mostly flat atoms
1440
+ // (the old one had 235/256), which would make "skip silence" mean "skip
1441
+ // nearly everything" -- so above kMaxSilenceCodes the skip is disabled and
1442
+ // training behaves exactly as it did before.
1443
+ static bool g_silence_code[256] = {false};
1444
+ static int g_n_silence = 0;
1445
+ static constexpr int kMaxSilenceCodes = 64;
1446
+
1447
+ static void load_silence_codes(const std::string& slots_dir) {
1448
+ std::string p = slots_dir;
1449
+ if (!p.empty() && p.back() != '/') p.push_back('/');
1450
+ p += "speech_vocoder.nyvc";
1451
+ std::ifstream f(p, std::ios::binary);
1452
+ if (!f) {
1453
+ std::printf("[speech] no vocoder at %s -- silence frames will be nudged\n", p.c_str());
1454
+ return;
1455
+ }
1456
+ char mag[4] = {};
1457
+ uint32_t hdr[4] = {};
1458
+ f.read(mag, 4);
1459
+ f.read((char*)hdr, sizeof(hdr));
1460
+ const uint32_t n_codes = hdr[1], fs = hdr[2];
1461
+ if (!f || std::memcmp(mag, "NYVC", 4) != 0 || n_codes != 256u || fs == 0 || fs > 4096) {
1462
+ std::printf("[speech] unreadable vocoder %s -- silence frames will be nudged\n", p.c_str());
1463
+ return;
1464
+ }
1465
+ bool flat[256] = {false};
1466
+ int count = 0;
1467
+ std::vector<unsigned char> atom(fs);
1468
+ for (uint32_t c = 0; c < n_codes; ++c) {
1469
+ f.read((char*)atom.data(), (std::streamsize)fs);
1470
+ if (!f) {
1471
+ std::printf("[speech] short vocoder %s -- silence frames will be nudged\n", p.c_str());
1472
+ return;
1473
+ }
1474
+ double m = 0.0;
1475
+ for (uint32_t b = 0; b < fs; ++b) m += atom[b];
1476
+ m /= (double)fs;
1477
+ double v = 0.0;
1478
+ for (uint32_t b = 0; b < fs; ++b) v += (atom[b] - m) * (atom[b] - m);
1479
+ if (std::sqrt(v / (double)fs) < 1.0) {
1480
+ flat[c] = true;
1481
+ ++count;
1482
+ }
1483
+ }
1484
+ if (count > kMaxSilenceCodes) {
1485
+ std::printf("[speech] vocoder has %d flat atoms (untrained codebook) -- silence skip "
1486
+ "DISABLED\n", count);
1487
+ return;
1488
+ }
1489
+ std::memcpy(g_silence_code, flat, sizeof(flat));
1490
+ g_n_silence = count;
1491
+ std::printf("[speech] silence codes (not nudged):");
1492
+ for (int c = 0; c < 256; ++c) {
1493
+ if (g_silence_code[c]) std::printf(" %d", c);
1494
+ }
1495
+ std::printf(" [%d]\n", count);
1496
+ }
1497
+
1498
+ static bool speech_is_silence(uint32_t id, uint32_t N) {
1499
+ if (g_n_silence == 0) return false;
1500
+ uint32_t base = speech_base_for_n(N);
1501
+ uint32_t c = id >= base ? (id - base) % 256u : (id & 255u);
1502
+ return g_silence_code[c];
1503
+ }
1504
+
1505
+ static bool next_voiced_speech(NysvTail& tail, NysvRec& rec,
1506
+ std::vector<uint32_t>& voiced, uint32_t N,
1507
+ uint64_t& skipped) {
1508
+ if (!tail.ensure()) return false;
1509
+ // Advance past valid but silent alignment rows within this slot. A bounded
1510
+ // pass rejects wholly silent/unusable sources without reducing the diet.
1511
+ for (uint32_t attempt=0; attempt<tail.record_count; ++attempt) {
1512
+ if (!tail.try_read(rec)) { ++skipped; continue; }
1513
+ voiced.clear();
1514
+ for (uint32_t id : rec.frames)
1515
+ if (id<N && !speech_is_silence(id,N)) voiced.push_back(id);
1516
+ if (!voiced.empty() && !rec.text.empty()) return true;
1517
+ ++skipped;
1518
+ }
1519
+ return false;
1520
+ }
1521
+
1522
+ // Every frame ID that is gold for ANY record, loaded once at startup.
1523
+ //
1524
+ // A physical (t, code) node is reused by every record whose gold happens
1525
+ // to land on it, so a negative built from ONE record's frames can be a
1526
+ // different record's gold. Measured on the live corpus: 39.9% of the
1527
+ // positive pushes were fought by a negative push on the identical node,
1528
+ // which is why K on text->gold sat flat (mean 0.195, essentially k0) while
1529
+ // K on text->negative drifted up (mean 0.230) after 50k steps post-reset.
1530
+ // Checking against the corpus-wide gold set instead of just the current
1531
+ // record's frames removes that conflict at the data level. Loaded once
1532
+ // because NYSV is not live-tailed -- regenerating it always requires a
1533
+ // stop, unlike the mix corpus.
1534
+ static std::unordered_set<uint32_t> g_global_gold;
1535
+
1536
+ static void load_global_gold_frames(const std::string& slots_dir) {
1537
+ g_global_gold.clear();
1538
+ const char* names[] = {"speech_align.nysv", "speech_slot.nysv"};
1539
+ for (const char* name : names) {
1540
+ std::string p = slots_dir;
1541
+ if (!p.empty() && p.back() != '/') p.push_back('/');
1542
+ p += name;
1543
+ FILE* f = std::fopen(p.c_str(), "rb");
1544
+ if (!f) continue;
1545
+ char mag[4] = {};
1546
+ uint32_t ver = 0, nrec = 0;
1547
+ if (std::fread(mag, 1, 4, f) != 4 || std::memcmp(mag, "NYSV", 4) != 0 ||
1548
+ std::fread(&ver, 4, 1, f) != 1 || std::fread(&nrec, 4, 1, f) != 1) {
1549
+ std::fclose(f);
1550
+ continue;
1551
+ }
1552
+ NysvRec r;
1553
+ while (read_nysv_rec(f, r)) {
1554
+ for (uint32_t fid : r.frames) g_global_gold.insert(fid);
1555
+ }
1556
+ std::fclose(f);
1557
+ }
1558
+ std::printf("[speech] global gold frames: %zu (negatives will avoid these)\n",
1559
+ g_global_gold.size());
1560
+ }
1561
+
1562
+ static bool speech_is_global_gold(uint32_t id) {
1563
+ return g_global_gold.count(id) != 0;
1564
+ }
1565
+
1566
+ // Neighbours of each gold code, used as NEGATIVE nudge targets.
1567
+ //
1568
+ // These used to be appended to the prompt, where they were ~97.5% of the
1569
+ // clamped set (mean 91.2 negatives against 2.3 text ids). The kernel builds
1570
+ // the nudge target as the circular mean over every prompt node, so sh_mean
1571
+ // was effectively the negatives' own phase: each step pulled the gold codes
1572
+ // toward their own neighbours and the text contributed 2.5% of the target.
1573
+ // The contrastive dc on text->gold was then zero-mean, which is why K on
1574
+ // those edges sat at the k0 the wiring pass wrote (median 0.2006, 50.2%
1575
+ // above k0) after 25k steps. As signed targets they push away from the
1576
+ // text's phase instead -- the same shape NYSA already uses for 0-bits.
1577
+ static std::vector<uint32_t> speech_negatives(const std::vector<uint32_t>& frames,
1578
+ uint32_t N) {
1579
+ std::vector<uint32_t> negs;
1580
+ uint32_t base = speech_base_for_n(N);
1581
+ std::unordered_set<uint32_t> gold(frames.begin(), frames.end());
1582
+ auto is_gold = [&](uint32_t x) { return gold.count(x) != 0; };
1583
+ int extra = 0;
1584
+ for (uint32_t id : frames) {
1585
+ if (extra >= 96) break;
1586
+ uint32_t t = 0, c = id & 255u;
1587
+ if (id >= base) {
1588
+ t = (id - base) / 256u;
1589
+ c = (id - base) % 256u;
1590
+ }
1591
+ // One negative per gold frame, not two: the dropped +127 offset
1592
+ // doubled the total nudge mass on the negative side relative to
1593
+ // gold for no benefit -- the 256 vocoder codes are unordered
1594
+ // k-means cluster ids, not similarity-ranked, so +1 is no less a
1595
+ // 'confusable' example than +127 was.
1596
+ uint32_t n1 = base + t * 256u + ((c + 1u) & 255u);
1597
+ if (n1 < N && !is_gold(n1) && !speech_is_silence(n1, N) &&
1598
+ !speech_is_global_gold(n1)) {
1599
+ negs.push_back(n1);
1600
+ extra++;
1601
+ }
1602
+ }
1603
+ return negs;
1604
+ }
1605
+
1606
+ static int train_nysa_file(GpuState& g, const Args& a, const std::string& path) {
1607
+ NysaTail t;
1608
+ t.path = path;
1609
+ t.wrap = false;
1610
+ int steps = 0;
1611
+ NysaRec rec;
1612
+ while (t.try_read(rec)) {
1613
+ if (a.max_steps > 0 && steps >= a.max_steps) break;
1614
+ uint32_t cbase = compiler_base_for_n(g.N);
1615
+ std::vector<std::pair<uint32_t, double>> nudges;
1616
+ int nbits = (int)std::min(rec.bits.size(), (size_t)NysaTail::kActiveBits);
1617
+ for (int b = 0; b < nbits; ++b) {
1618
+ uint32_t id = cbase + (uint32_t)b;
1619
+ if (id >= g.N) break;
1620
+ nudges.emplace_back(id, rec.bits[b] ? a.beta : -a.beta);
1621
+ }
1622
+ if (nudges.empty()) continue;
1623
+ auto pack = pack_step(rec.spec, nudges, g.N);
1624
+ launch_step(g, pack, a.lr);
1625
+ CUDA_CHECK(cudaDeviceSynchronize());
1626
+ steps++;
1627
+ if (a.save_every > 0 && steps % a.save_every == 0) save_snapshot(g, a, steps);
1628
+ }
1629
+ save_snapshot(g, a, steps);
1630
+ std::printf("[train] NYSA done steps=%d\n", steps);
1631
+ return 0;
1632
+ }
1633
+
1634
+ static int train_nysv_file(GpuState& g, const Args& a, const std::string& path) {
1635
+ NysvTail t;
1636
+ t.path = path;
1637
+ t.wrap = false;
1638
+ int steps = 0;
1639
+ NysvRec rec;
1640
+ while (t.try_read(rec)) {
1641
+ if (a.max_steps > 0 && steps >= a.max_steps) break;
1642
+ std::vector<std::pair<uint32_t, double>> nudges;
1643
+ // Silence frames stay in the record (t is unchanged) but are not
1644
+ // targets; beta is spread over the voiced frames only.
1645
+ std::vector<uint32_t> voiced;
1646
+ for (uint32_t id : rec.frames) {
1647
+ if (id < g.N && !speech_is_silence(id, g.N)) voiced.push_back(id);
1648
+ }
1649
+ double w = voiced.empty() ? 0.0 : a.beta / (double)voiced.size();
1650
+ if (w > 0.0 && w < 0.25) w = 0.25;
1651
+ for (uint32_t id : voiced) nudges.emplace_back(id, w);
1652
+ if (nudges.empty() || rec.text.empty()) continue;
1653
+ for (uint32_t nid : speech_negatives(voiced, g.N)) {
1654
+ nudges.emplace_back(nid, -w);
1655
+ }
1656
+ // Prompt is the text alone, so sh_mean is the text's phase and the
1657
+ // contrastive update on text->gold finally has a consistent sign.
1658
+ auto pack = pack_step(rec.text, nudges, g.N);
1659
+ launch_step(g, pack, a.lr);
1660
+ CUDA_CHECK(cudaDeviceSynchronize());
1661
+ steps++;
1662
+ if (a.save_every > 0 && steps % a.save_every == 0) save_snapshot(g, a, steps);
1663
+ }
1664
+ save_snapshot(g, a, steps);
1665
+ std::printf("[train] NYSV done steps=%d\n", steps);
1666
+ return 0;
1667
+ }
1668
+
1669
+ #include "distill_stdin.h"
1670
+
1671
+ static int train_live(GpuState& g, const Args& a) {
1672
+ std::unique_ptr<DistillStdin> teacher;
1673
+ if (a.distill_stdin) {
1674
+ teacher = std::make_unique<DistillStdin>(g.N);
1675
+ std::puts("[stream] ready for Qwen SDS1 probabilities");
1676
+ }
1677
+ std::vector<HardTail> hards;
1678
+ std::vector<Sds1Tail> sds1s;
1679
+ std::vector<NysaTail> nysas;
1680
+ std::vector<NysvTail> nysvs;
1681
+ HardTail mix;
1682
+ mix.live = true;
1683
+ mix.wrap = false;
1684
+ mix.path = a.live_mix;
1685
+
1686
+ for (const auto& path : a.corpora) {
1687
+ if (!a.live_mix.empty() && path == a.live_mix) continue;
1688
+ std::ifstream probe(path, std::ios::binary);
1689
+ char magic[4] = {};
1690
+ probe.read(magic, 4);
1691
+ bool sds1 = probe.gcount() == 4 && std::memcmp(magic, "SDS1", 4) == 0;
1692
+ bool nysa = probe.gcount() == 4 && std::memcmp(magic, "NYSA", 4) == 0;
1693
+ bool nysv = probe.gcount() == 4 && std::memcmp(magic, "NYSV", 4) == 0;
1694
+ bool nysd = probe.gcount() == 4 && std::memcmp(magic, "NYSD", 4) == 0;
1695
+ probe.close();
1696
+ if (nysd) {
1697
+ std::printf("[data] skip NYSD dump envelope %s\n", path.c_str());
1698
+ } else if (nysa) {
1699
+ NysaTail t;
1700
+ t.path = path;
1701
+ nysas.push_back(std::move(t));
1702
+ } else if (nysv) {
1703
+ NysvTail t;
1704
+ t.path = path;
1705
+ nysvs.push_back(std::move(t));
1706
+ } else if (sds1) {
1707
+ Sds1Tail t;
1708
+ t.path = path;
1709
+ t.wrap = true;
1710
+ sds1s.push_back(std::move(t));
1711
+ } else {
1712
+ HardTail t;
1713
+ t.path = path;
1714
+ t.wrap = true;
1715
+ hards.push_back(std::move(t));
1716
+ }
1717
+ }
1718
+
1719
+ if (teacher && !sds1s.empty())
1720
+ throw std::runtime_error("--distill-stdin replaces the SDS1 slot; do not also supply an SDS1 corpus");
1721
+ if (a.require_full_curriculum && (!teacher || mix.path.empty() || hards.size()!=2 || nysas.size()!=1 || nysvs.size()!=2))
1722
+ throw std::runtime_error("Complete curriculum requires mix, two hard corpora, Qwen, compiler, and both speech inputs");
1723
+
1724
+ int steps = a.resume_steps;
1725
+ size_t hi = 0, di = 0, ci = 0, vi = 0;
1726
+ int idle = 0;
1727
+ std::printf("[train] live loop static_hard=%zu sds1=%zu nysa=%zu nysv=%zu mix=%s resume_steps=%d\n",
1728
+ hards.size(), sds1s.size(), nysas.size(), nysvs.size(),
1729
+ mix.path.empty() ? "(none)" : mix.path.c_str(), a.resume_steps);
1730
+
1731
+ double c_mix = 0, c_hard = 0, c_dist = 0, c_comp = 0, c_speech = 0;
1732
+ auto log_rot = [&]() {
1733
+ if (a.log_every <= 0 || steps % a.log_every != 0) return;
1734
+ std::printf(" [step %d] mix=%.5f hard=%.5f distill=%.5f compiler=%.5f speech=%.5f mix_recs=%llu\n",
1735
+ steps, c_mix, c_hard, c_dist, c_comp, c_speech,
1736
+ (unsigned long long)mix.emitted);
1737
+ std::fflush(stdout);
1738
+ };
1739
+
1740
+ auto hard_step = [&](const std::vector<uint32_t>& toks, const char* src) {
1741
+ if (toks.size() < 2) return;
1742
+ uint32_t target = toks.back();
1743
+ std::vector<uint32_t> prompt(toks.begin(), toks.end() - 1);
1744
+ std::vector<std::pair<uint32_t, double>> nudges{{target, a.beta}};
1745
+ auto pack = pack_step(prompt, nudges, g.N);
1746
+ double contrast = launch_step(g, pack, a.lr);
1747
+ CUDA_CHECK(cudaDeviceSynchronize());
1748
+ steps++;
1749
+ if (std::strcmp(src, "mix") == 0) c_mix = contrast;
1750
+ else c_hard = contrast;
1751
+ log_rot();
1752
+ if (a.save_every > 0 && steps % a.save_every == 0) save_snapshot(g, a, steps);
1753
+ };
1754
+
1755
+ while (a.max_steps < 0 || steps < a.max_steps) {
1756
+ int rotation_start = steps;
1757
+ bool did = false;
1758
+ std::vector<uint32_t> toks;
1759
+ std::vector<uint32_t> teacher_prompt, teacher_ids;
1760
+ std::vector<float> teacher_probs;
1761
+ // One teacher record paces one complete rotation. In particular, do not
1762
+ // train other slots while Qwen loads, stalls, or has already finished.
1763
+ if (teacher) {
1764
+ int status;
1765
+ do {
1766
+ status = teacher->next(teacher_prompt, teacher_ids, teacher_probs);
1767
+ if (!status) std::this_thread::sleep_for(std::chrono::milliseconds(5));
1768
+ } while (!status);
1769
+ if (status < 0) {
1770
+ if (!teacher->emitted) throw std::runtime_error("teacher closed without records");
1771
+ std::puts("[stream] teacher EOF; saving and stopping");
1772
+ break;
1773
+ }
1774
+ }
1775
+
1776
+ if (!mix.path.empty() && mix.try_read(toks)) {
1777
+ hard_step(toks, "mix");
1778
+ did = true;
1779
+ idle = 0;
1780
+ if (a.max_steps > 0 && steps >= a.max_steps) break;
1781
+ } else if (a.require_full_curriculum) {
1782
+ throw std::runtime_error("Required mixed-language slot has no complete record");
1783
+ }
1784
+
1785
+ if (!hards.empty()) {
1786
+ for (size_t n = 0; n < hards.size(); ++n) {
1787
+ size_t i = (hi + n) % hards.size();
1788
+ if (hards[i].try_read(toks)) {
1789
+ hard_step(toks, "hard");
1790
+ hi = (i + 1) % hards.size();
1791
+ did = true;
1792
+ idle = 0;
1793
+ break;
1794
+ }
1795
+ }
1796
+ if (a.max_steps > 0 && steps >= a.max_steps) break;
1797
+ }
1798
+
1799
+ if (teacher) {
1800
+ std::vector<std::pair<uint32_t, double>> nudges;
1801
+ for (size_t i = 0; i < teacher_ids.size(); ++i)
1802
+ nudges.emplace_back(teacher_ids[i], a.beta * teacher_probs[i]);
1803
+ auto pack = pack_step(teacher_prompt, nudges, g.N);
1804
+ c_dist = launch_step(g, pack, a.lr);
1805
+ CUDA_CHECK(cudaDeviceSynchronize());
1806
+ ++steps; did = true; idle = 0;
1807
+ log_rot();
1808
+ if (a.save_every > 0 && steps % a.save_every == 0) save_snapshot(g, a, steps);
1809
+ if (a.max_steps > 0 && steps >= a.max_steps) break;
1810
+ }
1811
+
1812
+ if (!sds1s.empty()) {
1813
+ std::vector<uint32_t> prompt, ids;
1814
+ std::vector<float> pr;
1815
+ for (size_t n = 0; n < sds1s.size(); ++n) {
1816
+ size_t i = (di + n) % sds1s.size();
1817
+ if (!sds1s[i].try_read(prompt, ids, pr)) continue;
1818
+ std::vector<std::pair<uint32_t, double>> nudges;
1819
+ for (size_t k = 0; k < ids.size(); ++k) {
1820
+ if (pr[k] > 0.0f) nudges.emplace_back(ids[k], a.beta * (double)pr[k]);
1821
+ }
1822
+ if (!nudges.empty()) {
1823
+ auto pack = pack_step(prompt, nudges, g.N);
1824
+ double contrast = launch_step(g, pack, a.lr);
1825
+ CUDA_CHECK(cudaDeviceSynchronize());
1826
+ steps++;
1827
+ did = true;
1828
+ idle = 0;
1829
+ c_dist = contrast;
1830
+ log_rot();
1831
+ if (a.save_every > 0 && steps % a.save_every == 0) save_snapshot(g, a, steps);
1832
+ }
1833
+ di = (i + 1) % sds1s.size();
1834
+ break;
1835
+ }
1836
+ if (a.max_steps > 0 && steps >= a.max_steps) break;
1837
+ }
1838
+
1839
+ if (!nysas.empty()) {
1840
+ NysaRec rec;
1841
+ for (size_t n = 0; n < nysas.size(); ++n) {
1842
+ size_t i = (ci + n) % nysas.size();
1843
+ if (!nysas[i].try_read(rec)) continue;
1844
+ uint32_t cbase = compiler_base_for_n(g.N);
1845
+ std::vector<std::pair<uint32_t, double>> nudges;
1846
+ int nbits = (int)std::min(rec.bits.size(), (size_t)NysaTail::kActiveBits);
1847
+ for (int b = 0; b < nbits; ++b) {
1848
+ uint32_t id = cbase + (uint32_t)b;
1849
+ if (id >= g.N) break;
1850
+ nudges.emplace_back(id, rec.bits[b] ? a.beta : -a.beta);
1851
+ }
1852
+ if (!nudges.empty()) {
1853
+ auto pack = pack_step(rec.spec, nudges, g.N);
1854
+ double contrast = launch_step(g, pack, a.lr);
1855
+ CUDA_CHECK(cudaDeviceSynchronize());
1856
+ steps++;
1857
+ did = true;
1858
+ idle = 0;
1859
+ c_comp = contrast;
1860
+ log_rot();
1861
+ if (a.save_every > 0 && steps % a.save_every == 0) save_snapshot(g, a, steps);
1862
+ }
1863
+ ci = (i + 1) % nysas.size();
1864
+ break;
1865
+ }
1866
+ if (a.max_steps > 0 && steps >= a.max_steps) break;
1867
+ }
1868
+
1869
+ if (!nysvs.empty()) {
1870
+ NysvRec rec;
1871
+ for (size_t n = 0; n < nysvs.size(); ++n) {
1872
+ size_t i = (vi + n) % nysvs.size();
1873
+ std::vector<std::pair<uint32_t, double>> nudges;
1874
+ std::vector<uint32_t> voiced;
1875
+ if (a.require_full_curriculum) {
1876
+ uint64_t skipped=0;
1877
+ if (!next_voiced_speech(nysvs[i],rec,voiced,g.N,skipped))
1878
+ throw std::runtime_error("Required speech source has no voiced training records: "+nysvs[i].path);
1879
+ if (skipped) std::printf("[speech] skipped %llu silent/unusable rows within %s; speech slot retained\n",
1880
+ (unsigned long long)skipped,nysvs[i].path.c_str());
1881
+ } else {
1882
+ if (!nysvs[i].try_read(rec)) continue;
1883
+ for (uint32_t id : rec.frames)
1884
+ if (id < g.N && !speech_is_silence(id, g.N)) voiced.push_back(id);
1885
+ }
1886
+ double w = voiced.empty() ? 0.0 : a.beta / (double)voiced.size();
1887
+ if (w > 0.0 && w < 0.25) w = 0.25;
1888
+ for (uint32_t id : voiced) nudges.emplace_back(id, w);
1889
+ if (!nudges.empty() && !rec.text.empty()) {
1890
+ for (uint32_t nid : speech_negatives(voiced, g.N)) {
1891
+ nudges.emplace_back(nid, -w);
1892
+ }
1893
+ // Text-only prompt: sh_mean is the text's phase, not the
1894
+ // negatives'. This is the live rotation's NYSV step.
1895
+ auto pack = pack_step(rec.text, nudges, g.N);
1896
+ double contrast = launch_step(g, pack, a.lr);
1897
+ CUDA_CHECK(cudaDeviceSynchronize());
1898
+ steps++;
1899
+ did = true;
1900
+ idle = 0;
1901
+ c_speech = contrast;
1902
+ log_rot();
1903
+ if (a.save_every > 0 && steps % a.save_every == 0) save_snapshot(g, a, steps);
1904
+ }
1905
+ vi = (i + 1) % nysvs.size();
1906
+ break;
1907
+ }
1908
+ if (a.max_steps > 0 && steps >= a.max_steps) break;
1909
+ }
1910
+
1911
+ if (a.require_full_curriculum && steps-rotation_start != 5)
1912
+ throw std::runtime_error("Required five-slot rotation was incomplete; refusing reduced curriculum");
1913
+ if (a.save_first_rotation && rotation_start == a.resume_steps)
1914
+ save_snapshot(g, a, steps);
1915
+ if (did) continue;
1916
+ if (teacher) {
1917
+ std::this_thread::sleep_for(std::chrono::milliseconds(5));
1918
+ continue;
1919
+ }
1920
+
1921
+ bool mix_done = mix.path.empty();
1922
+ if (!mix.path.empty()) {
1923
+ std::ifstream df(mix.path + ".done");
1924
+ mix_done = (bool)df;
1925
+ }
1926
+ std::this_thread::sleep_for(std::chrono::milliseconds(50));
1927
+ idle++;
1928
+ if (mix_done && idle > 40) {
1929
+ std::puts("[train] live: mix done and no static records; stopping");
1930
+ break;
1931
+ }
1932
+ if (a.live_idle_sec > 0 && idle * 50 > a.live_idle_sec * 1000 && steps == 0) {
1933
+ std::puts("[train] live: idle with zero steps; stopping");
1934
+ break;
1935
+ }
1936
+ }
1937
+ save_snapshot(g, a, steps);
1938
+ std::printf("[train] live done steps=%d mix_recs=%llu\n", steps,
1939
+ (unsigned long long)mix.emitted);
1940
+ return 0;
1941
+ }
1942
+
1943
+ #include "cuda_self_test.h"
1944
+
1945
+ static int steps_from_ckpt_path(const std::string& path) {
1946
+ std::string src = path;
1947
+ char link[768];
1948
+ ssize_t n = readlink(path.c_str(), link, sizeof(link) - 1);
1949
+ if (n > 0) {
1950
+ link[n] = 0;
1951
+ src = link;
1952
+ }
1953
+ auto slash = src.find_last_of('/');
1954
+ std::string base = (slash == std::string::npos) ? src : src.substr(slash + 1);
1955
+ int st = 0;
1956
+ if (std::sscanf(base.c_str(), "gpu_eqprop_%d_", &st) == 1) return st;
1957
+ return 0;
1958
+ }
1959
+
1960
+ static bool check_ckpt(const Args& a) {
1961
+ std::ifstream f(a.ckpt, std::ios::binary | std::ios::ate);
1962
+ const uint64_t need = ckpt_bytes(a.N, a.k);
1963
+ if (!f || (uint64_t)f.tellg() != need) {
1964
+ std::cerr << "[ckpt] missing or wrong size: expected " << need << " bytes\n";
1965
+ return false;
1966
+ }
1967
+ f.seekg(0);
1968
+ uint32_t N = 0, k = 0;
1969
+ f.read((char*)&N, 4);
1970
+ f.read((char*)&k, 4);
1971
+ if (!f || N != a.N || k != a.k) {
1972
+ std::cerr << "[ckpt] N/k mismatch\n";
1973
+ return false;
1974
+ }
1975
+ std::printf("[ckpt] header/size OK: %ux%u %llu bytes\n", N, k,
1976
+ (unsigned long long)need);
1977
+ return true;
1978
+ }
1979
+
1980
+ static int main_impl(int argc, char** argv) {
1981
+ std::setvbuf(stdout, nullptr, _IOLBF, 0);
1982
+ std::setvbuf(stderr, nullptr, _IONBF, 0);
1983
+ std::puts("=== NYS GPU EqProp (CUDA, no PyTorch) ===");
1984
+ Args a = parse_args(argc, argv);
1985
+ if (a.validate_distill_stdin) return validate_distill_stdin(a.N);
1986
+ if (a.check_inputs) {
1987
+ resolve_corpora(a);
1988
+ for (const auto& path : a.corpora) {
1989
+ std::ifstream f(path, std::ios::binary);
1990
+ char magic[4] = {};
1991
+ f.read(magic, 4);
1992
+ if (!f) throw std::runtime_error("missing/empty corpus: " + path);
1993
+ if (a.distill_stdin && !std::memcmp(magic, "SDS1", 4))
1994
+ throw std::runtime_error("Qwen replaces the SDS1 slot; omit the legacy corpus");
1995
+ std::printf("[inputs] %s\n", path.c_str());
1996
+ }
1997
+ std::string sd = a.slots_dir;
1998
+ if (sd.empty() && !a.data_dir.empty()) sd = a.data_dir + "/slots";
1999
+ if (!sd.empty()) {
2000
+ load_silence_codes(sd);
2001
+ load_global_gold_frames(sd);
2002
+ if (a.require_full_curriculum) {
2003
+ for (const char* name : {"speech_slot.nysv","speech_align.nysv"}) {
2004
+ NysvTail tail; tail.path=sd+"/"+name;
2005
+ if (!tail.ensure() || !tail.record_count) throw std::runtime_error("Empty speech source");
2006
+ uint64_t skipped=0;
2007
+ const uint32_t reads=2*tail.record_count;
2008
+ for (uint32_t j=0;j<reads;++j) {
2009
+ NysvRec rec; std::vector<uint32_t> voiced;
2010
+ if (!next_voiced_speech(tail,rec,voiced,a.N,skipped))
2011
+ throw std::runtime_error("Speech source cannot supply a voiced rotation");
2012
+ }
2013
+ std::printf("[inputs] voiced-reader %s passed %u reads; skipped=%llu\n",name,reads,(unsigned long long)skipped);
2014
+ }
2015
+ }
2016
+ }
2017
+ std::puts("[inputs] CPU inspection complete; no CUDA allocation or training");
2018
+ return 0;
2019
+ }
2020
+ if (a.distill_stdin && (!a.live || a.settle || a.init)) {
2021
+ std::cerr << "--distill-stdin requires --live and a resume checkpoint\n";
2022
+ return 2;
2023
+ }
2024
+ g_omega_scale = a.omega_scale;
2025
+ kFillNeighbors = a.fill_neighbors;
2026
+ std::printf("[dyn] omega_scale=%g (1 = original dynamics)\n", g_omega_scale);
2027
+
2028
+ if (!a.probe && !a.self_test) {
2029
+ if (a.init && !a.ckpt.empty()) {
2030
+ std::cerr << "--init cannot be combined with --ckpt\n";
2031
+ return 2;
2032
+ }
2033
+ if (!a.init && a.ckpt.empty()) {
2034
+ std::cerr << "explicit --ckpt is required; use --init only for a new graph\n";
2035
+ return 2;
2036
+ }
2037
+ if ((!a.init || a.check_ckpt) && !check_ckpt(a)) return 1;
2038
+ if (a.check_ckpt) return 0;
2039
+ }
2040
+ probe_device(a.device, a.vram_limit_gb);
2041
+ if (a.probe) return 0;
2042
+
2043
+ size_t free_b = 0, total_b = 0;
2044
+ CUDA_CHECK(cudaMemGetInfo(&free_b, &total_b));
2045
+ size_t cap = (a.vram_limit_gb > 0.0)
2046
+ ? (size_t)(a.vram_limit_gb * (double)(1ull << 30))
2047
+ : (size_t)(kVramFraction * (double)total_b);
2048
+ if (cap > kVramLimitMax) cap = kVramLimitMax;
2049
+
2050
+ uint64_t model = a.self_test ? ckpt_bytes(4096, 32) : ckpt_bytes(a.N, a.k);
2051
+ std::printf("[plan] model_state=%.3f GB cap=%.3f GiB leave_free_for_others=%.3f GiB\n",
2052
+ model / 1e9, cap / (double)(1ull << 30), kLeaveFreeBytes / (double)(1ull << 30));
2053
+ if (!a.self_test && model > cap) {
2054
+ std::cerr << "[plan] model larger than VRAM cap\n";
2055
+ return 3;
2056
+ }
2057
+ if (free_b < (size_t)model + kLeaveFreeBytes) {
2058
+ std::cerr << "[plan] not enough free VRAM without crowding other jobs. abort.\n";
2059
+ return 3;
2060
+ }
2061
+
2062
+ if (a.self_test) return run_self_test(cap);
2063
+
2064
+ if (!a.init && !a.ckpt.empty() && a.resume_steps == 0) {
2065
+ a.resume_steps = steps_from_ckpt_path(a.ckpt);
2066
+ std::printf("[ckpt] resume_steps=%d from %s\n", a.resume_steps, a.ckpt.c_str());
2067
+ }
2068
+
2069
+ if (!a.init && !a.settle && !a.live) {
2070
+ std::cerr << "resuming the GPU lineage requires --live to preserve its step counter\n";
2071
+ return 2;
2072
+ }
2073
+
2074
+ if (a.settle) {
2075
+ if (a.ckpt.empty() || a.init) {
2076
+ std::cerr << "--settle needs --ckpt COMPLETE.bin, not a random init" << std::endl;
2077
+ return 2;
2078
+ }
2079
+ } else {
2080
+ resolve_corpora(a);
2081
+ {
2082
+ std::string sd = a.slots_dir;
2083
+ if (sd.empty() && !a.data_dir.empty()) {
2084
+ sd = a.data_dir;
2085
+ if (sd.back() != '/') sd.push_back('/');
2086
+ sd += "slots";
2087
+ }
2088
+ if (!sd.empty()) {
2089
+ load_silence_codes(sd);
2090
+ load_global_gold_frames(sd);
2091
+ }
2092
+ }
2093
+ if (a.live) {
2094
+ if (a.corpora.empty() && a.live_mix.empty() && !a.distill_stdin) {
2095
+ std::cerr << "need --data-dir corpora or --live-mix\n";
2096
+ return 2;
2097
+ }
2098
+ } else if (a.corpora.empty()) {
2099
+ std::cerr << "need --corpus, --data-dir, --self-test, or --probe\n";
2100
+ return 2;
2101
+ }
2102
+
2103
+ }
2104
+
2105
+ GpuState g = alloc_state(a.N, a.k, cap);
2106
+ if (a.init) {
2107
+ std::puts("[init] random ring+stubs on host (not loading a ckpt)");
2108
+ std::vector<double> theta, amp, omega;
2109
+ std::vector<uint64_t> off;
2110
+ std::vector<uint32_t> col;
2111
+ std::vector<float> K;
2112
+ host_init_graph(a.N, a.k, theta, amp, omega, off, col, K);
2113
+ upload(g, theta, amp, omega, off, col, K);
2114
+ } else {
2115
+ load_ckpt(g, a.ckpt);
2116
+ }
2117
+
2118
+ if (a.settle) return run_settle(g, a);
2119
+
2120
+ if (a.live) return train_live(g, a);
2121
+
2122
+ int rc = 0;
2123
+ for (const auto& path : a.corpora) {
2124
+ std::printf("[train] ===== %s =====\n", path.c_str());
2125
+ std::ifstream probe(path, std::ios::binary);
2126
+ char magic[4] = {};
2127
+ probe.read(magic, 4);
2128
+ bool sds1 = probe.gcount() == 4 && std::memcmp(magic, "SDS1", 4) == 0;
2129
+ bool nysa = probe.gcount() == 4 && std::memcmp(magic, "NYSA", 4) == 0;
2130
+ bool nysv = probe.gcount() == 4 && std::memcmp(magic, "NYSV", 4) == 0;
2131
+ bool nysd = probe.gcount() == 4 && std::memcmp(magic, "NYSD", 4) == 0;
2132
+ probe.close();
2133
+ if (nysd) {
2134
+ std::printf("[train] skip NYSD %s\n", path.c_str());
2135
+ } else if (nysa) {
2136
+ rc |= train_nysa_file(g, a, path);
2137
+ } else if (nysv) {
2138
+ rc |= train_nysv_file(g, a, path);
2139
+ } else if (sds1) {
2140
+ std::ifstream sds(path, std::ios::binary);
2141
+ char mag[4] = {};
2142
+ sds.read(mag, 4);
2143
+ rc |= train_sds1(g, a, sds, path);
2144
+ } else {
2145
+ rc |= train_hard(g, a, path);
2146
+ }
2147
+ }
2148
+ return rc;
2149
+ }
2150
+
2151
+ int main(int argc, char** argv) {
2152
+ try { return main_impl(argc, argv); }
2153
+ catch (const std::exception& e) {
2154
+ std::cerr << "[error] " << e.what() << std::endl;
2155
+ return 1;
2156
+ }
2157
+ }
lineages/corrected-20260930-v2/recovery/gpu_eqprop/hf_checkpoint_service.py ADDED
@@ -0,0 +1,458 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """L40S checkpoint uploads and separate, SHA-verified LOCAL retention.
3
+
4
+ No GPU imports, remote deletion, repo creation, or directory-wide uploads.
5
+ Zoey's latest checkpoint commit also includes a consistent warehouse snapshot
6
+ and an explicit allowlist of recovery files. Credentials are never artifacts.
7
+ """
8
+ import argparse
9
+ from contextlib import contextmanager
10
+ from datetime import datetime, timezone
11
+ import hashlib
12
+ import json
13
+ import os
14
+ from pathlib import Path
15
+ import re
16
+ import struct
17
+ import tempfile
18
+ import time
19
+
20
+ SIZE = 20_640_000_016
21
+ REPOS = {'nys': 'YNSScarSaiyan/nys-sft-public', 'zoey': 'YNSScarSaiyan/zoey'}
22
+ PATTERNS = {
23
+ 'nys': re.compile(r'gpu_eqprop_(\d+)_\d{8}_\d{6}\.bin\Z'),
24
+ 'zoey': re.compile(r'zoey_vsa_distill_step_(\d+)\.json\Z'),
25
+ }
26
+
27
+
28
+ def now():
29
+ return datetime.now(timezone.utc).isoformat()
30
+
31
+
32
+ def log(event, **fields):
33
+ print(json.dumps(dict(time=now(), event=event, **fields)), flush=True)
34
+
35
+
36
+ def atomic_json(path, data):
37
+ tmp = path.with_suffix('.writing')
38
+ with tmp.open('w') as f:
39
+ json.dump(data, f, indent=2)
40
+ f.write('\n'); f.flush(); os.fsync(f.fileno())
41
+ tmp.replace(path)
42
+
43
+
44
+ def verified_for_lineage(root, verified):
45
+ prefix = remote_prefix(Path(root))
46
+ return {key: record for key, record in verified.items()
47
+ if not prefix or record.get('remote_path', '').startswith(prefix)}
48
+
49
+
50
+ def service_status(root):
51
+ """Read saved service identities and verify each PID's command line."""
52
+ service = Path(root)/'hf-services'
53
+ result = {}
54
+ for mode in ('upload', 'retain'):
55
+ path = service/f'{mode}.json'
56
+ entry = json.loads(path.read_text()) if path.exists() else {}
57
+ cmdline = Path('/proc')/str(entry.get('pid', 0))/'cmdline'
58
+ try:
59
+ command = cmdline.read_bytes().split(b'\0')
60
+ except FileNotFoundError:
61
+ command = []
62
+ entry['alive'] = (any(x.endswith(b'/hf_checkpoint_service.py') for x in command)
63
+ and mode.encode() in command and entry.get('kind', '').encode() in command)
64
+ result[mode] = entry
65
+ path = service/'verified.json'
66
+ verified = json.loads(path.read_text()).get('verified', {}) if path.exists() else {}
67
+ verified = verified_for_lineage(root, verified)
68
+ result['verified_checkpoint_count'] = len(verified)
69
+ result['latest_verified'] = max(verified.values(), key=lambda x: x['step']) if verified else None
70
+ return result
71
+
72
+
73
+ def signature(path):
74
+ st = path.stat()
75
+ return [st.st_size, st.st_mtime_ns, st.st_ino]
76
+
77
+
78
+ def digest(path):
79
+ h = hashlib.sha256()
80
+ with path.open('rb') as f:
81
+ while chunk := f.read(16 * 1024 * 1024):
82
+ h.update(chunk)
83
+ return h.hexdigest()
84
+
85
+
86
+ def candidates(root, kind, settle=20):
87
+ base = root/'checkpoints' if kind == 'nys' else root/'runtime/hf-pending'
88
+ paths = base.glob('qwen-*/*.bin') if kind == 'nys' else base.glob('*.json')
89
+ result = []
90
+ for path in paths:
91
+ if (path.parent/'DO_NOT_RESUME.json').exists():
92
+ continue
93
+ match = PATTERNS[kind].fullmatch(path.name)
94
+ if not match or path.is_symlink() or not path.resolve().is_relative_to(base.resolve()):
95
+ continue
96
+ try:
97
+ st = path.stat()
98
+ if st.st_size <= 0 or time.time()-st.st_mtime < settle:
99
+ continue
100
+ if kind == 'nys' and st.st_size != SIZE:
101
+ continue
102
+ result.append((int(match[1]), path))
103
+ except FileNotFoundError:
104
+ continue
105
+ return sorted(result, key=lambda item: (item[0], item[1].name), reverse=True)
106
+
107
+
108
+ def inspect_checkpoint(path, kind, step):
109
+ before = signature(path)
110
+ meta = {}
111
+ if kind == 'nys':
112
+ with path.open('rb') as f:
113
+ if before[0] != SIZE or struct.unpack('<II', f.read(8)) != (5_000_000, 512):
114
+ raise ValueError('invalid NYS checkpoint size/header')
115
+ else:
116
+ with path.open() as f:
117
+ ckpt = json.load(f)
118
+ if (ckpt.get('step') != step or ckpt.get('map_ver') != 5
119
+ or not isinstance(ckpt.get('map5'), dict) or not ckpt['map5']
120
+ or type(ckpt.get('warehouse_consumed')) is not int
121
+ or ckpt['warehouse_consumed'] < 0):
122
+ raise ValueError('invalid Zoey checkpoint schema/cursor')
123
+ meta['warehouse_consumed'] = ckpt['warehouse_consumed']
124
+ sha = digest(path)
125
+ if signature(path) != before:
126
+ raise ValueError('checkpoint changed during inspection')
127
+ return dict(signature=before, bytes=before[0], sha256=sha, step=step, **meta)
128
+
129
+
130
+ def field(obj, name):
131
+ return obj.get(name) if isinstance(obj, dict) else getattr(obj, name, None)
132
+
133
+
134
+ def remote_matches(info, record):
135
+ lfs = field(info, 'lfs')
136
+ return (field(info, 'size') == record['bytes']
137
+ and field(lfs, 'size') == record['bytes']
138
+ and field(lfs, 'sha256') == record['sha256'])
139
+
140
+
141
+ def remote_info(api, repo, name, revision=None):
142
+ infos = api.get_paths_info(repo, [name], repo_type='model', revision=revision)
143
+ return infos[0] if infos else None
144
+
145
+
146
+ def protected_paths(root, kind, rows, keep):
147
+ protected = {p.resolve() for _, p in rows[:keep]}
148
+ if kind == 'nys' and rows:
149
+ active = json.loads((root/'runs/active_run.json').read_text())
150
+ if not re.fullmatch(r'qwen-[a-z0-9-]+', active['run_id']):
151
+ raise ValueError('invalid active NYS run')
152
+ started = json.loads((root/'runs'/active['run_id']/'started.json').read_text())
153
+ manifest_path = Path(started['checkpoint_manifest']).resolve()
154
+ if not manifest_path.is_relative_to(root.resolve()):
155
+ raise ValueError('resume manifest outside workspace')
156
+ manifest = json.loads(manifest_path.read_text())
157
+ protected.add(Path(manifest['local_path']).resolve())
158
+ return protected
159
+
160
+
161
+ def snapshot_jsonl(source, dest):
162
+ import fcntl
163
+ rows = 0
164
+ with source.open('rb') as src, dest.open('wb') as out:
165
+ fcntl.flock(src, fcntl.LOCK_SH)
166
+ try:
167
+ # Bounded prefix, including for the raw stream whose writer does
168
+ # not lock. An incomplete trailing line is excluded.
169
+ remaining = os.fstat(src.fileno()).st_size
170
+ while remaining > 0:
171
+ line = src.readline(remaining)
172
+ remaining -= len(line)
173
+ if not line or not line.endswith(b'\n'):
174
+ break
175
+ out.write(line)
176
+ rows += bool(line.strip())
177
+ finally:
178
+ fcntl.flock(src, fcntl.LOCK_UN)
179
+ return rows
180
+
181
+
182
+ def snapshot_zoey(root, stage, record, name):
183
+ runtime = root/'runtime'
184
+ warehouse = stage/'zoey_warehouse.jsonl'
185
+ count = snapshot_jsonl(runtime/'corpus/zoey_warehouse.jsonl', warehouse)
186
+ if count < record['warehouse_consumed']:
187
+ raise ValueError('warehouse snapshot is shorter than checkpoint cursor')
188
+ stream = stage/'zoey_warehouse_stream.jsonl'
189
+ snapshot_jsonl(runtime/'corpus/zoey_warehouse_stream.jsonl', stream)
190
+ files = {'corpus/zoey_warehouse.jsonl': warehouse,
191
+ 'corpus/zoey_warehouse_stream.jsonl': stream}
192
+ # Explicit paths only; never include tokens, caches or arbitrary runtime files.
193
+ for relative, src in {
194
+ 'map5_see.json': runtime/'map5_see.json', 'flags.json': runtime/'flags.json',
195
+ 'migration-manifest.json': runtime/'migration-manifest.json',
196
+ 'teacher_manifest.json': root/'teacher_manifest.json',
197
+ 'corpus/hose_seeds.txt': runtime/'corpus/hose_seeds.txt',
198
+ 'merge-manifest.json': root/'merged-corpus/merge-manifest.json',
199
+ }.items():
200
+ before = signature(src)
201
+ target = stage/src.name
202
+ data = src.read_bytes()
203
+ if signature(src) != before:
204
+ raise ValueError(f'recovery file changed during snapshot: {relative}')
205
+ if src.suffix == '.json':
206
+ json.loads(data)
207
+ target.write_bytes(data)
208
+ files[relative] = target
209
+ checkpoint = json.loads((runtime/'hf-pending'/name).read_text())
210
+ if (root/'lineage.json').exists():
211
+ readout=checkpoint.get('readout_state')
212
+ if not readout or not readout.get('flags.json',{}).get('gemini_chooser'):
213
+ raise ValueError('Corrected checkpoint has no saved Gemini reader')
214
+ for relative,data in readout.items():
215
+ if relative not in ('flags.json','map5_see.json','chooser.json'):
216
+ raise ValueError('Unknown reader recovery file')
217
+ target=stage/relative
218
+ target.write_text(json.dumps(data))
219
+ files[relative]=target
220
+ for relative,src in {
221
+ 'lineage.json':root/'lineage.json',
222
+ 'chooser-acceptance.json':root/'validation/chooser-acceptance.json',
223
+ 'source/zoey_neural_chooser.py':root/'source/zoey_neural_chooser.py',
224
+ 'source/zoey_readout.py':root/'source/zoey_readout.py',
225
+ 'source/zoey_see.py':root/'source/zoey_see.py',
226
+ 'source/train_zoey_distill.py':root/'source/train_zoey_distill.py',
227
+ }.items():
228
+ target=stage/src.name
229
+ target.write_bytes(src.read_bytes())
230
+ files[relative]=target
231
+ source=root/'source'
232
+ code_paths=list(source.glob('zoey_*.py'))+list((source/'src').rglob('*.rs'))
233
+ code_paths += [p for p in (source/'Cargo.toml',source/'Cargo.lock') if p.is_file()]
234
+ code_paths += list((source/'l40s').glob('*.py'))+list((source/'l40s').glob('*.sh'))
235
+ for src in code_paths:
236
+ relative='source/'+src.relative_to(source).as_posix()
237
+ target=stage/'code'/src.relative_to(source)
238
+ target.parent.mkdir(parents=True,exist_ok=True)
239
+ target.write_bytes(src.read_bytes())
240
+ files[relative]=target
241
+ manifest = dict(checkpoint=f'{remote_prefix(root)}checkpoints/{name}', checkpoint_sha256=record['sha256'],
242
+ warehouse_consumed=record['warehouse_consumed'], warehouse_rows=count,
243
+ captured_utc=now(), files={k: dict(bytes=p.stat().st_size, sha256=digest(p))
244
+ for k, p in files.items()})
245
+ target = stage/'backup-manifest.json'
246
+ target.write_text(json.dumps(manifest, indent=2)+'\n')
247
+ files['backup-manifest.json'] = target
248
+ return files
249
+
250
+
251
+ def needs_runtime_bundle(kind, step, state, previous, has_lineage=False):
252
+ # Legacy Zoey checkpoints can have larger step counters from a different
253
+ # schema/lineage. Compare only this deployment's verified checkpoints.
254
+ highest = max((r['step'] for r in state['verified'].values()), default=-1)
255
+ return ((kind == 'zoey' or has_lineage) and step >= highest
256
+ and not (previous and previous.get('runtime_bundle_verified')
257
+ and (not has_lineage or previous.get('bundle_version') == 2)))
258
+
259
+
260
+ def remote_prefix(root):
261
+ path=root/'lineage.json'
262
+ if not path.exists(): return ''
263
+ lineage=json.loads(path.read_text())['lineage']
264
+ if not re.fullmatch(r'[a-z0-9-]+',lineage): raise ValueError('Invalid backup lineage')
265
+ return 'lineages/'+lineage+'/'
266
+
267
+
268
+ def snapshot_nys(root, stage, record, name):
269
+ files={}
270
+ for subdir,names in {
271
+ 'data/corrected-mix':['mix_sft_sparse.bin','mix_manifest.json'],
272
+ 'data/language_baseline':['vocab_sparse.json','train_corpus_sparse.bin','sft_corpus_sparse.bin','recovery_manifest.json'],
273
+ 'data/slots':['compiler_physics.nysa','speech_align.nysv','speech_slot.nysv','speech_vocoder.nyvc','recovery_manifest.json'],
274
+ }.items():
275
+ for item in names:
276
+ path=root/subdir/item
277
+ if not path.is_file(): raise ValueError(f'Missing recovery input: {path}')
278
+ files[subdir+'/'+item]=path
279
+ code=['build_cuda.sh','run_cuda_l40s.sh','run_qwen_live.py','start_qwen_l40s.sh',
280
+ 'qwen_live.py','validate_curriculum.py','eqprop_gpu.cu','distill_stdin.h','cuda_self_test.h',
281
+ 'hf_checkpoint_service.py','start_backup_services.py','live_status.py','build_corrected_mix.py']
282
+ for relative in ['gpu_eqprop/'+n for n in code]+['hierarchical_tokenizer.py','source_checkpoint.json','lineage.json','recovery_policy.json']:
283
+ src=root/relative
284
+ target=stage/'code'/relative
285
+ target.parent.mkdir(parents=True,exist_ok=True)
286
+ target.write_bytes(src.read_bytes())
287
+ files[relative]=target
288
+ manifest=dict(checkpoint=remote_prefix(root)+name,checkpoint_sha256=record['sha256'],captured_utc=now(),
289
+ files={n:dict(bytes=p.stat().st_size,sha256=digest(p)) for n,p in files.items()})
290
+ target=stage/'backup-manifest.json'
291
+ target.write_text(json.dumps(manifest,indent=2))
292
+ files['backup-manifest.json']=target
293
+ return files
294
+
295
+
296
+ def upload_one(api, root, kind, path, record, service, publish_runtime=False):
297
+ from huggingface_hub import CommitOperationAdd
298
+ repo = REPOS[kind]
299
+ prefix=remote_prefix(root)
300
+ dest = prefix+(path.name if kind == 'nys' else 'checkpoints/'+path.name)
301
+ existing = remote_info(api, repo, dest)
302
+ if existing is not None:
303
+ if not remote_matches(existing, record):
304
+ raise ValueError(f'remote filename collision: {dest}; refusing to overwrite')
305
+ if not publish_runtime:
306
+ return dict(record, remote_path=dest, verified_utc=now())
307
+ with tempfile.TemporaryDirectory(prefix='snapshot-', dir=service) as folder:
308
+ stage = Path(folder)
309
+ operations = [CommitOperationAdd(path_in_repo=dest, path_or_fileobj=str(path))]
310
+ head = api.model_info(repo, files_metadata=False)
311
+ files = {}
312
+ bundle_prefix=prefix+('runtime/l40s/' if kind=='zoey' else 'recovery/')
313
+ if publish_runtime and kind=='zoey':
314
+ files = snapshot_zoey(root, stage, record, path.name)
315
+ operations.append(CommitOperationAdd(path_in_repo=prefix+'checkpoints/LATEST.txt',
316
+ path_or_fileobj=(path.name+'\n').encode()))
317
+ elif publish_runtime and kind=='nys':
318
+ files=snapshot_nys(root,stage,record,path.name)
319
+ operations.extend(CommitOperationAdd(path_in_repo=bundle_prefix+name,
320
+ path_or_fileobj=str(src)) for name,src in files.items())
321
+ if prefix and publish_runtime:
322
+ lineage=(root/'lineage.json').read_bytes()
323
+ operations.append(CommitOperationAdd(path_in_repo=prefix+'lineage.json',path_or_fileobj=lineage))
324
+ policy=root/'recovery_policy.json'
325
+ if policy.exists():
326
+ operations.append(CommitOperationAdd(path_in_repo=prefix+'recovery_policy.json',path_or_fileobj=str(policy)))
327
+ operations.append(CommitOperationAdd(path_in_repo=prefix+'LATEST.json',path_or_fileobj=json.dumps(
328
+ dict(path=dest,step=record['step'],bytes=record['bytes'],sha256=record['sha256'])).encode()))
329
+ if kind=='nys':
330
+ for name,src in {
331
+ 'mix_manifest.json':root/'data/corrected-mix/mix_manifest.json',
332
+ 'curriculum.json':path.parent/'curriculum.json',
333
+ }.items():
334
+ operations.append(CommitOperationAdd(path_in_repo=prefix+name,path_or_fileobj=str(src)))
335
+ if signature(path) != record['signature']:
336
+ raise ValueError('checkpoint changed before upload')
337
+ log('upload_start', checkpoint=path.name, bytes=record['bytes'], repo=repo)
338
+ commit = api.create_commit(repo_id=repo, repo_type='model', operations=operations,
339
+ parent_commit=head.sha,
340
+ commit_message=f'L40S {kind} checkpoint {record["step"]}')
341
+ info = remote_info(api, repo, dest, revision=commit.oid)
342
+ if not remote_matches(info, record):
343
+ raise ValueError('HF size/SHA verification failed after commit')
344
+ if signature(path) != record['signature']:
345
+ raise ValueError('local checkpoint changed during upload')
346
+ if files:
347
+ infos = api.get_paths_info(repo, [bundle_prefix+name for name in files],
348
+ revision=commit.oid, repo_type='model')
349
+ if len(infos) != len(files):
350
+ raise ValueError('HF recovery bundle is incomplete')
351
+ for info in infos:
352
+ source = files[info.path.removeprefix(bundle_prefix)]
353
+ if info.size != source.stat().st_size:
354
+ raise ValueError(f'HF recovery size mismatch: {info.path}')
355
+ if info.lfs:
356
+ if field(info.lfs, 'sha256') != digest(source):
357
+ raise ValueError(f'HF recovery SHA mismatch: {info.path}')
358
+ else:
359
+ blob = source.read_bytes()
360
+ expected = hashlib.sha1(b'blob '+str(len(blob)).encode()+b'\0'+blob).hexdigest()
361
+ if info.blob_id != expected:
362
+ raise ValueError(f'HF recovery Git blob mismatch: {info.path}')
363
+ return dict(record, remote_path=dest, commit=commit.oid, verified_utc=now(),
364
+ runtime_bundle_verified=bool(files),bundle_version=2 if files else None)
365
+
366
+
367
+ def retention_pass(api, root, kind, state, keep, dry_run=False):
368
+ rows = candidates(root, kind)
369
+ protected = protected_paths(root, kind, rows, keep)
370
+ for step, path in rows:
371
+ if path.resolve() in protected:
372
+ continue
373
+ key = path.relative_to(root).as_posix()
374
+ record = state.get('verified', {}).get(key)
375
+ if not record or signature(path) != record['signature']:
376
+ continue
377
+ info = remote_info(api, REPOS[kind], record['remote_path'])
378
+ if not remote_matches(info, record):
379
+ log('retain_unverified', checkpoint=path.name)
380
+ continue
381
+ # The uploader verified this exact inode/mtime/size against SHA. Verify
382
+ # identity again after the remote check; remove only this single file.
383
+ if signature(path) != record['signature'] or path.is_symlink():
384
+ continue
385
+ log('retain_would_delete' if dry_run else 'retain_delete', checkpoint=path.name,
386
+ bytes=record['bytes'], remote_sha256=record['sha256'])
387
+ if not dry_run:
388
+ path.unlink()
389
+
390
+
391
+ def main():
392
+ ap = argparse.ArgumentParser(description=__doc__)
393
+ ap.add_argument('--kind', choices=REPOS, required=True)
394
+ ap.add_argument('--mode', choices=['upload', 'retain'], required=True)
395
+ ap.add_argument('--root', type=Path, required=True)
396
+ ap.add_argument('--token-file', type=Path, required=True)
397
+ ap.add_argument('--keep', type=int, default=2)
398
+ ap.add_argument('--interval', type=int, default=30)
399
+ ap.add_argument('--once', action='store_true')
400
+ ap.add_argument('--dry-run', action='store_true')
401
+ args = ap.parse_args()
402
+ if args.keep < 2 or args.interval < 5:
403
+ raise ValueError('keep >= 2 and interval >= 5 required')
404
+ root = args.root.resolve()
405
+ if (root/'DO_NOT_RESUME_CURRENT_CHECKPOINTS.json').exists():
406
+ raise ValueError('User quarantined this deployment; upload and retention are disabled')
407
+ if (root/'lineage.json').exists() and json.loads((root/'lineage.json').read_text()).get('status') != 'VALIDATED':
408
+ raise ValueError('Corrected lineage has not passed validation')
409
+ service = root/'hf-services'
410
+ service.mkdir(exist_ok=True)
411
+ import fcntl
412
+ from huggingface_hub import HfApi
413
+ with (service/f'{args.mode}.lock').open('a') as lock:
414
+ fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB)
415
+ api = HfApi(token=args.token_file.read_text().strip())
416
+ api.model_info(REPOS[args.kind]) # Require the known repo to exist.
417
+ state_path = service/'verified.json'
418
+ atomic_json(service/f'{args.mode}.json', dict(pid=os.getpid(), started_utc=now(),
419
+ kind=args.kind, mode=args.mode, repo=REPOS[args.kind], keep=args.keep))
420
+ log('service_started', mode=args.mode, kind=args.kind, keep=args.keep,
421
+ repo=REPOS[args.kind], dry_run=args.dry_run)
422
+ while True:
423
+ try:
424
+ state = json.loads(state_path.read_text()) if state_path.exists() else {'verified': {}}
425
+ if args.mode == 'retain':
426
+ retention_pass(api, root, args.kind, state, args.keep, args.dry_run)
427
+ else:
428
+ for step, path in candidates(root, args.kind):
429
+ key = path.relative_to(root).as_posix()
430
+ current_verified = verified_for_lineage(root, state['verified'])
431
+ previous = current_verified.get(key)
432
+ publish_runtime = needs_runtime_bundle(args.kind, step, {'verified': current_verified}, previous,
433
+ has_lineage=bool(remote_prefix(root)))
434
+ if previous and signature(path) == previous['signature'] and not publish_runtime:
435
+ continue
436
+ if args.dry_run:
437
+ log('upload_candidate', checkpoint=path.name, step=step, bytes=path.stat().st_size)
438
+ continue
439
+ record = inspect_checkpoint(path, args.kind, step)
440
+ state['verified'][key] = upload_one(api, root, args.kind, path, record, service,
441
+ publish_runtime=publish_runtime)
442
+ atomic_json(state_path, state)
443
+ log('upload_verified', checkpoint=path.name, bytes=record['bytes'], sha256=record['sha256'])
444
+ # Re-scan after each upload so newly saved checkpoints
445
+ # take priority over the historical backlog.
446
+ break
447
+ except Exception as exc:
448
+ message = re.sub(r'hf_[A-Za-z0-9]+', '[REDACTED]', str(exc))
449
+ log('error', mode=args.mode, error=type(exc).__name__, message=message[:600])
450
+ if args.once:
451
+ raise SystemExit(1)
452
+ if args.once:
453
+ return
454
+ time.sleep(args.interval)
455
+
456
+
457
+ if __name__ == '__main__':
458
+ main()
lineages/corrected-20260930-v2/recovery/gpu_eqprop/live_status.py ADDED
@@ -0,0 +1,93 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """Read the authorized L40S run's logs and checkpoints; never controls processes."""
3
+ import argparse
4
+ from collections import deque
5
+ from datetime import datetime, timezone
6
+ import hashlib
7
+ import json
8
+ from pathlib import Path
9
+ import re
10
+ import struct
11
+ import subprocess
12
+ from hf_checkpoint_service import service_status
13
+
14
+
15
+ def main():
16
+ ap = argparse.ArgumentParser(description=__doc__)
17
+ ap.add_argument("--verify-latest", action="store_true")
18
+ args = ap.parse_args()
19
+ root = Path(__file__).resolve().parents[1]
20
+ run_id = json.loads((root/"runs/active_run.json").read_text())["run_id"]
21
+ if not re.fullmatch(r"qwen-[a-z0-9-]+", run_id):
22
+ raise ValueError("invalid active run ID")
23
+ run = root/"runs"/run_id
24
+ started = json.loads((run/"started.json").read_text())
25
+ result = dict(checked_utc=datetime.now(timezone.utc).isoformat(), run_id=run_id, started=started)
26
+ if (run/"finished.json").exists():
27
+ result["finished"] = json.loads((run/"finished.json").read_text())
28
+ teacher, recent, skips, issues = None, deque(maxlen=500), 0, []
29
+ step = started["resume_step"]
30
+ save_step = step
31
+ contrast = None
32
+ if (run/"train.log").exists():
33
+ with (run/"train.log").open(errors="replace") as log:
34
+ for line in log:
35
+ m = re.search(r"\[step (\d+)\] (.+)", line)
36
+ if m:
37
+ step, contrast = int(m[1]), m[2].strip()
38
+ saving = re.search(r'\[ckpt\] saving .*gpu_eqprop_(\d+)_\d{8}_\d{6}\.bin',line)
39
+ if saving:
40
+ save_step=max(save_step,int(saving[1]))
41
+ if line.startswith('{"event":'):
42
+ try:
43
+ event = json.loads(line)
44
+ except json.JSONDecodeError:
45
+ continue # Writer can be between writes during a status read.
46
+ if event.get("event") == "record":
47
+ teacher = event["record"]
48
+ recent.append(event["retained_mass"])
49
+ elif event.get("event") == "skipped":
50
+ skips += 1
51
+ if any(s in line for s in ("[error]", "Traceback", "[qwen-live]", "CUDA error", "Fatal Python", "nan", "inf ")):
52
+ issues.append(line.strip()[:400])
53
+ result.update(last_logged_step=step, updates=step-started["resume_step"],
54
+ last_completed_update_step=max(step,save_step),
55
+ contrast=contrast, teacher_records_emitted=teacher, teacher_skipped=skips,
56
+ recent_teacher_records=len(recent),
57
+ recent_retained_mass_min=min(recent) if recent else None,
58
+ recent_retained_mass_mean=sum(recent)/len(recent) if recent else None,
59
+ issues=issues[-10:])
60
+ result['checkpoint_writes']=[dict(path=str(p),bytes=p.stat().st_size,
61
+ modified_unix=p.stat().st_mtime) for p in (root/'checkpoints'/run_id).glob('*.bin.writing')]
62
+ files = []
63
+ for path in (root/"checkpoints"/run_id).glob("gpu_eqprop_*.bin"):
64
+ match = re.fullmatch(r"gpu_eqprop_(\d+)_\d{8}_\d{6}\.bin", path.name)
65
+ if match and path.stat().st_size == 20_640_000_016:
66
+ files.append((int(match[1]), path))
67
+ result["complete_checkpoint_count"] = len(files)
68
+ if files:
69
+ step, path = max(files, key=lambda item: (item[0], item[1].name))
70
+ with path.open("rb") as f:
71
+ dims = struct.unpack("<II", f.read(8))
72
+ digest = None
73
+ if args.verify_latest:
74
+ h = hashlib.sha256()
75
+ f.seek(0)
76
+ while chunk := f.read(16*1024*1024):
77
+ h.update(chunk)
78
+ digest = h.hexdigest()
79
+ result["latest_checkpoint"] = dict(path=str(path), step=step, bytes=path.stat().st_size,
80
+ dimensions=dims, header_ok=dims==(5_000_000,512),
81
+ sha256=digest)
82
+ result["gpu1"] = subprocess.check_output([
83
+ "nvidia-smi", "-i", "1", "--query-gpu=uuid,memory.used,memory.free,utilization.gpu",
84
+ "--format=csv,noheader"], text=True).strip()
85
+ result["gpu_processes"] = subprocess.check_output([
86
+ "nvidia-smi", "--query-compute-apps=gpu_uuid,pid,process_name,used_memory",
87
+ "--format=csv,noheader"], text=True).strip().splitlines()
88
+ result["hf_backup"] = service_status(root)
89
+ print(json.dumps(result, indent=2))
90
+
91
+
92
+ if __name__ == "__main__":
93
+ main()
lineages/corrected-20260930-v2/recovery/gpu_eqprop/qwen_live.py ADDED
@@ -0,0 +1,283 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """Qwen next-token probabilities -> frozen NYS IDs -> live SDS1 on stdout.
3
+
4
+ No response generation and no corpus output. Multi-ID mappings are rejected:
5
+ one teacher-token probability is not the probability of its first NYS piece.
6
+ Duplicate single-ID mappings are summed; retained mass is never inflated.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import argparse
11
+ from collections import defaultdict, deque
12
+ import hashlib
13
+ import json
14
+ import math
15
+ import os
16
+ from pathlib import Path
17
+ import re
18
+ import struct
19
+ import sys
20
+
21
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
22
+ from hierarchical_tokenizer import HierarchicalTokenizer
23
+
24
+ GPU1 = "GPU-15278276-8f59-9540-a03f-7188fec578f9"
25
+ HEADER = b"SDS1" + struct.pack("<I", 0xffffffff)
26
+ FOOTER = b"\x00\x00DONE"
27
+ N = 5_000_000
28
+
29
+
30
+ def file_sha(path):
31
+ h = hashlib.sha256()
32
+ with open(path, "rb") as f:
33
+ while chunk := f.read(1024*1024):
34
+ h.update(chunk)
35
+ return h.hexdigest()
36
+
37
+
38
+ def load_vocab(path, expected):
39
+ if file_sha(path) != expected.lower():
40
+ raise ValueError("frozen NYS vocabulary SHA-256 mismatch")
41
+ tok = HierarchicalTokenizer()
42
+ tok.load(str(path))
43
+ if not tok.vocab or any(type(v) is not int or v < 0 or v >= N for v in tok.vocab.values()):
44
+ raise ValueError("invalid NYS vocabulary IDs")
45
+ return tok
46
+
47
+
48
+ def project_probabilities(tok, candidates):
49
+ """Conservative partial projection, not a full cross-tokenizer distribution."""
50
+ mass = defaultdict(float)
51
+ rejected = defaultdict(float)
52
+ total = 0.0
53
+ for text, probability in candidates:
54
+ if not math.isfinite(probability) or not 0 <= probability <= 1:
55
+ raise ValueError("invalid teacher probability")
56
+ total += probability
57
+ if probability == 0:
58
+ continue
59
+ if not text or "\ufffd" in text or "<|" in text:
60
+ rejected["special_or_undecodable"] += probability
61
+ continue
62
+ ids = tok.encode(text)
63
+ if len(ids) != 1:
64
+ rejected["multiple_or_empty_nys_ids"] += probability
65
+ continue
66
+ if not 0 <= ids[0] < N:
67
+ raise ValueError("teacher mapped outside NYS graph")
68
+ mass[ids[0]] += probability
69
+ if total > 1.00001:
70
+ raise ValueError("teacher top-k probability mass exceeds one")
71
+ # Float32 softmax plus duplicate-ID summation can exceed 1 by rounding
72
+ # (observed: 1.0000000000005385). Clamp only downward after checking total
73
+ # mass; this neither renormalizes nor inflates any target probability.
74
+ targets = [(token_id, min(1.0, probability)) for token_id, probability in mass.items()]
75
+ return sorted(targets, key=lambda x: (-x[1], x[0])), dict(rejected)
76
+
77
+
78
+ def pack_record(prompt, targets):
79
+ if not 1 <= len(prompt) <= 127 or not 1 <= len(targets) <= 64:
80
+ raise ValueError("record exceeds active-set bounds")
81
+ if any(type(x) is not int or not 0 <= x < N for x in prompt):
82
+ raise ValueError("invalid prompt ID")
83
+ if len({x for x, _ in targets}) != len(targets):
84
+ raise ValueError("duplicate target ID")
85
+ for x, p in targets:
86
+ if type(x) is not int or not 0 <= x < N or not math.isfinite(p) or not 0 < p <= 1:
87
+ raise ValueError(f"invalid target ID/probability: id={x}, probability={p!r}")
88
+ if sum(p for _, p in targets) > 1.00001:
89
+ raise ValueError("target probability mass exceeds one")
90
+ return (struct.pack("<H", len(prompt)) + struct.pack(f"<{len(prompt)}I", *prompt)
91
+ + struct.pack("<H", len(targets))
92
+ + struct.pack(f"<{len(targets)}I", *(x for x, _ in targets))
93
+ + struct.pack(f"<{len(targets)}f", *(p for _, p in targets)))
94
+
95
+
96
+ def prompt_text(row):
97
+ if isinstance(row.get("prompt"), str):
98
+ return row["prompt"]
99
+ for message in row.get("messages", []):
100
+ if message.get("role") == "user" and isinstance(message.get("content"), str):
101
+ return message["content"]
102
+ raise ValueError("input row has no user prompt")
103
+
104
+
105
+ def iter_prompts(args):
106
+ if args.prompts:
107
+ with args.prompts.open(encoding="utf-8") as f:
108
+ for line in f:
109
+ if line.strip():
110
+ yield prompt_text(json.loads(line))
111
+ else:
112
+ from datasets import load_dataset
113
+ if not args.dataset_revision or not re.fullmatch(r"[0-9a-f]{40}", args.dataset_revision):
114
+ raise ValueError("dataset source needs a pinned 40-character revision")
115
+ rows = iter(load_dataset(args.dataset, split=args.split, revision=args.dataset_revision,
116
+ streaming=True, token=False))
117
+ try:
118
+ for row in rows:
119
+ yield prompt_text(row)
120
+ finally:
121
+ close = getattr(rows, "close", None)
122
+ if close:
123
+ close()
124
+
125
+
126
+ def unique_prompts(tok, prompts, skip=0):
127
+ """Rebuild the same deduplication prefix without running teacher inference.
128
+
129
+ prompt_index counts distinct, nonempty represented prompts; source_row
130
+ counts input rows. A pinned source and vocabulary are required for resume.
131
+ """
132
+ seen = set()
133
+ prompt_index = 0
134
+ for source_row, raw in enumerate(prompts, 1):
135
+ core = tok.encode(raw)[:96]
136
+ if not core:
137
+ continue
138
+ represented = tok.decode(core)
139
+ if represented in seen:
140
+ continue
141
+ seen.add(represented)
142
+ prompt_index += 1
143
+ if prompt_index > skip:
144
+ yield prompt_index, source_row, core, represented
145
+
146
+
147
+ class MappingGuard:
148
+ """Stop a degraded bridge, without a lifetime limit on isolated rejections."""
149
+ def __init__(self, window=1000, max_skipped=100):
150
+ if window < 1 or not 0 <= max_skipped < window:
151
+ raise ValueError("require 0 <= max-skipped < skip-window")
152
+ self.recent = deque(maxlen=window)
153
+ self.skipped = 0
154
+ self.max_skipped = max_skipped
155
+
156
+ def observe(self, accepted):
157
+ if len(self.recent) == self.recent.maxlen:
158
+ self.skipped -= self.recent[0]
159
+ self.recent.append(not accepted)
160
+ self.skipped += not accepted
161
+ if self.skipped > self.max_skipped:
162
+ raise ValueError(f"too many low-coverage teacher mappings: {self.skipped} "
163
+ f"of last {len(self.recent)}; inspect the bridge")
164
+
165
+
166
+ def main():
167
+ ap = argparse.ArgumentParser(description=__doc__)
168
+ ap.add_argument("--vocab", type=Path, required=True)
169
+ ap.add_argument("--vocab-sha256", required=True)
170
+ ap.add_argument("--model", required=True)
171
+ ap.add_argument("--revision", required=True)
172
+ source = ap.add_mutually_exclusive_group(required=True)
173
+ source.add_argument("--prompts", type=Path, help="JSONL user prompts, streamed once")
174
+ source.add_argument("--dataset", help="HF source of user prompts, not teacher targets")
175
+ ap.add_argument("--dataset-revision")
176
+ ap.add_argument("--split", default="train_sft")
177
+ ap.add_argument("--limit", type=int, default=1000, help="maximum accepted streamed records")
178
+ ap.add_argument("--top-k", type=int, default=64)
179
+ ap.add_argument("--min-mass", type=float, default=.05)
180
+ ap.add_argument("--max-skipped", type=int, default=100,
181
+ help="maximum low-coverage rejections within --skip-window")
182
+ ap.add_argument("--skip-window", type=int, default=1000)
183
+ ap.add_argument("--skip-prompts", type=int, default=0,
184
+ help="resume after this many unique represented source prompts")
185
+ args = ap.parse_args()
186
+ if os.environ.get("CUDA_VISIBLE_DEVICES") != GPU1:
187
+ raise ValueError("Qwen requires the pinned GPU 1 UUID in CUDA_VISIBLE_DEVICES")
188
+ if not re.fullmatch(r"[0-9a-f]{40}", args.revision) or "qwen" not in args.model.lower():
189
+ raise ValueError("specify a Qwen model and immutable 40-character HF revision")
190
+ if not 1 <= args.top_k <= 64 or args.limit < 1 or not 0 < args.min_mass <= 1:
191
+ raise ValueError("invalid stream limits")
192
+ guard = MappingGuard(args.skip_window, args.max_skipped)
193
+ if args.skip_prompts < 0:
194
+ raise ValueError("skip-prompts must be nonnegative")
195
+ tok = load_vocab(args.vocab, args.vocab_sha256)
196
+ # Validate the input source before allocating the teacher.
197
+ prompts = iter_prompts(args)
198
+ unique = unique_prompts(tok, prompts, args.skip_prompts)
199
+ first = next(unique, None)
200
+ if first is None:
201
+ raise ValueError("no remaining unique prompts after resume prefix")
202
+ import itertools
203
+ import torch
204
+ from transformers import AutoModelForCausalLM, AutoTokenizer
205
+ if torch.cuda.device_count() != 1:
206
+ raise ValueError("teacher must see exactly one GPU")
207
+ # A bounded teacher allocation on GPU 1; the NYS trainer has its own 28 GiB cap.
208
+ torch.cuda.set_per_process_memory_fraction(.45, 0)
209
+ teacher_tok = AutoTokenizer.from_pretrained(args.model, revision=args.revision, token=False)
210
+ model = AutoModelForCausalLM.from_pretrained(
211
+ args.model, revision=args.revision, token=False, torch_dtype=torch.bfloat16,
212
+ trust_remote_code=False).to("cuda:0").eval()
213
+ if model.config.model_type not in ("qwen2", "qwen3"):
214
+ raise ValueError("this bridge validates Qwen2/Qwen3 text model tokenizers only")
215
+ metadata = dict(event="teacher", model=args.model, revision=args.revision,
216
+ vocab_sha256=args.vocab_sha256, top_k=args.top_k,
217
+ projection="single-NYS-ID, duplicate mass summed, no renormalization",
218
+ dataset=args.dataset, dataset_revision=args.dataset_revision,
219
+ skip_prompts=args.skip_prompts, first_prompt_index=first[0],
220
+ skip_window=args.skip_window, max_skipped=args.max_skipped,
221
+ prompts_sha256=file_sha(args.prompts) if args.prompts else None)
222
+ print(json.dumps(metadata), file=sys.stderr, flush=True)
223
+ out = sys.stdout.buffer
224
+ out.write(HEADER); out.flush()
225
+ emitted = skipped = 0
226
+ for prompt_index, source_row, core, represented in itertools.chain([first], unique):
227
+ # Both models see the same represented user content. Role markers are context only.
228
+ prompt = tok.encode(" user ") + core + tok.encode(" assistant ")
229
+ if len(prompt) > 127:
230
+ raise ValueError("role framing exceeds active-set budget")
231
+ text = teacher_tok.apply_chat_template(
232
+ [{"role": "user", "content": represented}], tokenize=False,
233
+ add_generation_prompt=True, enable_thinking=False)
234
+ inputs = teacher_tok(text, return_tensors="pt").to("cuda:0")
235
+ with torch.inference_mode():
236
+ logits = model(**inputs).logits[0, -1].float()
237
+ probabilities = torch.softmax(logits, dim=-1)
238
+ values, indices = probabilities.topk(args.top_k)
239
+ candidates = [(teacher_tok.decode([i], skip_special_tokens=False,
240
+ clean_up_tokenization_spaces=False), p)
241
+ for i, p in zip(indices.tolist(), values.tolist())
242
+ if i not in teacher_tok.all_special_ids]
243
+ targets, rejected = project_probabilities(tok, candidates)
244
+ retained = sum(p for _, p in targets)
245
+ if not targets or retained < args.min_mass:
246
+ skipped += 1
247
+ print(json.dumps(dict(event="skipped", prompt_index=prompt_index,
248
+ source_row=source_row, retained_mass=retained, rejected=rejected)),
249
+ file=sys.stderr, flush=True)
250
+ guard.observe(False)
251
+ continue
252
+ guard.observe(True)
253
+ out.write(pack_record(prompt, targets)); out.flush()
254
+ emitted += 1
255
+ print(json.dumps(dict(event="record", record=emitted, targets=len(targets),
256
+ prompt_index=prompt_index, source_row=source_row,
257
+ retained_mass=retained, rejected=rejected,
258
+ top_target_ids=targets[:3])), file=sys.stderr, flush=True)
259
+ if emitted >= args.limit:
260
+ break
261
+ if not emitted:
262
+ raise ValueError("no teacher records passed the mapping checks")
263
+ # Tear down a partially consumed Arrow stream while Python still owns the
264
+ # GIL, before emitting DONE or entering interpreter finalization.
265
+ unique.close()
266
+ prompts.close()
267
+ import gc
268
+ gc.collect()
269
+ out.write(FOOTER); out.flush()
270
+ free_bytes, total_bytes = torch.cuda.mem_get_info(0)
271
+ print(json.dumps(dict(event="complete", records=emitted, skipped=skipped,
272
+ peak_teacher_allocated_bytes=torch.cuda.max_memory_allocated(0),
273
+ gpu_free_bytes=free_bytes, gpu_total_bytes=total_bytes)), file=sys.stderr)
274
+
275
+
276
+ if __name__ == "__main__":
277
+ try:
278
+ main()
279
+ except BrokenPipeError:
280
+ raise SystemExit(1)
281
+ except Exception as exc:
282
+ print(f"[qwen-live] {type(exc).__name__}: {exc}", file=sys.stderr)
283
+ raise SystemExit(1)
lineages/corrected-20260930-v2/recovery/gpu_eqprop/run_cuda_l40s.sh ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env bash
2
+ # Physical GPU 1 is pinned by UUID; CUDA then exposes it as logical device 0.
3
+ set -euo pipefail
4
+ cd -- "$(dirname -- "${BASH_SOURCE[0]}")"
5
+ expected=GPU-15278276-8f59-9540-a03f-7188fec578f9
6
+ actual="$(nvidia-smi -i 1 --query-gpu=uuid --format=csv,noheader)"
7
+ [[ "$actual" == "$expected" ]] || { echo 'GPU 1 UUID changed; aborting' >&2; exit 1; }
8
+ export CUDA_VISIBLE_DEVICES="$expected"
9
+ for arg in "$@"; do
10
+ [[ "$arg" != --init ]] || { echo 'This resume wrapper does not allow --init' >&2; exit 2; }
11
+ done
12
+ if (( $# == 0 )); then set -- --probe; fi
13
+ exec ./eqprop_gpu_cuda "$@" --device 0 --vram-limit-gb 28 --omega-scale 0.01
lineages/corrected-20260930-v2/recovery/gpu_eqprop/run_qwen_live.py ADDED
@@ -0,0 +1,127 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """Supervise Qwen -> SDS1 pipe -> NYS. Prints the plan unless --run is supplied.
3
+
4
+ Starts Qwen only after NYS has loaded its checkpoint, so the trainer's memory
5
+ reserve check is deterministic. Never starts upload or retention services.
6
+ """
7
+ import argparse
8
+ import json
9
+ import os
10
+ from pathlib import Path
11
+ import re
12
+ import subprocess
13
+ import sys
14
+
15
+
16
+ def main():
17
+ ap = argparse.ArgumentParser(description=__doc__)
18
+ ap.add_argument("--checkpoint-manifest", type=Path, required=True)
19
+ ap.add_argument("--save-dir", type=Path, required=True)
20
+ ap.add_argument("--corpus", type=Path, action="append", default=[])
21
+ ap.add_argument("--live-mix", type=Path, required=True)
22
+ ap.add_argument("--mix-manifest", type=Path, required=True)
23
+ ap.add_argument("--slots-dir", type=Path,
24
+ help="native speech inputs and matching vocoder for silence/negative filtering")
25
+ ap.add_argument("--max-steps", type=int, required=True, help="absolute final step, including resume")
26
+ ap.add_argument("--run", action="store_true")
27
+ ap.add_argument("teacher_args", nargs=argparse.REMAINDER, help="after --, arguments to qwen_live.py")
28
+ args = ap.parse_args()
29
+ here = Path(__file__).resolve().parent
30
+ root = here.parent
31
+ if args.run and (root/"DO_NOT_RESUME_CURRENT_CHECKPOINTS.json").exists():
32
+ raise ValueError("User stopped this deployment; current run checkpoints must not be resumed")
33
+ manifest = json.loads(args.checkpoint_manifest.read_text())
34
+ ckpt = Path(manifest["local_path"]).resolve()
35
+ if not manifest.get("verified") or ckpt.stat().st_size != 20_640_000_016:
36
+ raise ValueError("requires a verified complete checkpoint manifest")
37
+ match = re.fullmatch(r"gpu_eqprop_(\d+)_\d{8}_\d{6}\.bin", ckpt.name)
38
+ if not match or int(match[1]) != manifest["step"] or not manifest["step"] < args.max_steps <= 2_147_483_647:
39
+ raise ValueError("invalid checkpoint step or absolute stop step")
40
+ save_dir = args.save_dir.resolve()
41
+ if save_dir == ckpt.parent or ckpt.parent in save_dir.parents:
42
+ raise ValueError("write new snapshots outside the source checkpoint directory")
43
+ teacher_args = args.teacher_args
44
+ if teacher_args[:1] == ["--"]:
45
+ teacher_args = teacher_args[1:]
46
+ if not teacher_args:
47
+ raise ValueError("teacher arguments are required after --")
48
+ from validate_curriculum import validate
49
+ curriculum = validate(args.live_mix,args.mix_manifest,args.corpus,args.slots_dir,
50
+ (args.max_steps-manifest['step']+4)//5)
51
+ trainer = ["bash", str(here/"run_cuda_l40s.sh"), "--live", "--distill-stdin",
52
+ "--require-full-curriculum", "--save-first-rotation", "--live-mix", str(args.live_mix.resolve()),
53
+ "--ckpt", str(ckpt), "--resume-steps", str(manifest["step"]),
54
+ "--save-dir", str(save_dir), "--save", str(save_dir/"nys_gpu_checkpoint.bin"),
55
+ "--save-every", "5000", "--log-every", "20", "--max-steps", str(args.max_steps)]
56
+ speech = False
57
+ for corpus in args.corpus:
58
+ if not corpus.is_file():
59
+ raise ValueError(f"missing corpus: {corpus}")
60
+ with corpus.open("rb") as f:
61
+ magic = f.read(4)
62
+ if magic == b"SDS1":
63
+ raise ValueError("Qwen replaces the SDS1 slot; omit the legacy distill corpus")
64
+ speech = speech or magic == b"NYSV"
65
+ trainer += ["--corpus", str(corpus.resolve())]
66
+ if speech and args.slots_dir is None:
67
+ raise ValueError("NYSV training requires --slots-dir with its recovered vocoder")
68
+ if args.slots_dir:
69
+ slots = args.slots_dir.resolve()
70
+ for name in ("compiler_physics.nysa", "speech_vocoder.nyvc", "speech_align.nysv", "speech_slot.nysv"):
71
+ if not (slots/name).is_file():
72
+ raise ValueError(f"missing native speech input: {slots/name}")
73
+ # Global gold filtering must read the same speech records being trained.
74
+ for corpus in args.corpus:
75
+ with corpus.open("rb") as f:
76
+ if f.read(4) == b"NYSV" and corpus.resolve() not in (
77
+ slots/"speech_align.nysv", slots/"speech_slot.nysv"):
78
+ raise ValueError("NYSV corpora must be the speech inputs in --slots-dir")
79
+ trainer += ["--slots-dir", str(slots)]
80
+ teacher = [str(root/"teacher-venv/bin/python"), str(here/"qwen_live.py"), *teacher_args]
81
+ print(json.dumps(dict(trainer=trainer, teacher=teacher, curriculum=curriculum, executes=args.run), indent=2), flush=True)
82
+ if not args.run:
83
+ return 0
84
+ # Rehash immediately before a real run; a stale manifest alone is insufficient.
85
+ from qwen_live import file_sha, GPU1
86
+ if file_sha(ckpt) != manifest["sha256"]:
87
+ raise ValueError("checkpoint no longer matches the verified SHA-256")
88
+ save_dir.mkdir(parents=True, exist_ok=True)
89
+ (save_dir/'curriculum.json').write_text(json.dumps(curriculum,indent=2))
90
+ env = os.environ.copy()
91
+ env.update(CUDA_VISIBLE_DEVICES=GPU1, HF_HOME=str(root/"hf-cache"),
92
+ TOKENIZERS_PARALLELISM="false", PYTHONUNBUFFERED="1")
93
+ prod = None
94
+ with subprocess.Popen(trainer, stdin=subprocess.PIPE, stdout=subprocess.PIPE,
95
+ stderr=subprocess.STDOUT, env=env) as consumer:
96
+ try:
97
+ # Echo trainer diagnostics while waiting for its full checkpoint load.
98
+ for line in iter(consumer.stdout.readline, b""):
99
+ sys.stdout.buffer.write(line); sys.stdout.buffer.flush()
100
+ if b"[stream] ready" in line:
101
+ prod = subprocess.Popen(teacher, stdout=consumer.stdin, env=env)
102
+ consumer.stdin.close(); consumer.stdin = None
103
+ break
104
+ if prod is None:
105
+ return consumer.wait() or 1
106
+ for line in iter(consumer.stdout.readline, b""):
107
+ sys.stdout.buffer.write(line); sys.stdout.buffer.flush()
108
+ code = consumer.wait()
109
+ # A bounded consumer can finish before the next teacher record.
110
+ # Terminate only the exact child we created, never a process-name match.
111
+ if prod.poll() is None:
112
+ prod.terminate()
113
+ prod.wait(timeout=30)
114
+ return code
115
+ return code or prod.returncode
116
+ finally:
117
+ for proc in (prod, consumer):
118
+ if proc is not None and proc.poll() is None:
119
+ proc.terminate()
120
+ try:
121
+ proc.wait(timeout=30)
122
+ except subprocess.TimeoutExpired:
123
+ proc.kill(); proc.wait()
124
+
125
+
126
+ if __name__ == "__main__":
127
+ sys.exit(main())
lineages/corrected-20260930-v2/recovery/gpu_eqprop/start_backup_services.py ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Start one uploader and one verified retention worker for a validated root."""
2
+ import argparse
3
+ import json
4
+ from pathlib import Path
5
+ import subprocess
6
+ import sys
7
+ import time
8
+
9
+ def main():
10
+ p=argparse.ArgumentParser()
11
+ p.add_argument('--root',type=Path,required=True)
12
+ p.add_argument('--kind',choices=['nys','zoey'],required=True)
13
+ p.add_argument('--token-file',type=Path,required=True)
14
+ args=p.parse_args()
15
+ root=args.root.resolve()
16
+ if (root/'DO_NOT_RESUME_CURRENT_CHECKPOINTS.json').exists() or json.loads((root/'lineage.json').read_text())['status']!='VALIDATED':
17
+ raise RuntimeError('Deployment is quarantined or not validated')
18
+ script=root/('gpu_eqprop/hf_checkpoint_service.py' if args.kind=='nys' else 'source/l40s/hf_checkpoint_service.py')
19
+ sys.path.insert(0,str(script.parent))
20
+ from hf_checkpoint_service import service_status
21
+ services=service_status(root)
22
+ base=root/'hf-services'; base.mkdir(exist_ok=True)
23
+ for mode in ('upload','retain'):
24
+ if services[mode]['alive']: continue
25
+ with (base/(mode+'.log')).open('ab',buffering=0) as log:
26
+ subprocess.Popen([str(root/'.venv/bin/python'),'-u',str(script),'--kind',args.kind,
27
+ '--mode',mode,'--root',str(root),'--token-file',str(args.token_file),'--keep','2'],
28
+ stdout=log,stderr=subprocess.STDOUT,stdin=subprocess.DEVNULL,
29
+ start_new_session=True,close_fds=True)
30
+ deadline=time.monotonic()+15
31
+ while time.monotonic()<deadline:
32
+ services=service_status(root)
33
+ if all(services[m]['alive'] for m in ('upload','retain')):
34
+ print(json.dumps(services,indent=2)); return
35
+ time.sleep(.2)
36
+ raise RuntimeError('Backup services did not start; inspect their logs')
37
+
38
+ if __name__=='__main__': main()
lineages/corrected-20260930-v2/recovery/gpu_eqprop/start_qwen_l40s.sh ADDED
@@ -0,0 +1,73 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env bash
2
+ # Authorized 2026-09-30: one bounded live run, GPU 1 only, no HF publishing.
3
+ set -euo pipefail
4
+ ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)"
5
+ cd "$ROOT"
6
+ [[ ! -e "$ROOT/DO_NOT_RESUME_CURRENT_CHECKPOINTS.json" ]] || { echo 'User stopped this deployment. Its run checkpoints are quarantined; use the validated corrected deployment.' >&2; exit 1; }
7
+ RUN_ID="${1:-qwen-20260930}"
8
+ MANIFEST="${2:-$ROOT/source_checkpoint.json}"
9
+ SKIP_PROMPTS="${3:-0}"
10
+ TEACHER_LIMIT="${4:-40000}"
11
+ MAX_STEP="${5:-2075000}"
12
+ [[ "$RUN_ID" =~ ^qwen-[a-z0-9-]+$ ]] || { echo 'Invalid run ID' >&2; exit 1; }
13
+ RUN="$ROOT/runs/$RUN_ID"
14
+ mkdir -p "$RUN"
15
+ exec 9>"$ROOT/runs/trainer.lock"
16
+ flock -n 9 || { echo 'An NYS launch already holds the run lock' >&2; exit 1; }
17
+ [[ ! -e "$RUN/started.json" ]] || { echo 'This run already started; inspect it before resuming' >&2; exit 1; }
18
+ # Reserve space for the remaining periodic saves, two final-save equivalents,
19
+ # and 10 GB headroom. Earlier run snapshots remain intact.
20
+ "$ROOT/.venv/bin/python" - "$RUN" "$MANIFEST" "$SKIP_PROMPTS" "$TEACHER_LIMIT" "$MAX_STEP" <<'PY'
21
+ from datetime import datetime, timezone
22
+ import json, os, shutil, sys
23
+ from pathlib import Path
24
+ run, manifest_path = map(Path, sys.argv[1:3])
25
+ skip, limit, maximum = map(int, sys.argv[3:])
26
+ manifest = json.loads(manifest_path.read_text())
27
+ resume = manifest['step']
28
+ if not manifest.get('verified') or not 0 <= resume < maximum <= 2_147_483_647 or skip < 0 or limit < 1:
29
+ raise ValueError('invalid resume plan')
30
+ needed = (maximum // 5000 - resume // 5000 + 2) * 20_640_000_016 + 10_000_000_000
31
+ reservation = 'all remaining checkpoints'
32
+ if shutil.disk_usage(run).free < needed:
33
+ sys.path.insert(0, str(run.parents[1]/'gpu_eqprop'))
34
+ from hf_checkpoint_service import service_status
35
+ services=service_status(run.parents[1])
36
+ if all(services[mode].get('alive') for mode in ('upload','retain')):
37
+ needed = 4 * 20_640_000_016 + 10_000_000_000
38
+ reservation = 'four checkpoints with verified local retention; trainer stops before low-space writes'
39
+ if shutil.disk_usage(run).free < needed:
40
+ raise ValueError(f'insufficient space: need {needed} bytes for remaining saves')
41
+ started = dict(pid=os.getppid(), started_utc=datetime.now(timezone.utc).isoformat(),
42
+ resume_step=resume, max_step=maximum, teacher_limit=limit,
43
+ teacher_skip_prompts=skip, checkpoint_manifest=str(manifest_path.resolve()),
44
+ reserved_disk_bytes=needed, reservation_policy=reservation)
45
+ (run/'started.json').write_text(json.dumps(started) + '\n')
46
+ active = run.parent/'active_run.json'
47
+ temp = active.with_suffix('.json.writing')
48
+ temp.write_text(json.dumps(dict(run_id=run.name)) + '\n')
49
+ temp.replace(active)
50
+ PY
51
+ export HF_HUB_DISABLE_PROGRESS_BARS=1
52
+ export PYTHONUNBUFFERED=1
53
+ set +e
54
+ "$ROOT/.venv/bin/python" -u gpu_eqprop/run_qwen_live.py \
55
+ --checkpoint-manifest "$MANIFEST" \
56
+ --save-dir "checkpoints/$RUN_ID" --max-steps "$MAX_STEP" \
57
+ --live-mix data/corrected-mix/mix_sft_sparse.bin \
58
+ --mix-manifest data/corrected-mix/mix_manifest.json \
59
+ --corpus data/language_baseline/train_corpus_sparse.bin \
60
+ --corpus data/language_baseline/sft_corpus_sparse.bin \
61
+ --slots-dir data/slots --run \
62
+ -- \
63
+ --model Qwen/Qwen3-8B \
64
+ --revision b968826d9c46dd6066d109eabc6255188de91218 \
65
+ --vocab "$ROOT/data/language_baseline/vocab_sparse.json" \
66
+ --vocab-sha256 23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e \
67
+ --dataset HuggingFaceH4/ultrachat_200k \
68
+ --dataset-revision 8049631c405ae6576f93f445c6b8166f76f5505a \
69
+ --split train_sft --limit "$TEACHER_LIMIT" --skip-prompts "$SKIP_PROMPTS" \
70
+ --skip-window 1000 --max-skipped 100 >> "$RUN/train.log" 2>&1
71
+ code=$?
72
+ printf '{"exit_code":%d,"finished_utc":"%s"}\n' "$code" "$(date -u +%FT%TZ)" > "$RUN/finished.json"
73
+ exit "$code"
lineages/corrected-20260930-v2/recovery/gpu_eqprop/validate_curriculum.py ADDED
@@ -0,0 +1,63 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Fail-closed validation of the complete corrected NYS curriculum."""
2
+ import hashlib
3
+ import json
4
+ from pathlib import Path
5
+ import struct
6
+
7
+ VOCAB_SHA='23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e'
8
+ SOURCE_SHA='406a691b1ac42112705c5193814f76ff064f922161d48ab0e3f67452403bbeec'
9
+ SOURCES={'tulu3','openhermes','ultrachat','xlam','toolace','glaive','hermes_fc','orca','agent_flan'}
10
+
11
+ def sha(path):
12
+ with Path(path).open('rb') as f: return hashlib.file_digest(f,'sha256').hexdigest()
13
+
14
+ def check_file(path,info):
15
+ if path.stat().st_size!=info['bytes'] or sha(path)!=info['sha256']:
16
+ raise ValueError(f'Curriculum size/hash mismatch: {path}')
17
+
18
+ def hard_records(path):
19
+ with path.open('rb') as f:
20
+ count,=struct.unpack('<I',f.read(4))
21
+ if not count: raise ValueError('Empty hard corpus')
22
+ for _ in range(count):
23
+ n,=struct.unpack('<H',f.read(2))
24
+ if not 2<=n<=2048: raise ValueError('Invalid record length')
25
+ ids=struct.unpack('<'+'I'*n,f.read(4*n))
26
+ if max(ids)>=5_000_000: raise ValueError('Out-of-range token')
27
+ if f.read(1): raise ValueError('Trailing hard corpus bytes')
28
+ return count
29
+
30
+ def validate(mix_path,mix_manifest,corpora,slots,max_rotations):
31
+ if mix_path is None or mix_manifest is None or slots is None:
32
+ raise ValueError('Corrected training requires the mixed corpus, its manifest, and native slots')
33
+ m=json.loads(mix_manifest.read_text())
34
+ if not m.get('verified') or m.get('vocab_sha256')!=VOCAB_SHA:
35
+ raise ValueError('Unverified mix or wrong vocabulary')
36
+ if {r['name'] for r in m['sources']}!=SOURCES or any(r['records']<=0 for r in m['sources']):
37
+ raise ValueError('All nine original SFT sources must contribute')
38
+ if any(m.get('first_32000_source_counts',{}).get(name,0)<=0 for name in SOURCES):
39
+ raise ValueError('The bounded run prefix must include every source')
40
+ if m.get('converter_sha256')!='0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649':
41
+ raise ValueError('Mix was not built with the verified frozen-tokenizer converter')
42
+ if sum(r['records'] for r in m['sources'])!=m['records']:
43
+ raise ValueError('Mix source accounting mismatch')
44
+ check_file(mix_path,m)
45
+ if hard_records(mix_path)!=m['records']: raise ValueError('Mix record count mismatch')
46
+ # No wrapping is needed during this bounded recovery run.
47
+ if m['records']<max_rotations: raise ValueError('Insufficient mixed records for the whole run')
48
+ if {p.name for p in corpora}!={'train_corpus_sparse.bin','sft_corpus_sparse.bin'} or len(corpora)!=2:
49
+ raise ValueError('Both original hard-language corpora are required exactly once')
50
+ counts={'mix':m['records']}
51
+ for path in corpora:
52
+ baseline=json.loads((path.parent/'recovery_manifest.json').read_text())
53
+ check_file(path,baseline['files'][path.name])
54
+ check_file(path.parent/'vocab_sparse.json',baseline['files']['vocab_sparse.json'])
55
+ if baseline['files']['vocab_sparse.json']['sha256']!=VOCAB_SHA: raise ValueError('Wrong baseline vocabulary')
56
+ counts[path.name]=hard_records(path)
57
+ slot_manifest=json.loads((slots/'recovery_manifest.json').read_text())
58
+ if not slot_manifest.get('verified'): raise ValueError('Unverified native slots')
59
+ for name in ('compiler_physics.nysa','speech_align.nysv','speech_slot.nysv','speech_vocoder.nyvc'):
60
+ check_file(slots/name,slot_manifest['staged_files'][name])
61
+ counts.update(slot_manifest['record_counts'])
62
+ return dict(verified=True,rotation=['mix','hard','qwen_direct','compiler','speech'],records=counts,
63
+ mix_sha256=m['sha256'],mix_sources=sorted(SOURCES),vocab_sha256=VOCAB_SHA)
lineages/corrected-20260930-v2/recovery/hierarchical_tokenizer.py ADDED
@@ -0,0 +1,126 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import re
3
+ import json
4
+ import logging
5
+ import struct
6
+ from collections import Counter
7
+
8
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s")
9
+ log = logging.getLogger("HierarchicalTokenizer")
10
+
11
+ class HierarchicalTokenizer:
12
+ def __init__(self):
13
+ self.vocab = {} # word/phrase -> token_id
14
+ self.inverse_vocab = {} # token_id -> word/phrase
15
+
16
+ # 1. Initialize base vocabulary with raw bytes (0-255)
17
+ for i in range(256):
18
+ b_str = bytes([i]).decode('latin-1') # safe round-trip representation
19
+ self.vocab[b_str] = i
20
+ self.inverse_vocab[i] = b_str
21
+
22
+ def train(self, texts, target_vocab_size=5000000, max_phrase_length=4):
23
+ log.info(f"Training Hierarchical Phrase Tokenizer to target size: {target_vocab_size}...")
24
+
25
+ # 1. Clean and normalize texts, then extract word n-grams (preserving whitespace)
26
+ word_token_pattern = re.compile(r"\w+|[^\w\s]|\s+")
27
+
28
+ all_words = []
29
+ n_gram_counters = [Counter() for _ in range(max_phrase_length)]
30
+
31
+ log.info("Analyzing text corpora and counting phrase frequencies...")
32
+ for text in texts:
33
+ words = word_token_pattern.findall(text.lower())
34
+ all_words.extend(words)
35
+
36
+ # Record n-grams
37
+ for n in range(1, max_phrase_length + 1):
38
+ for i in range(len(words) - n + 1):
39
+ phrase = " ".join(words[i:i+n])
40
+ n_gram_counters[n-1][phrase] += 1
41
+
42
+ # 2. Extract best words and phrases based on frequencies
43
+ candidates = []
44
+
45
+ # Keep words that appear at least twice
46
+ for word, freq in n_gram_counters[0].items():
47
+ if freq >= 2:
48
+ candidates.append((word, freq, len(word.split())))
49
+
50
+ # Keep multi-word phrases that appear at least 3 times
51
+ for n in range(2, max_phrase_length + 1):
52
+ for phrase, freq in n_gram_counters[n-1].items():
53
+ if freq >= 3:
54
+ candidates.append((phrase, freq, len(phrase.split())))
55
+
56
+ # Sort candidates by frequency (descending)
57
+ candidates.sort(key=lambda x: x[1], reverse=True)
58
+
59
+ # 3. Add to vocabulary up to the target_vocab_size limit
60
+ current_id = 256
61
+ for item, freq, phrase_len in candidates:
62
+ if current_id >= target_vocab_size:
63
+ break
64
+
65
+ # Avoid overwriting the raw byte slots
66
+ if item not in self.vocab:
67
+ self.vocab[item] = current_id
68
+ self.inverse_vocab[current_id] = item
69
+ current_id += 1
70
+
71
+ log.info(f"Vocabulary successfully built. Final size: {len(self.vocab)} tokens (IDs: 0 to {current_id - 1})")
72
+
73
+ def encode(self, text):
74
+ """Greedy longest-match phrase encoding."""
75
+ word_token_pattern = re.compile(r"\w+|[^\w\s]|\s+")
76
+ words = word_token_pattern.findall(text.lower())
77
+
78
+ encoded_tokens = []
79
+ i = 0
80
+ n_words = len(words)
81
+
82
+ while i < n_words:
83
+ matched = False
84
+ # Try to match the longest phrase starting at index i (up to 4 words/spaces)
85
+ for length in range(4, 0, -1):
86
+ if i + length <= n_words:
87
+ phrase = " ".join(words[i:i+length])
88
+ if phrase in self.vocab:
89
+ encoded_tokens.append(self.vocab[phrase])
90
+ i += length
91
+ matched = True
92
+ break
93
+
94
+ if not matched:
95
+ # Character-level fallback for out-of-vocabulary single words/spaces
96
+ oov_word = words[i]
97
+ for char in oov_word:
98
+ # Encode character as its byte equivalent
99
+ char_byte = ord(char) if ord(char) < 256 else 63 # '?' fallback
100
+ encoded_tokens.append(char_byte)
101
+ i += 1
102
+
103
+ return encoded_tokens
104
+
105
+ def decode(self, token_ids):
106
+ """Decode token stream back to text."""
107
+ parts = []
108
+ for tid in token_ids:
109
+ if tid in self.inverse_vocab:
110
+ item = self.inverse_vocab[tid]
111
+ parts.append(item)
112
+ else:
113
+ parts.append(f"<{tid}>")
114
+
115
+ return "".join(parts)
116
+
117
+ def save(self, vocab_path):
118
+ log.info(f"Saving vocabulary to {vocab_path}...")
119
+ with open(vocab_path, "w") as f:
120
+ json.dump(self.vocab, f, indent=2)
121
+
122
+ def load(self, vocab_path):
123
+ log.info(f"Loading vocabulary from {vocab_path}...")
124
+ with open(vocab_path, "r") as f:
125
+ self.vocab = json.load(f)
126
+ self.inverse_vocab = {int(v): k for k, v in self.vocab.items()}
lineages/corrected-20260930-v2/recovery/lineage.json ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "lineage": "corrected-20260930-v2",
3
+ "parent_root": "/data/nys-l40s",
4
+ "quarantined_parent_runs": true,
5
+ "source_step": 1915000,
6
+ "source_revision": "6f1f1e314e6d680f0e1541ddc6b1f6590b46cf40",
7
+ "status": "VALIDATED",
8
+ "full_curriculum_sha256": "e1e79d842ec60f3f646a563850a683e49519c6892807ee3a70a3a60047c3de81",
9
+ "mix_sha256": "425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9",
10
+ "superseded_lineage": "corrected-20260930"
11
+ }
lineages/corrected-20260930-v2/recovery/recovery_policy.json ADDED
@@ -0,0 +1,95 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "created_utc": "2026-09-30T16:51:26.166606+00:00",
3
+ "reason": "User rejected the prior deployments: NYS lacked the required mix; Zoey lacked Gemini chooser",
4
+ "quarantined_root": "/data/nys-l40s",
5
+ "exclude_all_run_outputs": true,
6
+ "original_source_revision": "6f1f1e314e6d680f0e1541ddc6b1f6590b46cf40",
7
+ "corrected_namespace": "lineages/corrected-20260930-v2/",
8
+ "selection_rule": "Use the corrected lineage manifest or the pinned original source, never repository-wide highest step",
9
+ "known_verified_remote_outputs": [
10
+ {
11
+ "bytes": 20640000016,
12
+ "sha256": "0c4763caf56bf393581dac9e032e3ad37cd1b6064f82162381830c289354584e",
13
+ "step": 2000000,
14
+ "remote_path": "gpu_eqprop_2000000_20260930_132916.bin",
15
+ "commit": "e33f42fcb859dc8fd43ce6921923d4552c7b9a93"
16
+ },
17
+ {
18
+ "bytes": 20640000016,
19
+ "sha256": "b5bae8a2a984f0ed01b8f0b16721095719857bb0f38c17c99f9922879b82c941",
20
+ "step": 1995000,
21
+ "remote_path": "gpu_eqprop_1995000_20260930_132014.bin",
22
+ "commit": "592300a451225e4c01825db06b2fd6ccc8d63ee9"
23
+ },
24
+ {
25
+ "bytes": 20640000016,
26
+ "sha256": "09184c4c4e74ec42fe8b5ebd5396573aedcc8ca6e497075b4dcbe8b8be0eba8f",
27
+ "step": 1990000,
28
+ "remote_path": "gpu_eqprop_1990000_20260930_131104.bin",
29
+ "commit": "4482e473d2d0040eece15e8644e6ce2ea4a0c09f"
30
+ },
31
+ {
32
+ "bytes": 20640000016,
33
+ "sha256": "4df3a5a9f724b60239d4c9a4462330425e21641238b3552f13e8b48074df7bc2",
34
+ "step": 1985000,
35
+ "remote_path": "gpu_eqprop_1985000_20260930_130154.bin",
36
+ "commit": "3a374292cb2e5f4d0c3dd43de8e28c237fa5d4eb"
37
+ },
38
+ {
39
+ "bytes": 20640000016,
40
+ "sha256": "f728c272680e0399268acc90f5e51b6345f50d506009f98fe9300f475134232c",
41
+ "step": 2005000,
42
+ "remote_path": "gpu_eqprop_2005000_20260930_150916.bin",
43
+ "commit": "b4b72f23d4997ec0f5d537f089be9b07b398e4b8"
44
+ },
45
+ {
46
+ "bytes": 20640000016,
47
+ "sha256": "363fe72d53955eed71d60a556e66cb6fa2c4322b41a132e67c77900390f2bc20",
48
+ "step": 1980000,
49
+ "remote_path": "gpu_eqprop_1980000_20260930_125247.bin",
50
+ "commit": "5c5bf1bcd676e4dc1d13beafc2d6000d71a5f0dc"
51
+ },
52
+ {
53
+ "bytes": 20640000016,
54
+ "sha256": "652f9acd94c4a977bb1df5466cfe27c911c3e5b2c396d31d81d53f414bca5c0e",
55
+ "step": 1975000,
56
+ "remote_path": "gpu_eqprop_1975000_20260930_124338.bin",
57
+ "commit": "3c98c79f9f4115d0191620874886cedddfc1e4c9"
58
+ },
59
+ {
60
+ "bytes": 20640000016,
61
+ "sha256": "6ce3d17580637beee6e223c33d6197dc4e2fdbfa8172cf8392acdc13eec87a99",
62
+ "step": 1970000,
63
+ "remote_path": "gpu_eqprop_1970000_20260930_123429.bin",
64
+ "commit": "6c91ae6376060ddd4db51f1d6766776c06673d81"
65
+ },
66
+ {
67
+ "bytes": 20640000016,
68
+ "sha256": "89b1f9d3bc939d695e5bfbc580c2b746d2626ca943c58d6e8f744026e12851c8",
69
+ "step": 2010000,
70
+ "remote_path": "gpu_eqprop_2010000_20260930_151818.bin",
71
+ "commit": "842dc9b0ef08798827b489a62e98ecdf40e48e4e"
72
+ },
73
+ {
74
+ "bytes": 20640000016,
75
+ "sha256": "547b726c248de15fe0ff94d7f87a50ee1a9c03cf89312664c5c1f3564110226b",
76
+ "step": 1965000,
77
+ "remote_path": "gpu_eqprop_1965000_20260930_122522.bin",
78
+ "commit": "5015b2ca767d822fbff6f3f286bcf72ca5a7ebcf"
79
+ },
80
+ {
81
+ "bytes": 20640000016,
82
+ "sha256": "7dd3b898d5b2cb7487a58ebdb628077d5e60b721cce664701a65aa0d59373545",
83
+ "step": 2015000,
84
+ "remote_path": "gpu_eqprop_2015000_20260930_152729.bin",
85
+ "commit": "19f0c19d07081b9f974d1cbc8d9da4a0bd6f06d1"
86
+ }
87
+ ],
88
+ "additional_excluded_namespaces": [
89
+ "lineages/corrected-20260930/"
90
+ ],
91
+ "additional_excluded_runs": [
92
+ "qwen-corrected-20260930"
93
+ ],
94
+ "additional_exclusion_reason": "Incomplete rotation guard stopped the first correction before silent speech rows were handled within their slot"
95
+ }
lineages/corrected-20260930-v2/recovery/source_checkpoint.json ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "repo": "YNSScarSaiyan/nys-sft-public",
3
+ "revision": "6f1f1e314e6d680f0e1541ddc6b1f6590b46cf40",
4
+ "filename": "gpu_eqprop_1915000_20260913_095122.bin",
5
+ "step": 1915000,
6
+ "bytes": 20640000016,
7
+ "sha256": "406a691b1ac42112705c5193814f76ff064f922161d48ab0e3f67452403bbeec",
8
+ "N": 5000000,
9
+ "k": 512,
10
+ "local_path": "/data/nys-l40s/checkpoints/source/gpu_eqprop_1915000_20260913_095122.bin",
11
+ "verified": true
12
+ }
lineages/corrected-20260930-v2/recovery_policy.json ADDED
@@ -0,0 +1,95 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "created_utc": "2026-09-30T16:51:26.166606+00:00",
3
+ "reason": "User rejected the prior deployments: NYS lacked the required mix; Zoey lacked Gemini chooser",
4
+ "quarantined_root": "/data/nys-l40s",
5
+ "exclude_all_run_outputs": true,
6
+ "original_source_revision": "6f1f1e314e6d680f0e1541ddc6b1f6590b46cf40",
7
+ "corrected_namespace": "lineages/corrected-20260930-v2/",
8
+ "selection_rule": "Use the corrected lineage manifest or the pinned original source, never repository-wide highest step",
9
+ "known_verified_remote_outputs": [
10
+ {
11
+ "bytes": 20640000016,
12
+ "sha256": "0c4763caf56bf393581dac9e032e3ad37cd1b6064f82162381830c289354584e",
13
+ "step": 2000000,
14
+ "remote_path": "gpu_eqprop_2000000_20260930_132916.bin",
15
+ "commit": "e33f42fcb859dc8fd43ce6921923d4552c7b9a93"
16
+ },
17
+ {
18
+ "bytes": 20640000016,
19
+ "sha256": "b5bae8a2a984f0ed01b8f0b16721095719857bb0f38c17c99f9922879b82c941",
20
+ "step": 1995000,
21
+ "remote_path": "gpu_eqprop_1995000_20260930_132014.bin",
22
+ "commit": "592300a451225e4c01825db06b2fd6ccc8d63ee9"
23
+ },
24
+ {
25
+ "bytes": 20640000016,
26
+ "sha256": "09184c4c4e74ec42fe8b5ebd5396573aedcc8ca6e497075b4dcbe8b8be0eba8f",
27
+ "step": 1990000,
28
+ "remote_path": "gpu_eqprop_1990000_20260930_131104.bin",
29
+ "commit": "4482e473d2d0040eece15e8644e6ce2ea4a0c09f"
30
+ },
31
+ {
32
+ "bytes": 20640000016,
33
+ "sha256": "4df3a5a9f724b60239d4c9a4462330425e21641238b3552f13e8b48074df7bc2",
34
+ "step": 1985000,
35
+ "remote_path": "gpu_eqprop_1985000_20260930_130154.bin",
36
+ "commit": "3a374292cb2e5f4d0c3dd43de8e28c237fa5d4eb"
37
+ },
38
+ {
39
+ "bytes": 20640000016,
40
+ "sha256": "f728c272680e0399268acc90f5e51b6345f50d506009f98fe9300f475134232c",
41
+ "step": 2005000,
42
+ "remote_path": "gpu_eqprop_2005000_20260930_150916.bin",
43
+ "commit": "b4b72f23d4997ec0f5d537f089be9b07b398e4b8"
44
+ },
45
+ {
46
+ "bytes": 20640000016,
47
+ "sha256": "363fe72d53955eed71d60a556e66cb6fa2c4322b41a132e67c77900390f2bc20",
48
+ "step": 1980000,
49
+ "remote_path": "gpu_eqprop_1980000_20260930_125247.bin",
50
+ "commit": "5c5bf1bcd676e4dc1d13beafc2d6000d71a5f0dc"
51
+ },
52
+ {
53
+ "bytes": 20640000016,
54
+ "sha256": "652f9acd94c4a977bb1df5466cfe27c911c3e5b2c396d31d81d53f414bca5c0e",
55
+ "step": 1975000,
56
+ "remote_path": "gpu_eqprop_1975000_20260930_124338.bin",
57
+ "commit": "3c98c79f9f4115d0191620874886cedddfc1e4c9"
58
+ },
59
+ {
60
+ "bytes": 20640000016,
61
+ "sha256": "6ce3d17580637beee6e223c33d6197dc4e2fdbfa8172cf8392acdc13eec87a99",
62
+ "step": 1970000,
63
+ "remote_path": "gpu_eqprop_1970000_20260930_123429.bin",
64
+ "commit": "6c91ae6376060ddd4db51f1d6766776c06673d81"
65
+ },
66
+ {
67
+ "bytes": 20640000016,
68
+ "sha256": "89b1f9d3bc939d695e5bfbc580c2b746d2626ca943c58d6e8f744026e12851c8",
69
+ "step": 2010000,
70
+ "remote_path": "gpu_eqprop_2010000_20260930_151818.bin",
71
+ "commit": "842dc9b0ef08798827b489a62e98ecdf40e48e4e"
72
+ },
73
+ {
74
+ "bytes": 20640000016,
75
+ "sha256": "547b726c248de15fe0ff94d7f87a50ee1a9c03cf89312664c5c1f3564110226b",
76
+ "step": 1965000,
77
+ "remote_path": "gpu_eqprop_1965000_20260930_122522.bin",
78
+ "commit": "5015b2ca767d822fbff6f3f286bcf72ca5a7ebcf"
79
+ },
80
+ {
81
+ "bytes": 20640000016,
82
+ "sha256": "7dd3b898d5b2cb7487a58ebdb628077d5e60b721cce664701a65aa0d59373545",
83
+ "step": 2015000,
84
+ "remote_path": "gpu_eqprop_2015000_20260930_152729.bin",
85
+ "commit": "19f0c19d07081b9f974d1cbc8d9da4a0bd6f06d1"
86
+ }
87
+ ],
88
+ "additional_excluded_namespaces": [
89
+ "lineages/corrected-20260930/"
90
+ ],
91
+ "additional_excluded_runs": [
92
+ "qwen-corrected-20260930"
93
+ ],
94
+ "additional_exclusion_reason": "Incomplete rotation guard stopped the first correction before silent speech rows were handled within their slot"
95
+ }
lineages/corrected-20260930/LATEST.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"path": "lineages/corrected-20260930/gpu_eqprop_1915005_20260930_163300.bin", "step": 1915005, "bytes": 20640000016, "sha256": "c94c93db11ad89876d0b18b9ae1e1542e0e6a20df3366a8ecc0dad568b82a334"}
lineages/corrected-20260930/curriculum.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "verified": true,
3
+ "rotation": [
4
+ "mix",
5
+ "hard",
6
+ "qwen_direct",
7
+ "compiler",
8
+ "speech"
9
+ ],
10
+ "records": {
11
+ "mix": 2050000,
12
+ "train_corpus_sparse.bin": 255435,
13
+ "sft_corpus_sparse.bin": 50000,
14
+ "speech_align.nysv": 305,
15
+ "speech_slot.nysv": 6,
16
+ "compiler_physics.nysa": 412
17
+ },
18
+ "mix_sha256": "425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9",
19
+ "mix_sources": [
20
+ "agent_flan",
21
+ "glaive",
22
+ "hermes_fc",
23
+ "openhermes",
24
+ "orca",
25
+ "toolace",
26
+ "tulu3",
27
+ "ultrachat",
28
+ "xlam"
29
+ ],
30
+ "vocab_sha256": "23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e"
31
+ }
lineages/corrected-20260930/lineage.json ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "lineage": "corrected-20260930",
3
+ "parent_root": "/data/nys-l40s",
4
+ "quarantined_parent_runs": true,
5
+ "source_step": 1915000,
6
+ "source_revision": "6f1f1e314e6d680f0e1541ddc6b1f6590b46cf40",
7
+ "status": "VALIDATED",
8
+ "full_curriculum_sha256": "ee1184e158efb81e40a9aa6cede9884b774ce71acc766cbc2a1aea3d733b0b4a",
9
+ "mix_sha256": "425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9"
10
+ }
lineages/corrected-20260930/mix_manifest.json ADDED
@@ -0,0 +1,187 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "verified": true,
3
+ "records": 2050000,
4
+ "bytes": 810830240,
5
+ "sha256": "425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9",
6
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649",
7
+ "vocab_sha256": "23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e",
8
+ "sources": [
9
+ {
10
+ "name": "tulu3",
11
+ "repo": "allenai/tulu-3-sft-mixture",
12
+ "revision": "b14afda60f1bbebe55d5d2fa1e4df5042f97f8be",
13
+ "config": null,
14
+ "records": 273333,
15
+ "target_records": 273333,
16
+ "bytes": 74691934,
17
+ "sha256": "a6791c7f5ebfbf73982f16200c57da1f986ec82835e26d310337723811488efe",
18
+ "fed_conversations": {
19
+ "train": 14280
20
+ },
21
+ "normalized_input_sha256": "43a1ceebba233ccc1295c0e6e4c38df03ca107aeca88ba76aa8ebdff3db05739",
22
+ "max_per_conversation": 25,
23
+ "skipped_no_assistant_targets": {},
24
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
25
+ },
26
+ {
27
+ "name": "openhermes",
28
+ "repo": "teknium/OpenHermes-2.5",
29
+ "revision": "b82037821055c377bed0d495e72e46de3bc72e84",
30
+ "config": null,
31
+ "records": 205000,
32
+ "target_records": 205000,
33
+ "bytes": 37265984,
34
+ "sha256": "a14995d3982265093dade0e5456bf4e807adc1606502a008e010bf1cae91a86e",
35
+ "fed_conversations": {
36
+ "train": 9472
37
+ },
38
+ "normalized_input_sha256": "0c8290012cd577e5fa0e719b3e1a109d0be0e7352a180fcc87d558123f96b55d",
39
+ "max_per_conversation": 25,
40
+ "skipped_no_assistant_targets": {},
41
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
42
+ },
43
+ {
44
+ "name": "ultrachat",
45
+ "repo": "HuggingFaceH4/ultrachat_200k",
46
+ "revision": "8049631c405ae6576f93f445c6b8166f76f5505a",
47
+ "config": null,
48
+ "records": 205000,
49
+ "target_records": 205000,
50
+ "bytes": 54089028,
51
+ "sha256": "ca6fe7a2c74cf1612a51d6c8b425c4869a5147f56647581a318b148a1660fca0",
52
+ "fed_conversations": {
53
+ "train_sft": 8211
54
+ },
55
+ "normalized_input_sha256": "c5e3c60cfe787c0ee6dc15a934e2d9f0de46a866faafb8d871464f69972301de",
56
+ "max_per_conversation": 25,
57
+ "skipped_no_assistant_targets": {},
58
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
59
+ },
60
+ {
61
+ "name": "xlam",
62
+ "repo": "Salesforce/xlam-function-calling-60k",
63
+ "revision": "26d14ebfe18b1f7b524bd39b404b50af5dc97866",
64
+ "config": "dataset",
65
+ "records": 273333,
66
+ "target_records": 273333,
67
+ "bytes": 133978174,
68
+ "sha256": "35fa9266bf1f5080dd52798145a68fcfaae141e5ae7060e9049d8b5518dc9f83",
69
+ "fed_conversations": {
70
+ "train": 11216
71
+ },
72
+ "normalized_input_sha256": "4d384db283209ff38aecbd1fc07498e9d2e9158aa5c151e488a6e3bcfa7dba73",
73
+ "max_per_conversation": 25,
74
+ "skipped_no_assistant_targets": {},
75
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
76
+ },
77
+ {
78
+ "name": "toolace",
79
+ "repo": "Team-ACE/ToolACE",
80
+ "revision": "6bda777c88d21e5a204703c1ee45597a8fa4f734",
81
+ "config": null,
82
+ "records": 205000,
83
+ "target_records": 205000,
84
+ "bytes": 105319644,
85
+ "sha256": "4e2351be623bf5d2cb4e6c72fe5454289dbb467055d8ab5c4ef9f9a50cca8b36",
86
+ "fed_conversations": {
87
+ "train": 9014
88
+ },
89
+ "normalized_input_sha256": "129d36d3541f946be6d6199eaed8ab7a796cdd54ffc45abe735f9837c79e7e3d",
90
+ "max_per_conversation": 25,
91
+ "skipped_no_assistant_targets": {},
92
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
93
+ },
94
+ {
95
+ "name": "glaive",
96
+ "repo": "glaiveai/glaive-function-calling-v2",
97
+ "revision": "e7f4b6456019f5d8bcb991ef0dd67d8ff23221ac",
98
+ "config": null,
99
+ "records": 205000,
100
+ "target_records": 205000,
101
+ "bytes": 80593548,
102
+ "sha256": "c1e0f9b1b3e57bea3d726ee282748959aaeefdce5a01dea4d10e980a152ee1fa",
103
+ "fed_conversations": {
104
+ "train": 8320
105
+ },
106
+ "normalized_input_sha256": "9a8a47aca6a49e0054690a267bdc610e9f9ffc2b67ae3782f00aae34921d33b7",
107
+ "max_per_conversation": 25,
108
+ "skipped_no_assistant_targets": {},
109
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
110
+ },
111
+ {
112
+ "name": "hermes_fc",
113
+ "repo": "NousResearch/hermes-function-calling-v1",
114
+ "revision": "dae3e1d28cfbcf4b915c04ea1e072030529b4bda",
115
+ "config": "func_calling_singleturn",
116
+ "records": 47325,
117
+ "target_records": 136666,
118
+ "bytes": 24325054,
119
+ "sha256": "b83105e2de0c9317cbdf064b8a022c0e2e6994b20ee95a9e2aee28fa7a4ea2b3",
120
+ "fed_conversations": {
121
+ "train": 1893
122
+ },
123
+ "normalized_input_sha256": "f9faf972ab53f865c517574dd83540982ec94b3d19024acf941bd627117e09d5",
124
+ "max_per_conversation": 25,
125
+ "skipped_no_assistant_targets": {},
126
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
127
+ },
128
+ {
129
+ "name": "orca",
130
+ "repo": "microsoft/orca-agentinstruct-1M-v1",
131
+ "revision": "86d609183249ff8037eae33d76ebca3af9390ea8",
132
+ "config": null,
133
+ "records": 273333,
134
+ "target_records": 273333,
135
+ "bytes": 115045826,
136
+ "sha256": "b9194dd3c45ac353809a19c851ef5935ef06d568c46943ab021963a4a8adf677",
137
+ "fed_conversations": {
138
+ "analytical_reasoning": 1873,
139
+ "code_": 1873,
140
+ "rag": 1872,
141
+ "follow_up": 1872,
142
+ "fs_cot_flow": 1872,
143
+ "open_domain_qa": 1872
144
+ },
145
+ "normalized_input_sha256": "c4b165c28a5e4ba92304e38d23f2da93cf691eb9eb137b847aa9f6f5592b7180",
146
+ "max_per_conversation": 25,
147
+ "skipped_no_assistant_targets": {},
148
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
149
+ },
150
+ {
151
+ "name": "agent_flan",
152
+ "repo": "internlm/Agent-FLAN",
153
+ "revision": "8b25999e795a58b264fcb51e8746edb2faee9161",
154
+ "config": null,
155
+ "records": 362676,
156
+ "target_records": 362676,
157
+ "bytes": 185521080,
158
+ "sha256": "b760628a29e8f7e50237379d2ba4f4aaf4400b771706347032a80bc032524856",
159
+ "fed_conversations": {
160
+ "agent_instruct_tflan": 1730,
161
+ "agent_instruct_react": 1730,
162
+ "toolbench_tflan_60p_r10r5u7": 5530,
163
+ "toolbench_instruct_j1s1_3k": 5530
164
+ },
165
+ "normalized_input_sha256": "e67470488beec5b7dbc132d818185f9dc2c3fa5d39859dc59e27573e90582fca",
166
+ "max_per_conversation": 25,
167
+ "skipped_no_assistant_targets": {
168
+ "agent_instruct_react": 1,
169
+ "agent_instruct_tflan": 1
170
+ },
171
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
172
+ }
173
+ ],
174
+ "shuffle_seed": 7,
175
+ "first_32000_source_counts": {
176
+ "tulu3": 4287,
177
+ "orca": 4204,
178
+ "agent_flan": 5622,
179
+ "xlam": 4393,
180
+ "ultrachat": 3193,
181
+ "openhermes": 3171,
182
+ "toolace": 3164,
183
+ "glaive": 3219,
184
+ "hermes_fc": 747
185
+ },
186
+ "note": "Rebuilt original nine-source recipe; not the unavailable historical byte-for-byte mix"
187
+ }
lineages/corrected-20260930/recovery/backup-manifest.json ADDED
@@ -0,0 +1,119 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "checkpoint": "lineages/corrected-20260930/gpu_eqprop_1915005_20260930_163300.bin",
3
+ "checkpoint_sha256": "c94c93db11ad89876d0b18b9ae1e1542e0e6a20df3366a8ecc0dad568b82a334",
4
+ "captured_utc": "2026-09-30T16:36:56.115009+00:00",
5
+ "files": {
6
+ "data/corrected-mix/mix_sft_sparse.bin": {
7
+ "bytes": 810830240,
8
+ "sha256": "425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9"
9
+ },
10
+ "data/corrected-mix/mix_manifest.json": {
11
+ "bytes": 7067,
12
+ "sha256": "bf22f42a0fd2fdcbdc23e60821e0614cd1bc5ab69b7c1589efc762c5adc4c624"
13
+ },
14
+ "data/language_baseline/vocab_sparse.json": {
15
+ "bytes": 106816698,
16
+ "sha256": "23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e"
17
+ },
18
+ "data/language_baseline/train_corpus_sparse.bin": {
19
+ "bytes": 66816546,
20
+ "sha256": "a9b901c8c02b35eebe638abc0bc8ae551e359f51370890733ac917f0c8a97916"
21
+ },
22
+ "data/language_baseline/sft_corpus_sparse.bin": {
23
+ "bytes": 13200476,
24
+ "sha256": "91cc73e689d8e8e457221d3a20bb4d552cd9d2e5df1c1ed2f7ad2ce5c796cd60"
25
+ },
26
+ "data/language_baseline/recovery_manifest.json": {
27
+ "bytes": 703,
28
+ "sha256": "aac6d1a44e1037fcac4de6b3a9cc8de1e714c357f7495e9aff4fe6309dd561cb"
29
+ },
30
+ "data/slots/compiler_physics.nysa": {
31
+ "bytes": 48815,
32
+ "sha256": "052fc7dee4b1081392051b9fde2fac6533aa77e82dbea56ac32c27a67ced6cf8"
33
+ },
34
+ "data/slots/speech_align.nysv": {
35
+ "bytes": 1490468,
36
+ "sha256": "fea716002f75cde8a4567d64e28c14e353c3cdb349cd9ec8ecf3c991f52fbf8c"
37
+ },
38
+ "data/slots/speech_slot.nysv": {
39
+ "bytes": 15284,
40
+ "sha256": "2988cc801e2b95d6ebba35d335ec656486bc0fe0b5938d071a536c63879aca0c"
41
+ },
42
+ "data/slots/speech_vocoder.nyvc": {
43
+ "bytes": 20500,
44
+ "sha256": "028d0e759f9721b2110d8527028e7de3d80d5b611401d33db895d48e2f2df4d1"
45
+ },
46
+ "data/slots/recovery_manifest.json": {
47
+ "bytes": 3007,
48
+ "sha256": "b74dc14a623e7a4e1b9b1726d1c162a2e23cc19d373e9f53e1fd9064eb46dc35"
49
+ },
50
+ "gpu_eqprop/build_cuda.sh": {
51
+ "bytes": 287,
52
+ "sha256": "8c2148c696f5f4520b9cfa13bc6e80accc3de5630e449de0197391877db13173"
53
+ },
54
+ "gpu_eqprop/run_cuda_l40s.sh": {
55
+ "bytes": 640,
56
+ "sha256": "7f0360130f5d5f28c22e40310b9f6cec532ee0d8afd2cd1cf3afdd319375f96c"
57
+ },
58
+ "gpu_eqprop/run_qwen_live.py": {
59
+ "bytes": 6718,
60
+ "sha256": "deac34a492afb95687a97200955debb38d8a1f4e2c4bf39e8e6a64315e815ffb"
61
+ },
62
+ "gpu_eqprop/start_qwen_l40s.sh": {
63
+ "bytes": 3840,
64
+ "sha256": "b018e78b46355d022614352b362ee8bdc3bf8e701758cf97b60a7419ef33b0f8"
65
+ },
66
+ "gpu_eqprop/qwen_live.py": {
67
+ "bytes": 12912,
68
+ "sha256": "40ee3d2f4ba916f3d93db3f515a21af1feec3a1572b9d782798711ae7d787cf7"
69
+ },
70
+ "gpu_eqprop/validate_curriculum.py": {
71
+ "bytes": 3744,
72
+ "sha256": "d433f10fc417e2a87154c2c0e9721271d084ccb12f6bcda335c2a995cf79af02"
73
+ },
74
+ "gpu_eqprop/eqprop_gpu.cu": {
75
+ "bytes": 83588,
76
+ "sha256": "352fe64f735221128fe883c9d2baac13b2f66c4278bd24b471f351a2467c5fd9"
77
+ },
78
+ "gpu_eqprop/distill_stdin.h": {
79
+ "bytes": 4590,
80
+ "sha256": "6ad021d70316d74f29a43359f3436669d57827c84bdd47919e9c907f75c42eb3"
81
+ },
82
+ "gpu_eqprop/cuda_self_test.h": {
83
+ "bytes": 8128,
84
+ "sha256": "96e95af580d1d133cf72c7636f2da52ef0358f4d7e7978c5050029f8f61e0ff2"
85
+ },
86
+ "gpu_eqprop/hf_checkpoint_service.py": {
87
+ "bytes": 21698,
88
+ "sha256": "31e700a0471ad1e2c8fbe5fda61ed536f33bb57d72be8c0c73a728a145eb4172"
89
+ },
90
+ "gpu_eqprop/start_backup_services.py": {
91
+ "bytes": 1845,
92
+ "sha256": "de96145ac8e0ac90b48a72f9e199b5c7879e29967f9a52d5b5bc28493cd4c56f"
93
+ },
94
+ "gpu_eqprop/live_status.py": {
95
+ "bytes": 4503,
96
+ "sha256": "6de382c1565f95ca0ce4b6c94846c5c128b32c6af8e65bf7c980a0c48800c3a4"
97
+ },
98
+ "gpu_eqprop/build_corrected_mix.py": {
99
+ "bytes": 10711,
100
+ "sha256": "a2bcd06abe180432c009d84b6a298b9c6d893747a60f6a16f4eea5977136b27f"
101
+ },
102
+ "hierarchical_tokenizer.py": {
103
+ "bytes": 4963,
104
+ "sha256": "e7bacce12c795901359a57dbe2fc53915aa79f9a9c446ec5725c3e6e3f4a809f"
105
+ },
106
+ "source_checkpoint.json": {
107
+ "bytes": 422,
108
+ "sha256": "bda715ec4a981fd23d2164206b37956f639102fa20cd7c621f93bf81c018a8cc"
109
+ },
110
+ "lineage.json": {
111
+ "bytes": 403,
112
+ "sha256": "b0be0d97349ac609f0370311cc1a19795c68cc4deee088de246201e13a13d760"
113
+ },
114
+ "recovery_policy.json": {
115
+ "bytes": 3504,
116
+ "sha256": "6d5db68cb9911118b0f2df3c5b3262d00c616cb3801a27d3087c28bc099e1608"
117
+ }
118
+ }
119
+ }
lineages/corrected-20260930/recovery/data/corrected-mix/mix_manifest.json ADDED
@@ -0,0 +1,187 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "verified": true,
3
+ "records": 2050000,
4
+ "bytes": 810830240,
5
+ "sha256": "425f2adaa8ab2a380b273d7abba3401855fcd792475c742627c0befa419273a9",
6
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649",
7
+ "vocab_sha256": "23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e",
8
+ "sources": [
9
+ {
10
+ "name": "tulu3",
11
+ "repo": "allenai/tulu-3-sft-mixture",
12
+ "revision": "b14afda60f1bbebe55d5d2fa1e4df5042f97f8be",
13
+ "config": null,
14
+ "records": 273333,
15
+ "target_records": 273333,
16
+ "bytes": 74691934,
17
+ "sha256": "a6791c7f5ebfbf73982f16200c57da1f986ec82835e26d310337723811488efe",
18
+ "fed_conversations": {
19
+ "train": 14280
20
+ },
21
+ "normalized_input_sha256": "43a1ceebba233ccc1295c0e6e4c38df03ca107aeca88ba76aa8ebdff3db05739",
22
+ "max_per_conversation": 25,
23
+ "skipped_no_assistant_targets": {},
24
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
25
+ },
26
+ {
27
+ "name": "openhermes",
28
+ "repo": "teknium/OpenHermes-2.5",
29
+ "revision": "b82037821055c377bed0d495e72e46de3bc72e84",
30
+ "config": null,
31
+ "records": 205000,
32
+ "target_records": 205000,
33
+ "bytes": 37265984,
34
+ "sha256": "a14995d3982265093dade0e5456bf4e807adc1606502a008e010bf1cae91a86e",
35
+ "fed_conversations": {
36
+ "train": 9472
37
+ },
38
+ "normalized_input_sha256": "0c8290012cd577e5fa0e719b3e1a109d0be0e7352a180fcc87d558123f96b55d",
39
+ "max_per_conversation": 25,
40
+ "skipped_no_assistant_targets": {},
41
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
42
+ },
43
+ {
44
+ "name": "ultrachat",
45
+ "repo": "HuggingFaceH4/ultrachat_200k",
46
+ "revision": "8049631c405ae6576f93f445c6b8166f76f5505a",
47
+ "config": null,
48
+ "records": 205000,
49
+ "target_records": 205000,
50
+ "bytes": 54089028,
51
+ "sha256": "ca6fe7a2c74cf1612a51d6c8b425c4869a5147f56647581a318b148a1660fca0",
52
+ "fed_conversations": {
53
+ "train_sft": 8211
54
+ },
55
+ "normalized_input_sha256": "c5e3c60cfe787c0ee6dc15a934e2d9f0de46a866faafb8d871464f69972301de",
56
+ "max_per_conversation": 25,
57
+ "skipped_no_assistant_targets": {},
58
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
59
+ },
60
+ {
61
+ "name": "xlam",
62
+ "repo": "Salesforce/xlam-function-calling-60k",
63
+ "revision": "26d14ebfe18b1f7b524bd39b404b50af5dc97866",
64
+ "config": "dataset",
65
+ "records": 273333,
66
+ "target_records": 273333,
67
+ "bytes": 133978174,
68
+ "sha256": "35fa9266bf1f5080dd52798145a68fcfaae141e5ae7060e9049d8b5518dc9f83",
69
+ "fed_conversations": {
70
+ "train": 11216
71
+ },
72
+ "normalized_input_sha256": "4d384db283209ff38aecbd1fc07498e9d2e9158aa5c151e488a6e3bcfa7dba73",
73
+ "max_per_conversation": 25,
74
+ "skipped_no_assistant_targets": {},
75
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
76
+ },
77
+ {
78
+ "name": "toolace",
79
+ "repo": "Team-ACE/ToolACE",
80
+ "revision": "6bda777c88d21e5a204703c1ee45597a8fa4f734",
81
+ "config": null,
82
+ "records": 205000,
83
+ "target_records": 205000,
84
+ "bytes": 105319644,
85
+ "sha256": "4e2351be623bf5d2cb4e6c72fe5454289dbb467055d8ab5c4ef9f9a50cca8b36",
86
+ "fed_conversations": {
87
+ "train": 9014
88
+ },
89
+ "normalized_input_sha256": "129d36d3541f946be6d6199eaed8ab7a796cdd54ffc45abe735f9837c79e7e3d",
90
+ "max_per_conversation": 25,
91
+ "skipped_no_assistant_targets": {},
92
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
93
+ },
94
+ {
95
+ "name": "glaive",
96
+ "repo": "glaiveai/glaive-function-calling-v2",
97
+ "revision": "e7f4b6456019f5d8bcb991ef0dd67d8ff23221ac",
98
+ "config": null,
99
+ "records": 205000,
100
+ "target_records": 205000,
101
+ "bytes": 80593548,
102
+ "sha256": "c1e0f9b1b3e57bea3d726ee282748959aaeefdce5a01dea4d10e980a152ee1fa",
103
+ "fed_conversations": {
104
+ "train": 8320
105
+ },
106
+ "normalized_input_sha256": "9a8a47aca6a49e0054690a267bdc610e9f9ffc2b67ae3782f00aae34921d33b7",
107
+ "max_per_conversation": 25,
108
+ "skipped_no_assistant_targets": {},
109
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
110
+ },
111
+ {
112
+ "name": "hermes_fc",
113
+ "repo": "NousResearch/hermes-function-calling-v1",
114
+ "revision": "dae3e1d28cfbcf4b915c04ea1e072030529b4bda",
115
+ "config": "func_calling_singleturn",
116
+ "records": 47325,
117
+ "target_records": 136666,
118
+ "bytes": 24325054,
119
+ "sha256": "b83105e2de0c9317cbdf064b8a022c0e2e6994b20ee95a9e2aee28fa7a4ea2b3",
120
+ "fed_conversations": {
121
+ "train": 1893
122
+ },
123
+ "normalized_input_sha256": "f9faf972ab53f865c517574dd83540982ec94b3d19024acf941bd627117e09d5",
124
+ "max_per_conversation": 25,
125
+ "skipped_no_assistant_targets": {},
126
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
127
+ },
128
+ {
129
+ "name": "orca",
130
+ "repo": "microsoft/orca-agentinstruct-1M-v1",
131
+ "revision": "86d609183249ff8037eae33d76ebca3af9390ea8",
132
+ "config": null,
133
+ "records": 273333,
134
+ "target_records": 273333,
135
+ "bytes": 115045826,
136
+ "sha256": "b9194dd3c45ac353809a19c851ef5935ef06d568c46943ab021963a4a8adf677",
137
+ "fed_conversations": {
138
+ "analytical_reasoning": 1873,
139
+ "code_": 1873,
140
+ "rag": 1872,
141
+ "follow_up": 1872,
142
+ "fs_cot_flow": 1872,
143
+ "open_domain_qa": 1872
144
+ },
145
+ "normalized_input_sha256": "c4b165c28a5e4ba92304e38d23f2da93cf691eb9eb137b847aa9f6f5592b7180",
146
+ "max_per_conversation": 25,
147
+ "skipped_no_assistant_targets": {},
148
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
149
+ },
150
+ {
151
+ "name": "agent_flan",
152
+ "repo": "internlm/Agent-FLAN",
153
+ "revision": "8b25999e795a58b264fcb51e8746edb2faee9161",
154
+ "config": null,
155
+ "records": 362676,
156
+ "target_records": 362676,
157
+ "bytes": 185521080,
158
+ "sha256": "b760628a29e8f7e50237379d2ba4f4aaf4400b771706347032a80bc032524856",
159
+ "fed_conversations": {
160
+ "agent_instruct_tflan": 1730,
161
+ "agent_instruct_react": 1730,
162
+ "toolbench_tflan_60p_r10r5u7": 5530,
163
+ "toolbench_instruct_j1s1_3k": 5530
164
+ },
165
+ "normalized_input_sha256": "e67470488beec5b7dbc132d818185f9dc2c3fa5d39859dc59e27573e90582fca",
166
+ "max_per_conversation": 25,
167
+ "skipped_no_assistant_targets": {
168
+ "agent_instruct_react": 1,
169
+ "agent_instruct_tflan": 1
170
+ },
171
+ "converter_sha256": "0e478ad6b522e193fc052888a269aa2aa076633183e29f6e5cf2b7263d413649"
172
+ }
173
+ ],
174
+ "shuffle_seed": 7,
175
+ "first_32000_source_counts": {
176
+ "tulu3": 4287,
177
+ "orca": 4204,
178
+ "agent_flan": 5622,
179
+ "xlam": 4393,
180
+ "ultrachat": 3193,
181
+ "openhermes": 3171,
182
+ "toolace": 3164,
183
+ "glaive": 3219,
184
+ "hermes_fc": 747
185
+ },
186
+ "note": "Rebuilt original nine-source recipe; not the unavailable historical byte-for-byte mix"
187
+ }
lineages/corrected-20260930/recovery/data/language_baseline/recovery_manifest.json ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "repo": "YNSScarSaiyan/nys-corpus",
3
+ "revision": "7eb0b0dd1f3122a4189d33ae6cc13c21d0a25774",
4
+ "files": {
5
+ "vocab_sparse.json": {
6
+ "bytes": 106816698,
7
+ "sha256": "23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e"
8
+ },
9
+ "train_corpus_sparse.bin": {
10
+ "bytes": 66816546,
11
+ "sha256": "a9b901c8c02b35eebe638abc0bc8ae551e359f51370890733ac917f0c8a97916"
12
+ },
13
+ "sft_corpus_sparse.bin": {
14
+ "bytes": 13200476,
15
+ "sha256": "91cc73e689d8e8e457221d3a20bb4d552cd9d2e5df1c1ed2f7ad2ce5c796cd60"
16
+ },
17
+ "distill_corpus_sparse.bin": {
18
+ "bytes": 15363524,
19
+ "sha256": "e1e27ee163ef7288fa6d68230fe167ef368e553ac8f19922019d80f87eb7e4e4"
20
+ }
21
+ }
22
+ }
lineages/corrected-20260930/recovery/data/language_baseline/vocab_sparse.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e
3
+ size 106816698
lineages/corrected-20260930/recovery/data/slots/compiler_physics.nysa ADDED
Binary file (48.8 kB). View file
 
lineages/corrected-20260930/recovery/data/slots/recovery_manifest.json ADDED
@@ -0,0 +1,108 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "verified": true,
3
+ "checks": {
4
+ "2_compiler": true,
5
+ "4_speech": true,
6
+ "hard": true,
7
+ "sft": true,
8
+ "distill": true,
9
+ "source_checkpoint_sha256_unchanged": true
10
+ },
11
+ "record_counts": {
12
+ "speech_align.nysv": 305,
13
+ "speech_slot.nysv": 6,
14
+ "compiler_physics.nysa": 412
15
+ },
16
+ "checkpoint": "gpu_eqprop_1915000_20260913_095122.bin",
17
+ "checkpoint_sha256": "406a691b1ac42112705c5193814f76ff064f922161d48ab0e3f67452403bbeec",
18
+ "published_revision": "6f1f1e314e6d680f0e1541ddc6b1f6590b46cf40",
19
+ "vocoder_sha256": "028d0e759f9721b2110d8527028e7de3d80d5b611401d33db895d48e2f2df4d1",
20
+ "original_slot_sha256": {
21
+ "compiler_physics.nysa": "052fc7dee4b1081392051b9fde2fac6533aa77e82dbea56ac32c27a67ced6cf8",
22
+ "speech_slot.nysv": "2fd8be22d9408bb5d88a11b1c27c6c0ac80f42d7808f93a8d1ed31734a5f5cee",
23
+ "speech_align.nysv": "274f4d2e283a6b1de14eb73d4ea4f6c2e10fc18aff8bd7f1d1aca29b5e1430c5"
24
+ },
25
+ "actual": {
26
+ "slots": {
27
+ "1_language": {
28
+ "mix": {
29
+ "acc": null,
30
+ "hits": 0,
31
+ "n": 0,
32
+ "in_pool": 0,
33
+ "in_pool_frac": null
34
+ },
35
+ "hard": {
36
+ "acc": 0.0,
37
+ "hits": 0,
38
+ "n": 32,
39
+ "in_pool": 4,
40
+ "in_pool_frac": 0.125
41
+ },
42
+ "sft": {
43
+ "acc": 0.0,
44
+ "hits": 0,
45
+ "n": 32,
46
+ "in_pool": 30,
47
+ "in_pool_frac": 0.9375
48
+ },
49
+ "distill": {
50
+ "acc": 0.0,
51
+ "hits": 0,
52
+ "n": 32,
53
+ "in_pool": 18,
54
+ "in_pool_frac": 0.5625
55
+ }
56
+ },
57
+ "2_compiler": {
58
+ "clamped_free": {
59
+ "acc": 0.6385690789473685,
60
+ "hits": 1553,
61
+ "n": 2432,
62
+ "in_pool": 2432,
63
+ "in_pool_frac": 1.0
64
+ },
65
+ "spin_free": {
66
+ "acc": 0.4004934210526316,
67
+ "hits": 974,
68
+ "n": 2432,
69
+ "in_pool": 2432,
70
+ "in_pool_frac": 1.0
71
+ },
72
+ "n_recs": 64,
73
+ "n_free": 38
74
+ },
75
+ "4_speech": {
76
+ "acc": 0.10825043885313049,
77
+ "hits": 185,
78
+ "n": 1709,
79
+ "in_pool": 700,
80
+ "in_pool_frac": 0.4095962551199532
81
+ }
82
+ },
83
+ "headline": {
84
+ "language": 0.0,
85
+ "compiler": 0.6385690789473685,
86
+ "speech": 0.10825043885313049
87
+ }
88
+ },
89
+ "limitation": "Original live mix is unavailable; its score and aggregate language headline cannot be reproduced.",
90
+ "staged_files": {
91
+ "compiler_physics.nysa": {
92
+ "bytes": 48815,
93
+ "sha256": "052fc7dee4b1081392051b9fde2fac6533aa77e82dbea56ac32c27a67ced6cf8"
94
+ },
95
+ "speech_vocoder.nyvc": {
96
+ "bytes": 20500,
97
+ "sha256": "028d0e759f9721b2110d8527028e7de3d80d5b611401d33db895d48e2f2df4d1"
98
+ },
99
+ "speech_align.nysv": {
100
+ "bytes": 1490468,
101
+ "sha256": "fea716002f75cde8a4567d64e28c14e353c3cdb349cd9ec8ecf3c991f52fbf8c"
102
+ },
103
+ "speech_slot.nysv": {
104
+ "bytes": 15284,
105
+ "sha256": "2988cc801e2b95d6ebba35d335ec656486bc0fe0b5938d071a536c63879aca0c"
106
+ }
107
+ }
108
+ }
lineages/corrected-20260930/recovery/data/slots/speech_align.nysv ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fea716002f75cde8a4567d64e28c14e353c3cdb349cd9ec8ecf3c991f52fbf8c
3
+ size 1490468
lineages/corrected-20260930/recovery/data/slots/speech_slot.nysv ADDED
Binary file (15.3 kB). View file
 
lineages/corrected-20260930/recovery/data/slots/speech_vocoder.nyvc ADDED
Binary file (20.5 kB). View file
 
lineages/corrected-20260930/recovery/gpu_eqprop/build_corrected_mix.py ADDED
@@ -0,0 +1,217 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Rebuild all nine original SFT sources using frozen NYS IDs. No training/upload.
2
+
3
+ Pinned schemas are normalized explicitly. Every source and every requested split
4
+ must contribute; failed or empty sources fail the build, never shrink the diet.
5
+ """
6
+ import argparse
7
+ from array import array
8
+ from collections import Counter
9
+ import hashlib
10
+ import json
11
+ import mmap
12
+ import os
13
+ from pathlib import Path
14
+ import random
15
+ import re
16
+ import struct
17
+ import subprocess
18
+
19
+ VOCAB_SHA = '23d96f815a5904045899fa304988fb011e21a6b25b612d729b8d833ae5e7388e'
20
+ SPECS = [
21
+ ('tulu3',4,['train'],None), ('openhermes',3,['train'],None),
22
+ ('ultrachat',3,['train_sft'],None), ('xlam',4,['train'],'dataset'),
23
+ ('toolace',3,['train'],None), ('glaive',3,['train'],None),
24
+ ('hermes_fc',2,['train'],'func_calling_singleturn'),
25
+ ('orca',4,['analytical_reasoning','code_','rag','follow_up','fs_cot_flow','open_domain_qa'],None),
26
+ ('agent_flan',4,['agent_instruct_tflan','agent_instruct_react','toolbench_tflan_60p_r10r5u7','toolbench_instruct_j1s1_3k'],None),
27
+ ]
28
+
29
+ def sha(path):
30
+ with Path(path).open('rb') as f:
31
+ return hashlib.file_digest(f, 'sha256').hexdigest()
32
+
33
+ def text(v):
34
+ if v is None: return ''
35
+ return v if isinstance(v,str) else json.dumps(v,ensure_ascii=False,separators=(',',':'))
36
+
37
+ def decode(v):
38
+ return json.loads(v) if isinstance(v,str) else v
39
+
40
+ class NoTargets(ValueError):
41
+ pass
42
+
43
+ def messages(row):
44
+ out=[]
45
+ conv=row.get('messages') or row.get('conversations') or row.get('conversation')
46
+ if conv:
47
+ conv=decode(conv)
48
+ for m in conv:
49
+ m=decode(m)
50
+ role=(m.get('role') or m.get('from') or '').lower()
51
+ role={'human':'user','gpt':'assistant','function':'tool','tool_response':'tool'}.get(role,role)
52
+ if role not in ('user','assistant','system','tool'):
53
+ raise ValueError(f'Unknown conversation role {role!r}')
54
+ content=text(m.get('content',m.get('value')))
55
+ if m.get('tool_calls'):
56
+ content += '\n'+text(m['tool_calls'])
57
+ if m.get('loss') is False:
58
+ role='tool' # context only; not an assistant target
59
+ if content.strip(): out.append(dict(role=role,content=content))
60
+ system=row.get('system') or row.get('system_prompt')
61
+ if system and not any(m['role']=='system' for m in out):
62
+ out.insert(0,dict(role='system',content=text(system)))
63
+ elif row.get('query') is not None and row.get('answers') is not None:
64
+ out=[dict(role='user',content=text(row['query'])+'\n tools '+text(row.get('tools'))),
65
+ dict(role='assistant',content=text(row['answers']))]
66
+ elif isinstance(row.get('chat'),str):
67
+ if row.get('system'): out.append(dict(role='system',content=text(row['system'])))
68
+ parts=re.split(r'(?:^|\n)(USER|ASSISTANT|FUNCTION RESPONSE):\s*', row['chat'])
69
+ if len(parts)<3 or parts[0].strip(): raise ValueError('Unsupported Glaive chat framing')
70
+ for role,body in zip(parts[1::2],parts[2::2]):
71
+ out.append(dict(role={'USER':'user','ASSISTANT':'assistant','FUNCTION RESPONSE':'tool'}[role],
72
+ content=body.replace('<|endoftext|>','').strip()))
73
+ else:
74
+ raise ValueError(f'Unsupported source schema: {sorted(row)}')
75
+ if not any(m['role']=='assistant' and m['content'].strip() for m in out):
76
+ raise NoTargets('No assistant targets in conversation')
77
+ return out
78
+
79
+ def offsets(path, expected):
80
+ off=array('Q')
81
+ with open(path,'rb') as f:
82
+ with mmap.mmap(f.fileno(),0,access=mmap.ACCESS_READ) as mm:
83
+ count,=struct.unpack_from('<I',mm,0)
84
+ if count!=expected: raise ValueError(f'Incomplete source {path}: {count} != {expected}')
85
+ pos=4
86
+ for _ in range(count):
87
+ off.append(pos)
88
+ n,=struct.unpack_from('<H',mm,pos)
89
+ if not 2<=n<=128: raise ValueError('Invalid sparse record length')
90
+ end=pos+2+4*n
91
+ if end>len(mm): raise ValueError('Truncated sparse record')
92
+ ids=struct.unpack_from('<'+'I'*n,mm,pos+2)
93
+ if max(ids)>=5_000_000: raise ValueError('Invalid sparse token ID')
94
+ pos=end
95
+ if pos!=len(mm): raise ValueError('Trailing corpus bytes')
96
+ return off
97
+
98
+ def build(args):
99
+ if args.token_file:
100
+ os.environ['HF_TOKEN']=args.token_file.read_text().strip()
101
+ from datasets import load_dataset
102
+ from huggingface_hub import hf_hub_download
103
+ def open_stream(name, info, config, sp):
104
+ if name=='agent_flan':
105
+ # Its published Arrow feature declaration omits the optional `type`
106
+ # field in some splits. Read the pinned JSONL rows without that cast.
107
+ p=hf_hub_download(info['repo'],'data/'+sp+'.jsonl',repo_type='dataset',revision=info['revision'])
108
+ def rows():
109
+ with open(p,encoding='utf-8') as f:
110
+ for line in f:
111
+ if line.strip(): yield json.loads(line)
112
+ return iter(rows())
113
+ return iter(load_dataset(info['repo'],name=config,split=sp,revision=info['revision'],streaming=True))
114
+ out=args.output
115
+ out.mkdir(parents=True,exist_ok=True)
116
+ if sha(args.vocab)!=VOCAB_SHA: raise ValueError('Wrong frozen vocabulary')
117
+ discovered=json.loads(args.discovery.read_text())
118
+ converter_sha=sha(args.converter)
119
+ reports=[]
120
+ used=0
121
+ for idx,(spec,info) in enumerate(zip(SPECS,discovered)):
122
+ name,weight,splits,config=spec
123
+ quota=args.records-used if idx==len(SPECS)-1 else args.records*weight//30
124
+ path=out/(name+'.bin')
125
+ report_path=out/(name+'.json')
126
+ if report_path.exists():
127
+ report=json.loads(report_path.read_text())
128
+ if report.get('converter_sha256')!=converter_sha or report['sha256']!=sha(path) or report.get('target_records',report['records'])!=quota or report['revision']!=info['revision']:
129
+ raise ValueError('Prior build identity differs')
130
+ offsets(path,report['records'])
131
+ reports.append(report)
132
+ used+=report['records']
133
+ continue
134
+ streams={sp:open_stream(name,info,config,sp) for sp in splits}
135
+ empty=Counter()
136
+ def next_messages(sp,it):
137
+ while True:
138
+ row=next(it)
139
+ try: return messages(row)
140
+ except NoTargets: empty[sp]+=1
141
+ # Check each schema before starting the converter; preserve the first row.
142
+ first={sp:next_messages(sp,it) for sp,it in streams.items()}
143
+ counts=Counter()
144
+ source_hash=hashlib.sha256()
145
+ with (out/(name+'.convert.log')).open('w') as log:
146
+ proc=subprocess.Popen([str(args.converter),'--vocab',str(args.vocab),'--output',str(path),
147
+ '--limit',str(quota),'--max-per-convo','25','--flush-every','1000'],
148
+ stdin=subprocess.PIPE,stderr=log,text=True)
149
+ try:
150
+ while proc.poll() is None and streams:
151
+ for sp,it in list(streams.items()):
152
+ try:
153
+ ms=first.pop(sp) if sp in first else next_messages(sp,it)
154
+ except StopIteration:
155
+ del streams[sp]
156
+ continue
157
+ line=json.dumps({'messages':ms},ensure_ascii=False)+'\n'
158
+ proc.stdin.write(line)
159
+ counts[sp]+=1
160
+ source_hash.update(line.encode())
161
+ except BrokenPipeError:
162
+ pass
163
+ except BaseException:
164
+ proc.kill(); proc.wait(); raise
165
+ finally:
166
+ try: proc.stdin.close()
167
+ except BrokenPipeError: pass
168
+ if proc.wait()!=0: raise RuntimeError(f'Converter failed: {name}')
169
+ actual=struct.unpack('<I',path.read_bytes()[:4])[0]
170
+ if actual<=0 or actual>quota: raise ValueError('Empty source or quota overrun')
171
+ offsets(path,actual)
172
+ if not all(counts[sp]>0 for sp in splits): raise ValueError('Missing split')
173
+ report=dict(name=name,repo=info['repo'],revision=info['revision'],config=config,
174
+ records=actual,target_records=quota,bytes=path.stat().st_size,sha256=sha(path),fed_conversations=dict(counts),
175
+ normalized_input_sha256=source_hash.hexdigest(),max_per_conversation=25,
176
+ skipped_no_assistant_targets=dict(empty),converter_sha256=converter_sha)
177
+ report_path.write_text(json.dumps(report,indent=2))
178
+ reports.append(report)
179
+ used+=actual
180
+ print(json.dumps(report),flush=True)
181
+ # Shuffle record order globally so the first live rotations include all sources.
182
+ handles=[open(out/(r['name']+'.bin'),'rb') for r in reports]
183
+ maps=[mmap.mmap(f.fileno(),0,access=mmap.ACCESS_READ) for f in handles]
184
+ record_offsets=[offsets(out/(r['name']+'.bin'),r['records']) for r in reports]
185
+ order=array('Q', ((i<<32)|j for i,r in enumerate(reports) for j in range(r['records'])))
186
+ random.Random(7).shuffle(order)
187
+ temp=out/'mix_sft_sparse.bin.writing'
188
+ first_counts=Counter()
189
+ with temp.open('wb') as f:
190
+ f.write(struct.pack('<I',len(order)))
191
+ for at,key in enumerate(order):
192
+ i,j=key>>32,key&0xffffffff
193
+ pos=record_offsets[i][j]
194
+ n,=struct.unpack_from('<H',maps[i],pos)
195
+ f.write(maps[i][pos:pos+2+4*n])
196
+ if at<32000: first_counts[reports[i]['name']]+=1
197
+ f.flush(); os.fsync(f.fileno())
198
+ offsets(temp,args.records)
199
+ final=out/'mix_sft_sparse.bin'
200
+ temp.replace(final)
201
+ for m,f in zip(maps,handles): m.close(); f.close()
202
+ manifest=dict(verified=True,records=args.records,bytes=final.stat().st_size,sha256=sha(final),
203
+ converter_sha256=converter_sha,
204
+ vocab_sha256=VOCAB_SHA,sources=reports,shuffle_seed=7,first_32000_source_counts=dict(first_counts),
205
+ note='Rebuilt original nine-source recipe; not the unavailable historical byte-for-byte mix')
206
+ (out/'mix_manifest.json').write_text(json.dumps(manifest,indent=2))
207
+ print(json.dumps(manifest),flush=True)
208
+
209
+ if __name__=='__main__':
210
+ ap=argparse.ArgumentParser(description=__doc__)
211
+ ap.add_argument('--discovery',type=Path,required=True)
212
+ ap.add_argument('--vocab',type=Path,required=True)
213
+ ap.add_argument('--converter',type=Path,required=True)
214
+ ap.add_argument('--output',type=Path,required=True)
215
+ ap.add_argument('--records',type=int,default=2_050_000)
216
+ ap.add_argument('--token-file',type=Path)
217
+ build(ap.parse_args())
lineages/corrected-20260930/recovery/gpu_eqprop/build_cuda.sh ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env bash
2
+ set -euo pipefail
3
+ cd -- "$(dirname -- "${BASH_SOURCE[0]}")"
4
+ NVCC="${NVCC:-/data/yue-cuda/cuda/toolkit/bin/nvcc}"
5
+ if [[ ! -x "$NVCC" ]]; then NVCC="$(command -v nvcc)"; fi
6
+ "$NVCC" -O3 -std=c++17 -gencode=arch=compute_89,code=sm_89 \
7
+ -o eqprop_gpu_cuda eqprop_gpu.cu