AINovice2005 commited on
Commit
5b806a5
Β·
verified Β·
1 Parent(s): a1f8216

Update index.html

Browse files
Files changed (1) hide show
  1. index.html +219 -17
index.html CHANGED
@@ -1,19 +1,221 @@
1
  <!doctype html>
2
  <html>
3
- <head>
4
- <meta charset="utf-8" />
5
- <meta name="viewport" content="width=device-width" />
6
- <title>My static Space</title>
7
- <link rel="stylesheet" href="style.css" />
8
- </head>
9
- <body>
10
- <div class="card">
11
- <h1>Welcome to your static Space!</h1>
12
- <p>You can modify this app directly by editing <i>index.html</i> in the Files and versions tab.</p>
13
- <p>
14
- Also don't forget to check the
15
- <a href="https://huggingface.co/docs/hub/spaces" target="_blank">Spaces documentation</a>.
16
- </p>
17
- </div>
18
- </body>
19
- </html>
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  <!doctype html>
2
  <html>
3
+ <head>
4
+ <meta charset="utf-8">
5
+ <title>schema.py Β· config.py Β· definitions.py</title>
6
+ <style>
7
+ :root{
8
+ --bg:#ffffff; --card:#ffffff; --ink:#1a1a1a; --sub:#6b6b6b; --line:#d8d8d8;
9
+ --coral-bg:#FAECE7; --coral-bd:#D85A30; --coral-tx:#712B13;
10
+ --amber-bg:#FAEEDA; --amber-bd:#BA7517; --amber-tx:#633806;
11
+ --blue-bg:#E6F1FB; --blue-bd:#378ADD; --blue-tx:#0C447C;
12
+ --teal-bg:#E1F5EE; --teal-bd:#1D9E75; --teal-tx:#085041;
13
+ --purple-bg:#EEEDFE;--purple-bd:#7F77DD;--purple-tx:#3C3489;
14
+ --green-bg:#EAF3DE; --green-bd:#639922; --green-tx:#27500A;
15
+ --pink-bg:#FBEAF0; --pink-bd:#D4537E; --pink-tx:#72243E;
16
+ --gray-bg:#F1EFE8; --gray-bd:#888780; --gray-tx:#444441;
17
+ }
18
+ @media (prefers-color-scheme: dark){
19
+ :root:not([data-theme="light"]){
20
+ --bg:#1c1c1a; --card:#26261f; --ink:#eeeee6; --sub:#a3a299; --line:#3c3c36;
21
+ --coral-bg:#4A1B0C; --coral-bd:#F0997B; --coral-tx:#F5C4B3;
22
+ --amber-bg:#412402; --amber-bd:#EF9F27; --amber-tx:#FAC775;
23
+ --blue-bg:#042C53; --blue-bd:#85B7EB; --blue-tx:#B5D4F4;
24
+ --teal-bg:#04342C; --teal-bd:#5DCAA5; --teal-tx:#9FE1CB;
25
+ --purple-bg:#26215C;--purple-bd:#AFA9EC;--purple-tx:#CECBF6;
26
+ --green-bg:#173404; --green-bd:#97C459; --green-tx:#C0DD97;
27
+ --pink-bg:#4B1528; --pink-bd:#ED93B1; --pink-tx:#F4C0D1;
28
+ --gray-bg:#2C2C2A; --gray-bd:#B4B2A9; --gray-tx:#D3D1C7;
29
+ }
30
+ }
31
+ :root[data-theme="dark"]{
32
+ --bg:#1c1c1a; --card:#26261f; --ink:#eeeee6; --sub:#a3a299; --line:#3c3c36;
33
+ --coral-bg:#4A1B0C; --coral-bd:#F0997B; --coral-tx:#F5C4B3;
34
+ --amber-bg:#412402; --amber-bd:#EF9F27; --amber-tx:#FAC775;
35
+ --blue-bg:#042C53; --blue-bd:#85B7EB; --blue-tx:#B5D4F4;
36
+ --teal-bg:#04342C; --teal-bd:#5DCAA5; --teal-tx:#9FE1CB;
37
+ --purple-bg:#26215C;--purple-bd:#AFA9EC;--purple-tx:#CECBF6;
38
+ --green-bg:#173404; --green-bd:#97C459; --green-tx:#C0DD97;
39
+ --pink-bg:#4B1528; --pink-bd:#ED93B1; --pink-tx:#F4C0D1;
40
+ --gray-bg:#2C2C2A; --gray-bd:#B4B2A9; --gray-tx:#D3D1C7;
41
+ }
42
+ *{box-sizing:border-box;}
43
+ body{margin:0;background:var(--bg);color:var(--ink);font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",Helvetica,Arial,sans-serif;}
44
+ .wrap{max-width:780px;margin:0 auto;padding:28px 20px 40px;}
45
+ h1{font-size:17px;font-weight:600;margin:0 0 2px;}
46
+ .subtitle{font-size:13px;color:var(--sub);margin:0 0 22px;}
47
+ .hint{font-size:12px;color:var(--sub);text-align:center;margin:0 0 18px;}
48
+
49
+ .flow{display:flex;flex-direction:column;align-items:center;gap:0;}
50
+ .node{
51
+ border-radius:10px;border:1.5px solid var(--bd);
52
+ background:var(--fill);padding:10px 16px;cursor:pointer;transition:transform .12s ease, box-shadow .12s ease;
53
+ text-align:center;
54
+ }
55
+ .node:hover, .node.active{transform:translateY(-1px);box-shadow:0 2px 10px rgba(0,0,0,.08);}
56
+ .node .t{font-size:13.5px;font-weight:600;color:var(--tx);}
57
+ .node .s{font-size:11.5px;color:var(--tx);opacity:.75;margin-top:1px;}
58
+ .arrow{width:1.5px;height:22px;background:var(--line);position:relative;}
59
+ .arrow::after{
60
+ content:"";position:absolute;left:50%;bottom:-1px;transform:translateX(-50%);
61
+ width:0;height:0;border-left:5px solid transparent;border-right:5px solid transparent;
62
+ border-top:6px solid var(--line);
63
+ }
64
+ .arrow-lbl{font-size:10.5px;color:var(--sub);margin:2px 0;}
65
+
66
+ .row{display:flex;gap:18px;justify-content:center;align-items:flex-start;flex-wrap:wrap;width:100%;}
67
+ .row .node{width:100%;max-width:260px;flex:1 1 240px;}
68
+ .hjoin{position:relative;width:1.5px;background:var(--line);align-self:center;}
69
+
70
+ .full .node{width:100%;max-width:560px;}
71
+
72
+ .pool{
73
+ width:100%;max-width:600px;border:1.5px dashed var(--line);border-radius:12px;
74
+ padding:14px;margin:2px 0;
75
+ }
76
+ .pool-label{font-size:11px;color:var(--sub);text-align:center;margin-bottom:10px;}
77
+ .pool-row{display:flex;gap:10px;justify-content:center;flex-wrap:wrap;}
78
+ .pool-row .node{max-width:180px;flex:1 1 160px;padding:9px 10px;}
79
+ .pool-row .node .t{font-size:12.5px;}
80
+ .pool-row .node .s{font-size:10.5px;}
81
+
82
+ .detail{
83
+ margin-top:22px;border:1px solid var(--line);border-radius:10px;background:var(--card);
84
+ padding:14px 16px;font-size:13px;line-height:1.55;color:var(--ink);min-height:64px;
85
+ }
86
+ .detail .k{font-size:11px;color:var(--sub);text-transform:uppercase;letter-spacing:.04em;margin-bottom:4px;}
87
+
88
+ .legend{display:flex;flex-wrap:wrap;gap:8px 14px;justify-content:center;margin-top:20px;}
89
+ .legend span{display:inline-flex;align-items:center;gap:6px;font-size:11.5px;color:var(--sub);}
90
+ .legend i{width:10px;height:10px;border-radius:3px;display:inline-block;border:1.2px solid var(--bd2);background:var(--bg2);}
91
+ </style>
92
+ </head>
93
+ <body>
94
+ <div class="wrap">
95
+ <h1>schema.py Β· config.py Β· definitions.py</h1>
96
+ <p class="subtitle">The pipeline's contract, configuration, and wiring layer β€” not a data flow, a dependency graph</p>
97
+ <p class="hint">Click any block for details</p>
98
+
99
+ <div class="flow">
100
+ <div class="row">
101
+ <div class="node" style="--fill:var(--coral-bg);--bd:var(--coral-bd);--tx:var(--coral-tx)" data-key="schema">
102
+ <div class="t">schema.py</div>
103
+ <div class="s">Data contracts β€” no Dagster, no compute</div>
104
+ </div>
105
+ <div class="node" style="--fill:var(--amber-bg);--bd:var(--amber-bd);--tx:var(--amber-tx)" data-key="config">
106
+ <div class="t">config.py</div>
107
+ <div class="s">CarbonPipelineConfig, StorageConfig</div>
108
+ </div>
109
+ </div>
110
+ <div class="arrow-lbl">config.py imports DEFAULT_VALIDATION_LEVEL from schema.py β†’</div>
111
+ <div class="arrow"></div>
112
+
113
+ <div class="full">
114
+ <div class="node" style="--fill:var(--blue-bg);--bd:var(--blue-bd);--tx:var(--blue-tx)" data-key="assets">
115
+ <div class="t">assets/cpu/*.py + assets/gpu/*.py</div>
116
+ <div class="s">each asset takes CarbonPipelineConfig and reads schema.py constants directly</div>
117
+ </div>
118
+ </div>
119
+ <div class="arrow"></div>
120
+
121
+ <div class="row">
122
+ <div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="carbon_resource">
123
+ <div class="t">resources/carbon.py</div>
124
+ <div class="s">CarbonModelResource β€” model + tokenizer</div>
125
+ </div>
126
+ <div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="hf_resource">
127
+ <div class="t">resources/hf_client.py</div>
128
+ <div class="s">create_huggingface_resource()</div>
129
+ </div>
130
+ </div>
131
+ <div class="arrow"></div>
132
+
133
+ <div class="pool">
134
+ <div class="pool-label">definitions.py β€” imports assets + resources, wires everything together</div>
135
+ <div class="pool-row">
136
+ <div class="node" style="--fill:var(--teal-bg);--bd:var(--teal-bd);--tx:var(--teal-tx)" data-key="groupings">
137
+ <div class="t">Asset groupings</div>
138
+ <div class="s">CPU_ASSETS, GPU_PIPELINE_ASSETS</div>
139
+ </div>
140
+ <div class="node" style="--fill:var(--purple-bg);--bd:var(--purple-bd);--tx:var(--purple-tx)" data-key="jobs">
141
+ <div class="t">Job definitions</div>
142
+ <div class="s">4 jobs, each an AssetSelection</div>
143
+ </div>
144
+ </div>
145
+ </div>
146
+ <div class="arrow"></div>
147
+
148
+ <div class="full">
149
+ <div class="node" style="--fill:var(--pink-bg);--bd:var(--pink-bd);--tx:var(--pink-tx)" data-key="defs">
150
+ <div class="t">defs = dg.Definitions(assets, jobs, resources)</div>
151
+ <div class="s">the single object Dagster actually loads</div>
152
+ </div>
153
+ </div>
154
+ <div class="arrow-lbl">used by downstream analysis &amp; catalog sync β€” not registered in dg.Definitions() ↓</div>
155
+ <div class="arrow"></div>
156
+
157
+ <div class="pool">
158
+ <div class="pool-label">resources/*.py β€” analytical &amp; catalog access layer (standalone)</div>
159
+ <div class="pool-row">
160
+ <div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="facebergs">
161
+ <div class="t">faceberg.py</div>
162
+ <div class="s">PipelineCatalog, PIPELINE_TABLES lineage registry</div>
163
+ </div>
164
+ <div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="duckdb">
165
+ <div class="t">duckdb.py</div>
166
+ <div class="s">DuckDB + Iceberg query surface</div>
167
+ </div>
168
+ <div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="clickhouse">
169
+ <div class="t">clickhouse.py</div>
170
+ <div class="s">ClickHouseResource β€” local analytical SQL</div>
171
+ </div>
172
+ <div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="lancedb">
173
+ <div class="t">lancedb.py</div>
174
+ <div class="s">LanceDBResource β€” vector storage/search</div>
175
+ </div>
176
+ </div>
177
+ </div>
178
+ </div>
179
+
180
+ <div class="detail" id="detail">
181
+ <div class="k">Detail</div>
182
+ <div id="detail-body">Click any block above to see how it works.</div>
183
+ </div>
184
+
185
+ <div class="legend">
186
+ <span><i style="--bd2:var(--coral-bd);--bg2:var(--coral-bg)"></i>Contracts</span>
187
+ <span><i style="--bd2:var(--amber-bd);--bg2:var(--amber-bg)"></i>Runtime config</span>
188
+ <span><i style="--bd2:var(--blue-bd);--bg2:var(--blue-bg)"></i>Asset modules</span>
189
+ <span><i style="--bd2:var(--gray-bd);--bg2:var(--gray-bg)"></i>Resource modules</span>
190
+ <span><i style="--bd2:var(--teal-bd);--bg2:var(--teal-bg)"></i>Asset groupings</span>
191
+ <span><i style="--bd2:var(--purple-bd);--bg2:var(--purple-bg)"></i>Jobs</span>
192
+ <span><i style="--bd2:var(--pink-bd);--bg2:var(--pink-bg)"></i>Definitions object</span>
193
+ </div>
194
+ </div>
195
+
196
+ <script>
197
+ const details = {
198
+ schema: "Deliberately independent of Dagster and of compute β€” its own docstring states this. Covers the raw HF dataset identity (HuggingFaceBio/carbon-pretraining-corpus, config eukaryote_generator, split train), the validation-tier row table, the 14-column raw schema (EXPECTED_COLUMNS), the canonical biological composite key (record_id, start, end) β€” unique only after deduplication, and used directly as the GPU join key with no surrogate row_key β€” the IUPAC nucleotide alphabet, and the GPU output column contracts (TOKENIZED_CORPUS_COLUMNS, EMBEDDING_COLUMNS, LIKELIHOOD_COLUMNS). It also defines token_mask semantics (-2 padding, -1 BPE/text, 0 DNA special, 1..k k-mer contribution) as a shared contract precisely so no downstream stage hardcodes those values independently. It deliberately does NOT define gene_type's full vocabulary, since the dataset exposes multiple classes and the module's own comment says it shouldn't invent a complete vocabulary from a partial sample.",
199
+ config: "One dg.Config class, CarbonPipelineConfig, holds every runtime parameter β€” dataset selection, batch_size (1,000), rows_per_shard (250,000), compression (zstd), five separate output_dirs, pilot_sample_fraction/seed, cpu_workers, and gpu_token_budget (500,000,000). The docstring explains why it's one class: parameters split across several dg.Config classes get reinterpreted by Dagster as asset inputs rather than configuration. Each output_dir is kept independent from the others by design β€” e.g. tokenized_output_dir is a sibling of output_dir, not a subdirectory keyed off it, so a schema change in tokenization never forces touching or re-running the CPU-enriched Parquet shards, and vice versa.",
200
+ assets: "Every CPU and GPU asset function takes a CarbonPipelineConfig parameter and reads schema.py constants directly (EXPECTED_COLUMNS, VALIDATION_LEVEL_ROWS, TOKEN_MASK_* etc.) β€” this is where the two foundation modules actually get used; neither schema.py nor config.py is imported by definitions.py itself.",
201
+ carbon_resource: "CarbonModelResource β€” the single project-level source of the Carbon-3B model and its hybrid 6-mer tokenizer, shared by tokenize_and_tag (tokenizer only) and the GPU enrichment assets (tokenizer + model), so both are resolved exactly once per process at one pinned revision instead of each asset calling from_pretrained() independently and risking version skew. Both are lazily loaded via properties β€” a resource that only needs model_checkpoint for a provenance manifest never triggers a 3B-parameter weight load. The tokenizer loads with trust_remote_code=True (required for the custom 6-mer tokenizer); the model loads in bfloat16 β€” its native training precision, not a throughput tradeoff β€” with FlashAttention-2 wired through the HF Kernels Hub rather than pip flash-attn, then put in .eval() mode. It records model-lifecycle telemetry only (load times, device/dtype, param count, resolved attention implementation and FA2 kernel revision); inference telemetry like throughput, OOM retries, and peak CUDA memory belongs to embeddings.py, not this resource. compile_for_buckets() is an explicit method the GPU asset calls at startup with the bucket/batch lookup table β€” the resource does not torch.compile eagerly on load, since bucket-stable static shapes are asset-level configuration.",
202
+ hf_resource: "A thin, deliberately non-duplicating wrapper: the module's own docstring says it intentionally does not implement a second dataset client or reimplement load_dataset(). It just configures a cache directory (.hf_cache by default) and returns a project-configured dagster_hf_datasets.HuggingFaceResource, consumed by assets/cpu/ingest.py through Dagster's dependency injection.",
203
+ groupings: "CPU_ASSETS = [carbon_cpu_enriched_sequences]. GPU_PIPELINE_ASSETS = [carbon_pilot_corpus, carbon_tokenized_corpus, carbon_gpu_enrichment]. Both lists exist purely so dg.Definitions(assets=[*CPU_ASSETS, *GPU_PIPELINE_ASSETS]) reads as an explicit statement of pipeline membership rather than an unlabeled list.",
204
+ jobs: "Four jobs, each defined with dg.in_process_executor rather than Dagster's default multiprocess executor: carbon_cpu_job (CPU stage alone), carbon_tokenize_job (tokenization alone), carbon_inference_job (embeddings + likelihood from an existing tokenized corpus), and carbon_gpu_job (pilot sampling through inference, end to end). in_process_executor is used everywhere specifically because several assets already run their own internal ProcessPoolExecutor β€” nesting Dagster's multiprocess executor around that would risk IPC pipe deadlocks between Dagster's step workers and the asset's own worker pool.",
205
+ defs: "The dg.Definitions object Dagster's CLI and UI actually discover and load β€” resources={\"hf_resource\": create_huggingface_resource(), \"carbon\": CarbonModelResource()} is where the two resource modules above are actually instantiated and bound by name. Two environment variables are set at the very top of this file, before dagster is even imported: TOKENIZERS_PARALLELISM=false, preventing a deadlock when the HF tokenizer's Rust threadpool is cloned across a fork, and PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True, guarding against CUDA virtual-memory fragmentation once VRAM usage climbs past roughly 70GB. Worth flagging: the module's own docstring diagram shows a carbon_likelihood_summary asset downstream of carbon_likelihood_stats, but no such asset is imported or defined anywhere in this file β€” a documented stage that isn't implemented yet, not a data-flow error.",
206
+ facebergs: "Maps existing HF datasets to Iceberg table metadata without copying data β€” the read path for catalog-managed lineage across five nodes: cpu_enriched, sampled_cpu, tokenized, likelihood_stats, embeddings, each declared in a PIPELINE_TABLES registry with its HF repo, upstream node_id, and access_mode. The raw pretraining_corpus stays a streaming-only node, deliberately never cataloged. Worth knowing: 'sampled_cpu' maps to the published HF artifact carbon-pilot-corpus-dedup, but the module's own docstring is explicit that the lineage edge it represents is deterministic per-row-hash sampling with stratum representativeness validation β€” not deduplication; the artifact name is legacy naming, not a description of what this node does. The module also monkey-patches three PyIceberg/Faceberg internals (BinaryEncoder.write_utf8, conversions.to_bytes for StringType, and iceberg.write_manifest) before any catalog operation runs, and writes a separate lineage.yml manifest that additionally records the streaming-only root node β€” data the underlying _LocalCatalog itself never reads or rewrites. Note: in the current repository this file is actually named facebergs.py (plural) β€” shown here as faceberg.py to match the import path duckdb.py actually uses.",
207
+ duckdb: "Provides the DuckDB+Iceberg query surface used to read catalog tables, and to recompute sampling.py's stratification logic in SQL for independent validation β€” length_bucket_expression() is written to be the exact SQL equivalent of the proxy buckets sampling.py computes in Python. get_connection() installs and loads DuckDB's Iceberg extension and registers a Hugging Face credential-chain secret so catalog tables backed by HF-hosted Parquet resolve without a manually passed token. This module imports PIPELINE_TABLES from carbon_enrichment.resources.faceberg (singular) β€” that import only resolves if the neighboring module is named faceberg.py, as shown here; in the current repository it's actually saved as facebergs.py (plural), which would make this import fail as written until one of the two names is fixed.",
208
+ clickhouse: "A thin wrapper around the clickhouse local CLI binary, not a persistent server connection β€” every query shells out to a subprocess. It can register local Parquet globs, or register a remote Hugging Face Parquet dataset as a ClickHouse url() source without downloading it: wildcard HF URLs are resolved once via the HF dataset tree API into concrete shard URLs, then compressed into ClickHouse's brace-expansion syntax. query_arrow() fully buffers a result as one Arrow table β€” fine for aggregates β€” while stream_arrow_batches() pipes ClickHouse's ArrowStream output straight into PyArrow's streaming reader so peak Python-side memory is bounded by one batch rather than the full result; the module's own docstring says to prefer the streaming path for any row-level result. This is the resource behind the Phase 4.5 case-study script that joins the CPU, likelihood, and embeddings sources against a common cohort key set.",
209
+ lancedb: "A thin wrapper around a local LanceDB connection, defaulting to a carbon_embeddings table with cosine distance. Every write and search call goes through the same exponential-backoff retry helper (5 attempts by default, 2.0s base delay doubling each attempt), specifically to absorb OS-level file-lock contention from concurrent local writes β€” something LanceDB's local file-backed format is more exposed to than a networked vector store would be. facet() hands the underlying Lance dataset to DuckDB directly via to_lance() to run a GROUP BY/COUNT without leaving Arrow format or duplicating the data β€” the same DuckDB-over-Arrow pattern duckdb.py uses for catalog tables, applied here to a vector table instead."
210
+ };
211
+
212
+ document.querySelectorAll('[data-key]').forEach(el => {
213
+ el.addEventListener('click', () => {
214
+ document.querySelectorAll('.node.active').forEach(n => n.classList.remove('active'));
215
+ el.classList.add('active');
216
+ document.getElementById('detail-body').textContent = details[el.dataset.key];
217
+ });
218
+ });
219
+ </script>
220
+ </body>
221
+ </html>