Update index.html
Browse files- index.html +219 -17
index.html
CHANGED
|
@@ -1,19 +1,221 @@
|
|
| 1 |
<!doctype html>
|
| 2 |
<html>
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
|
| 13 |
-
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
|
| 17 |
-
|
| 18 |
-
|
| 19 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
<!doctype html>
|
| 2 |
<html>
|
| 3 |
+
<head>
|
| 4 |
+
<meta charset="utf-8">
|
| 5 |
+
<title>schema.py Β· config.py Β· definitions.py</title>
|
| 6 |
+
<style>
|
| 7 |
+
:root{
|
| 8 |
+
--bg:#ffffff; --card:#ffffff; --ink:#1a1a1a; --sub:#6b6b6b; --line:#d8d8d8;
|
| 9 |
+
--coral-bg:#FAECE7; --coral-bd:#D85A30; --coral-tx:#712B13;
|
| 10 |
+
--amber-bg:#FAEEDA; --amber-bd:#BA7517; --amber-tx:#633806;
|
| 11 |
+
--blue-bg:#E6F1FB; --blue-bd:#378ADD; --blue-tx:#0C447C;
|
| 12 |
+
--teal-bg:#E1F5EE; --teal-bd:#1D9E75; --teal-tx:#085041;
|
| 13 |
+
--purple-bg:#EEEDFE;--purple-bd:#7F77DD;--purple-tx:#3C3489;
|
| 14 |
+
--green-bg:#EAF3DE; --green-bd:#639922; --green-tx:#27500A;
|
| 15 |
+
--pink-bg:#FBEAF0; --pink-bd:#D4537E; --pink-tx:#72243E;
|
| 16 |
+
--gray-bg:#F1EFE8; --gray-bd:#888780; --gray-tx:#444441;
|
| 17 |
+
}
|
| 18 |
+
@media (prefers-color-scheme: dark){
|
| 19 |
+
:root:not([data-theme="light"]){
|
| 20 |
+
--bg:#1c1c1a; --card:#26261f; --ink:#eeeee6; --sub:#a3a299; --line:#3c3c36;
|
| 21 |
+
--coral-bg:#4A1B0C; --coral-bd:#F0997B; --coral-tx:#F5C4B3;
|
| 22 |
+
--amber-bg:#412402; --amber-bd:#EF9F27; --amber-tx:#FAC775;
|
| 23 |
+
--blue-bg:#042C53; --blue-bd:#85B7EB; --blue-tx:#B5D4F4;
|
| 24 |
+
--teal-bg:#04342C; --teal-bd:#5DCAA5; --teal-tx:#9FE1CB;
|
| 25 |
+
--purple-bg:#26215C;--purple-bd:#AFA9EC;--purple-tx:#CECBF6;
|
| 26 |
+
--green-bg:#173404; --green-bd:#97C459; --green-tx:#C0DD97;
|
| 27 |
+
--pink-bg:#4B1528; --pink-bd:#ED93B1; --pink-tx:#F4C0D1;
|
| 28 |
+
--gray-bg:#2C2C2A; --gray-bd:#B4B2A9; --gray-tx:#D3D1C7;
|
| 29 |
+
}
|
| 30 |
+
}
|
| 31 |
+
:root[data-theme="dark"]{
|
| 32 |
+
--bg:#1c1c1a; --card:#26261f; --ink:#eeeee6; --sub:#a3a299; --line:#3c3c36;
|
| 33 |
+
--coral-bg:#4A1B0C; --coral-bd:#F0997B; --coral-tx:#F5C4B3;
|
| 34 |
+
--amber-bg:#412402; --amber-bd:#EF9F27; --amber-tx:#FAC775;
|
| 35 |
+
--blue-bg:#042C53; --blue-bd:#85B7EB; --blue-tx:#B5D4F4;
|
| 36 |
+
--teal-bg:#04342C; --teal-bd:#5DCAA5; --teal-tx:#9FE1CB;
|
| 37 |
+
--purple-bg:#26215C;--purple-bd:#AFA9EC;--purple-tx:#CECBF6;
|
| 38 |
+
--green-bg:#173404; --green-bd:#97C459; --green-tx:#C0DD97;
|
| 39 |
+
--pink-bg:#4B1528; --pink-bd:#ED93B1; --pink-tx:#F4C0D1;
|
| 40 |
+
--gray-bg:#2C2C2A; --gray-bd:#B4B2A9; --gray-tx:#D3D1C7;
|
| 41 |
+
}
|
| 42 |
+
*{box-sizing:border-box;}
|
| 43 |
+
body{margin:0;background:var(--bg);color:var(--ink);font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",Helvetica,Arial,sans-serif;}
|
| 44 |
+
.wrap{max-width:780px;margin:0 auto;padding:28px 20px 40px;}
|
| 45 |
+
h1{font-size:17px;font-weight:600;margin:0 0 2px;}
|
| 46 |
+
.subtitle{font-size:13px;color:var(--sub);margin:0 0 22px;}
|
| 47 |
+
.hint{font-size:12px;color:var(--sub);text-align:center;margin:0 0 18px;}
|
| 48 |
+
|
| 49 |
+
.flow{display:flex;flex-direction:column;align-items:center;gap:0;}
|
| 50 |
+
.node{
|
| 51 |
+
border-radius:10px;border:1.5px solid var(--bd);
|
| 52 |
+
background:var(--fill);padding:10px 16px;cursor:pointer;transition:transform .12s ease, box-shadow .12s ease;
|
| 53 |
+
text-align:center;
|
| 54 |
+
}
|
| 55 |
+
.node:hover, .node.active{transform:translateY(-1px);box-shadow:0 2px 10px rgba(0,0,0,.08);}
|
| 56 |
+
.node .t{font-size:13.5px;font-weight:600;color:var(--tx);}
|
| 57 |
+
.node .s{font-size:11.5px;color:var(--tx);opacity:.75;margin-top:1px;}
|
| 58 |
+
.arrow{width:1.5px;height:22px;background:var(--line);position:relative;}
|
| 59 |
+
.arrow::after{
|
| 60 |
+
content:"";position:absolute;left:50%;bottom:-1px;transform:translateX(-50%);
|
| 61 |
+
width:0;height:0;border-left:5px solid transparent;border-right:5px solid transparent;
|
| 62 |
+
border-top:6px solid var(--line);
|
| 63 |
+
}
|
| 64 |
+
.arrow-lbl{font-size:10.5px;color:var(--sub);margin:2px 0;}
|
| 65 |
+
|
| 66 |
+
.row{display:flex;gap:18px;justify-content:center;align-items:flex-start;flex-wrap:wrap;width:100%;}
|
| 67 |
+
.row .node{width:100%;max-width:260px;flex:1 1 240px;}
|
| 68 |
+
.hjoin{position:relative;width:1.5px;background:var(--line);align-self:center;}
|
| 69 |
+
|
| 70 |
+
.full .node{width:100%;max-width:560px;}
|
| 71 |
+
|
| 72 |
+
.pool{
|
| 73 |
+
width:100%;max-width:600px;border:1.5px dashed var(--line);border-radius:12px;
|
| 74 |
+
padding:14px;margin:2px 0;
|
| 75 |
+
}
|
| 76 |
+
.pool-label{font-size:11px;color:var(--sub);text-align:center;margin-bottom:10px;}
|
| 77 |
+
.pool-row{display:flex;gap:10px;justify-content:center;flex-wrap:wrap;}
|
| 78 |
+
.pool-row .node{max-width:180px;flex:1 1 160px;padding:9px 10px;}
|
| 79 |
+
.pool-row .node .t{font-size:12.5px;}
|
| 80 |
+
.pool-row .node .s{font-size:10.5px;}
|
| 81 |
+
|
| 82 |
+
.detail{
|
| 83 |
+
margin-top:22px;border:1px solid var(--line);border-radius:10px;background:var(--card);
|
| 84 |
+
padding:14px 16px;font-size:13px;line-height:1.55;color:var(--ink);min-height:64px;
|
| 85 |
+
}
|
| 86 |
+
.detail .k{font-size:11px;color:var(--sub);text-transform:uppercase;letter-spacing:.04em;margin-bottom:4px;}
|
| 87 |
+
|
| 88 |
+
.legend{display:flex;flex-wrap:wrap;gap:8px 14px;justify-content:center;margin-top:20px;}
|
| 89 |
+
.legend span{display:inline-flex;align-items:center;gap:6px;font-size:11.5px;color:var(--sub);}
|
| 90 |
+
.legend i{width:10px;height:10px;border-radius:3px;display:inline-block;border:1.2px solid var(--bd2);background:var(--bg2);}
|
| 91 |
+
</style>
|
| 92 |
+
</head>
|
| 93 |
+
<body>
|
| 94 |
+
<div class="wrap">
|
| 95 |
+
<h1>schema.py Β· config.py Β· definitions.py</h1>
|
| 96 |
+
<p class="subtitle">The pipeline's contract, configuration, and wiring layer β not a data flow, a dependency graph</p>
|
| 97 |
+
<p class="hint">Click any block for details</p>
|
| 98 |
+
|
| 99 |
+
<div class="flow">
|
| 100 |
+
<div class="row">
|
| 101 |
+
<div class="node" style="--fill:var(--coral-bg);--bd:var(--coral-bd);--tx:var(--coral-tx)" data-key="schema">
|
| 102 |
+
<div class="t">schema.py</div>
|
| 103 |
+
<div class="s">Data contracts β no Dagster, no compute</div>
|
| 104 |
+
</div>
|
| 105 |
+
<div class="node" style="--fill:var(--amber-bg);--bd:var(--amber-bd);--tx:var(--amber-tx)" data-key="config">
|
| 106 |
+
<div class="t">config.py</div>
|
| 107 |
+
<div class="s">CarbonPipelineConfig, StorageConfig</div>
|
| 108 |
+
</div>
|
| 109 |
+
</div>
|
| 110 |
+
<div class="arrow-lbl">config.py imports DEFAULT_VALIDATION_LEVEL from schema.py β</div>
|
| 111 |
+
<div class="arrow"></div>
|
| 112 |
+
|
| 113 |
+
<div class="full">
|
| 114 |
+
<div class="node" style="--fill:var(--blue-bg);--bd:var(--blue-bd);--tx:var(--blue-tx)" data-key="assets">
|
| 115 |
+
<div class="t">assets/cpu/*.py + assets/gpu/*.py</div>
|
| 116 |
+
<div class="s">each asset takes CarbonPipelineConfig and reads schema.py constants directly</div>
|
| 117 |
+
</div>
|
| 118 |
+
</div>
|
| 119 |
+
<div class="arrow"></div>
|
| 120 |
+
|
| 121 |
+
<div class="row">
|
| 122 |
+
<div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="carbon_resource">
|
| 123 |
+
<div class="t">resources/carbon.py</div>
|
| 124 |
+
<div class="s">CarbonModelResource β model + tokenizer</div>
|
| 125 |
+
</div>
|
| 126 |
+
<div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="hf_resource">
|
| 127 |
+
<div class="t">resources/hf_client.py</div>
|
| 128 |
+
<div class="s">create_huggingface_resource()</div>
|
| 129 |
+
</div>
|
| 130 |
+
</div>
|
| 131 |
+
<div class="arrow"></div>
|
| 132 |
+
|
| 133 |
+
<div class="pool">
|
| 134 |
+
<div class="pool-label">definitions.py β imports assets + resources, wires everything together</div>
|
| 135 |
+
<div class="pool-row">
|
| 136 |
+
<div class="node" style="--fill:var(--teal-bg);--bd:var(--teal-bd);--tx:var(--teal-tx)" data-key="groupings">
|
| 137 |
+
<div class="t">Asset groupings</div>
|
| 138 |
+
<div class="s">CPU_ASSETS, GPU_PIPELINE_ASSETS</div>
|
| 139 |
+
</div>
|
| 140 |
+
<div class="node" style="--fill:var(--purple-bg);--bd:var(--purple-bd);--tx:var(--purple-tx)" data-key="jobs">
|
| 141 |
+
<div class="t">Job definitions</div>
|
| 142 |
+
<div class="s">4 jobs, each an AssetSelection</div>
|
| 143 |
+
</div>
|
| 144 |
+
</div>
|
| 145 |
+
</div>
|
| 146 |
+
<div class="arrow"></div>
|
| 147 |
+
|
| 148 |
+
<div class="full">
|
| 149 |
+
<div class="node" style="--fill:var(--pink-bg);--bd:var(--pink-bd);--tx:var(--pink-tx)" data-key="defs">
|
| 150 |
+
<div class="t">defs = dg.Definitions(assets, jobs, resources)</div>
|
| 151 |
+
<div class="s">the single object Dagster actually loads</div>
|
| 152 |
+
</div>
|
| 153 |
+
</div>
|
| 154 |
+
<div class="arrow-lbl">used by downstream analysis & catalog sync β not registered in dg.Definitions() β</div>
|
| 155 |
+
<div class="arrow"></div>
|
| 156 |
+
|
| 157 |
+
<div class="pool">
|
| 158 |
+
<div class="pool-label">resources/*.py β analytical & catalog access layer (standalone)</div>
|
| 159 |
+
<div class="pool-row">
|
| 160 |
+
<div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="facebergs">
|
| 161 |
+
<div class="t">faceberg.py</div>
|
| 162 |
+
<div class="s">PipelineCatalog, PIPELINE_TABLES lineage registry</div>
|
| 163 |
+
</div>
|
| 164 |
+
<div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="duckdb">
|
| 165 |
+
<div class="t">duckdb.py</div>
|
| 166 |
+
<div class="s">DuckDB + Iceberg query surface</div>
|
| 167 |
+
</div>
|
| 168 |
+
<div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="clickhouse">
|
| 169 |
+
<div class="t">clickhouse.py</div>
|
| 170 |
+
<div class="s">ClickHouseResource β local analytical SQL</div>
|
| 171 |
+
</div>
|
| 172 |
+
<div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="lancedb">
|
| 173 |
+
<div class="t">lancedb.py</div>
|
| 174 |
+
<div class="s">LanceDBResource β vector storage/search</div>
|
| 175 |
+
</div>
|
| 176 |
+
</div>
|
| 177 |
+
</div>
|
| 178 |
+
</div>
|
| 179 |
+
|
| 180 |
+
<div class="detail" id="detail">
|
| 181 |
+
<div class="k">Detail</div>
|
| 182 |
+
<div id="detail-body">Click any block above to see how it works.</div>
|
| 183 |
+
</div>
|
| 184 |
+
|
| 185 |
+
<div class="legend">
|
| 186 |
+
<span><i style="--bd2:var(--coral-bd);--bg2:var(--coral-bg)"></i>Contracts</span>
|
| 187 |
+
<span><i style="--bd2:var(--amber-bd);--bg2:var(--amber-bg)"></i>Runtime config</span>
|
| 188 |
+
<span><i style="--bd2:var(--blue-bd);--bg2:var(--blue-bg)"></i>Asset modules</span>
|
| 189 |
+
<span><i style="--bd2:var(--gray-bd);--bg2:var(--gray-bg)"></i>Resource modules</span>
|
| 190 |
+
<span><i style="--bd2:var(--teal-bd);--bg2:var(--teal-bg)"></i>Asset groupings</span>
|
| 191 |
+
<span><i style="--bd2:var(--purple-bd);--bg2:var(--purple-bg)"></i>Jobs</span>
|
| 192 |
+
<span><i style="--bd2:var(--pink-bd);--bg2:var(--pink-bg)"></i>Definitions object</span>
|
| 193 |
+
</div>
|
| 194 |
+
</div>
|
| 195 |
+
|
| 196 |
+
<script>
|
| 197 |
+
const details = {
|
| 198 |
+
schema: "Deliberately independent of Dagster and of compute β its own docstring states this. Covers the raw HF dataset identity (HuggingFaceBio/carbon-pretraining-corpus, config eukaryote_generator, split train), the validation-tier row table, the 14-column raw schema (EXPECTED_COLUMNS), the canonical biological composite key (record_id, start, end) β unique only after deduplication, and used directly as the GPU join key with no surrogate row_key β the IUPAC nucleotide alphabet, and the GPU output column contracts (TOKENIZED_CORPUS_COLUMNS, EMBEDDING_COLUMNS, LIKELIHOOD_COLUMNS). It also defines token_mask semantics (-2 padding, -1 BPE/text, 0 DNA special, 1..k k-mer contribution) as a shared contract precisely so no downstream stage hardcodes those values independently. It deliberately does NOT define gene_type's full vocabulary, since the dataset exposes multiple classes and the module's own comment says it shouldn't invent a complete vocabulary from a partial sample.",
|
| 199 |
+
config: "One dg.Config class, CarbonPipelineConfig, holds every runtime parameter β dataset selection, batch_size (1,000), rows_per_shard (250,000), compression (zstd), five separate output_dirs, pilot_sample_fraction/seed, cpu_workers, and gpu_token_budget (500,000,000). The docstring explains why it's one class: parameters split across several dg.Config classes get reinterpreted by Dagster as asset inputs rather than configuration. Each output_dir is kept independent from the others by design β e.g. tokenized_output_dir is a sibling of output_dir, not a subdirectory keyed off it, so a schema change in tokenization never forces touching or re-running the CPU-enriched Parquet shards, and vice versa.",
|
| 200 |
+
assets: "Every CPU and GPU asset function takes a CarbonPipelineConfig parameter and reads schema.py constants directly (EXPECTED_COLUMNS, VALIDATION_LEVEL_ROWS, TOKEN_MASK_* etc.) β this is where the two foundation modules actually get used; neither schema.py nor config.py is imported by definitions.py itself.",
|
| 201 |
+
carbon_resource: "CarbonModelResource β the single project-level source of the Carbon-3B model and its hybrid 6-mer tokenizer, shared by tokenize_and_tag (tokenizer only) and the GPU enrichment assets (tokenizer + model), so both are resolved exactly once per process at one pinned revision instead of each asset calling from_pretrained() independently and risking version skew. Both are lazily loaded via properties β a resource that only needs model_checkpoint for a provenance manifest never triggers a 3B-parameter weight load. The tokenizer loads with trust_remote_code=True (required for the custom 6-mer tokenizer); the model loads in bfloat16 β its native training precision, not a throughput tradeoff β with FlashAttention-2 wired through the HF Kernels Hub rather than pip flash-attn, then put in .eval() mode. It records model-lifecycle telemetry only (load times, device/dtype, param count, resolved attention implementation and FA2 kernel revision); inference telemetry like throughput, OOM retries, and peak CUDA memory belongs to embeddings.py, not this resource. compile_for_buckets() is an explicit method the GPU asset calls at startup with the bucket/batch lookup table β the resource does not torch.compile eagerly on load, since bucket-stable static shapes are asset-level configuration.",
|
| 202 |
+
hf_resource: "A thin, deliberately non-duplicating wrapper: the module's own docstring says it intentionally does not implement a second dataset client or reimplement load_dataset(). It just configures a cache directory (.hf_cache by default) and returns a project-configured dagster_hf_datasets.HuggingFaceResource, consumed by assets/cpu/ingest.py through Dagster's dependency injection.",
|
| 203 |
+
groupings: "CPU_ASSETS = [carbon_cpu_enriched_sequences]. GPU_PIPELINE_ASSETS = [carbon_pilot_corpus, carbon_tokenized_corpus, carbon_gpu_enrichment]. Both lists exist purely so dg.Definitions(assets=[*CPU_ASSETS, *GPU_PIPELINE_ASSETS]) reads as an explicit statement of pipeline membership rather than an unlabeled list.",
|
| 204 |
+
jobs: "Four jobs, each defined with dg.in_process_executor rather than Dagster's default multiprocess executor: carbon_cpu_job (CPU stage alone), carbon_tokenize_job (tokenization alone), carbon_inference_job (embeddings + likelihood from an existing tokenized corpus), and carbon_gpu_job (pilot sampling through inference, end to end). in_process_executor is used everywhere specifically because several assets already run their own internal ProcessPoolExecutor β nesting Dagster's multiprocess executor around that would risk IPC pipe deadlocks between Dagster's step workers and the asset's own worker pool.",
|
| 205 |
+
defs: "The dg.Definitions object Dagster's CLI and UI actually discover and load β resources={\"hf_resource\": create_huggingface_resource(), \"carbon\": CarbonModelResource()} is where the two resource modules above are actually instantiated and bound by name. Two environment variables are set at the very top of this file, before dagster is even imported: TOKENIZERS_PARALLELISM=false, preventing a deadlock when the HF tokenizer's Rust threadpool is cloned across a fork, and PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True, guarding against CUDA virtual-memory fragmentation once VRAM usage climbs past roughly 70GB. Worth flagging: the module's own docstring diagram shows a carbon_likelihood_summary asset downstream of carbon_likelihood_stats, but no such asset is imported or defined anywhere in this file β a documented stage that isn't implemented yet, not a data-flow error.",
|
| 206 |
+
facebergs: "Maps existing HF datasets to Iceberg table metadata without copying data β the read path for catalog-managed lineage across five nodes: cpu_enriched, sampled_cpu, tokenized, likelihood_stats, embeddings, each declared in a PIPELINE_TABLES registry with its HF repo, upstream node_id, and access_mode. The raw pretraining_corpus stays a streaming-only node, deliberately never cataloged. Worth knowing: 'sampled_cpu' maps to the published HF artifact carbon-pilot-corpus-dedup, but the module's own docstring is explicit that the lineage edge it represents is deterministic per-row-hash sampling with stratum representativeness validation β not deduplication; the artifact name is legacy naming, not a description of what this node does. The module also monkey-patches three PyIceberg/Faceberg internals (BinaryEncoder.write_utf8, conversions.to_bytes for StringType, and iceberg.write_manifest) before any catalog operation runs, and writes a separate lineage.yml manifest that additionally records the streaming-only root node β data the underlying _LocalCatalog itself never reads or rewrites. Note: in the current repository this file is actually named facebergs.py (plural) β shown here as faceberg.py to match the import path duckdb.py actually uses.",
|
| 207 |
+
duckdb: "Provides the DuckDB+Iceberg query surface used to read catalog tables, and to recompute sampling.py's stratification logic in SQL for independent validation β length_bucket_expression() is written to be the exact SQL equivalent of the proxy buckets sampling.py computes in Python. get_connection() installs and loads DuckDB's Iceberg extension and registers a Hugging Face credential-chain secret so catalog tables backed by HF-hosted Parquet resolve without a manually passed token. This module imports PIPELINE_TABLES from carbon_enrichment.resources.faceberg (singular) β that import only resolves if the neighboring module is named faceberg.py, as shown here; in the current repository it's actually saved as facebergs.py (plural), which would make this import fail as written until one of the two names is fixed.",
|
| 208 |
+
clickhouse: "A thin wrapper around the clickhouse local CLI binary, not a persistent server connection β every query shells out to a subprocess. It can register local Parquet globs, or register a remote Hugging Face Parquet dataset as a ClickHouse url() source without downloading it: wildcard HF URLs are resolved once via the HF dataset tree API into concrete shard URLs, then compressed into ClickHouse's brace-expansion syntax. query_arrow() fully buffers a result as one Arrow table β fine for aggregates β while stream_arrow_batches() pipes ClickHouse's ArrowStream output straight into PyArrow's streaming reader so peak Python-side memory is bounded by one batch rather than the full result; the module's own docstring says to prefer the streaming path for any row-level result. This is the resource behind the Phase 4.5 case-study script that joins the CPU, likelihood, and embeddings sources against a common cohort key set.",
|
| 209 |
+
lancedb: "A thin wrapper around a local LanceDB connection, defaulting to a carbon_embeddings table with cosine distance. Every write and search call goes through the same exponential-backoff retry helper (5 attempts by default, 2.0s base delay doubling each attempt), specifically to absorb OS-level file-lock contention from concurrent local writes β something LanceDB's local file-backed format is more exposed to than a networked vector store would be. facet() hands the underlying Lance dataset to DuckDB directly via to_lance() to run a GROUP BY/COUNT without leaving Arrow format or duplicating the data β the same DuckDB-over-Arrow pattern duckdb.py uses for catalog tables, applied here to a vector table instead."
|
| 210 |
+
};
|
| 211 |
+
|
| 212 |
+
document.querySelectorAll('[data-key]').forEach(el => {
|
| 213 |
+
el.addEventListener('click', () => {
|
| 214 |
+
document.querySelectorAll('.node.active').forEach(n => n.classList.remove('active'));
|
| 215 |
+
el.classList.add('active');
|
| 216 |
+
document.getElementById('detail-body').textContent = details[el.dataset.key];
|
| 217 |
+
});
|
| 218 |
+
});
|
| 219 |
+
</script>
|
| 220 |
+
</body>
|
| 221 |
+
</html>
|