AINovice2005's picture
Update index.html
5b806a5 verified
Raw History Blame Contribute Delete
19.3 kB
<!doctype html>
<html>
<head>
<meta charset="utf-8">
<title>schema.py Β· config.py Β· definitions.py</title>
<style>
:root{
--bg:#ffffff; --card:#ffffff; --ink:#1a1a1a; --sub:#6b6b6b; --line:#d8d8d8;
--coral-bg:#FAECE7; --coral-bd:#D85A30; --coral-tx:#712B13;
--amber-bg:#FAEEDA; --amber-bd:#BA7517; --amber-tx:#633806;
--blue-bg:#E6F1FB; --blue-bd:#378ADD; --blue-tx:#0C447C;
--teal-bg:#E1F5EE; --teal-bd:#1D9E75; --teal-tx:#085041;
--purple-bg:#EEEDFE;--purple-bd:#7F77DD;--purple-tx:#3C3489;
--green-bg:#EAF3DE; --green-bd:#639922; --green-tx:#27500A;
--pink-bg:#FBEAF0; --pink-bd:#D4537E; --pink-tx:#72243E;
--gray-bg:#F1EFE8; --gray-bd:#888780; --gray-tx:#444441;
}
@media (prefers-color-scheme: dark){
:root:not([data-theme="light"]){
--bg:#1c1c1a; --card:#26261f; --ink:#eeeee6; --sub:#a3a299; --line:#3c3c36;
--coral-bg:#4A1B0C; --coral-bd:#F0997B; --coral-tx:#F5C4B3;
--amber-bg:#412402; --amber-bd:#EF9F27; --amber-tx:#FAC775;
--blue-bg:#042C53; --blue-bd:#85B7EB; --blue-tx:#B5D4F4;
--teal-bg:#04342C; --teal-bd:#5DCAA5; --teal-tx:#9FE1CB;
--purple-bg:#26215C;--purple-bd:#AFA9EC;--purple-tx:#CECBF6;
--green-bg:#173404; --green-bd:#97C459; --green-tx:#C0DD97;
--pink-bg:#4B1528; --pink-bd:#ED93B1; --pink-tx:#F4C0D1;
--gray-bg:#2C2C2A; --gray-bd:#B4B2A9; --gray-tx:#D3D1C7;
}
}
:root[data-theme="dark"]{
--bg:#1c1c1a; --card:#26261f; --ink:#eeeee6; --sub:#a3a299; --line:#3c3c36;
--coral-bg:#4A1B0C; --coral-bd:#F0997B; --coral-tx:#F5C4B3;
--amber-bg:#412402; --amber-bd:#EF9F27; --amber-tx:#FAC775;
--blue-bg:#042C53; --blue-bd:#85B7EB; --blue-tx:#B5D4F4;
--teal-bg:#04342C; --teal-bd:#5DCAA5; --teal-tx:#9FE1CB;
--purple-bg:#26215C;--purple-bd:#AFA9EC;--purple-tx:#CECBF6;
--green-bg:#173404; --green-bd:#97C459; --green-tx:#C0DD97;
--pink-bg:#4B1528; --pink-bd:#ED93B1; --pink-tx:#F4C0D1;
--gray-bg:#2C2C2A; --gray-bd:#B4B2A9; --gray-tx:#D3D1C7;
}
*{box-sizing:border-box;}
body{margin:0;background:var(--bg);color:var(--ink);font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",Helvetica,Arial,sans-serif;}
.wrap{max-width:780px;margin:0 auto;padding:28px 20px 40px;}
h1{font-size:17px;font-weight:600;margin:0 0 2px;}
.subtitle{font-size:13px;color:var(--sub);margin:0 0 22px;}
.hint{font-size:12px;color:var(--sub);text-align:center;margin:0 0 18px;}
.flow{display:flex;flex-direction:column;align-items:center;gap:0;}
.node{
border-radius:10px;border:1.5px solid var(--bd);
background:var(--fill);padding:10px 16px;cursor:pointer;transition:transform .12s ease, box-shadow .12s ease;
text-align:center;
}
.node:hover, .node.active{transform:translateY(-1px);box-shadow:0 2px 10px rgba(0,0,0,.08);}
.node .t{font-size:13.5px;font-weight:600;color:var(--tx);}
.node .s{font-size:11.5px;color:var(--tx);opacity:.75;margin-top:1px;}
.arrow{width:1.5px;height:22px;background:var(--line);position:relative;}
.arrow::after{
content:"";position:absolute;left:50%;bottom:-1px;transform:translateX(-50%);
width:0;height:0;border-left:5px solid transparent;border-right:5px solid transparent;
border-top:6px solid var(--line);
}
.arrow-lbl{font-size:10.5px;color:var(--sub);margin:2px 0;}
.row{display:flex;gap:18px;justify-content:center;align-items:flex-start;flex-wrap:wrap;width:100%;}
.row .node{width:100%;max-width:260px;flex:1 1 240px;}
.hjoin{position:relative;width:1.5px;background:var(--line);align-self:center;}
.full .node{width:100%;max-width:560px;}
.pool{
width:100%;max-width:600px;border:1.5px dashed var(--line);border-radius:12px;
padding:14px;margin:2px 0;
}
.pool-label{font-size:11px;color:var(--sub);text-align:center;margin-bottom:10px;}
.pool-row{display:flex;gap:10px;justify-content:center;flex-wrap:wrap;}
.pool-row .node{max-width:180px;flex:1 1 160px;padding:9px 10px;}
.pool-row .node .t{font-size:12.5px;}
.pool-row .node .s{font-size:10.5px;}
.detail{
margin-top:22px;border:1px solid var(--line);border-radius:10px;background:var(--card);
padding:14px 16px;font-size:13px;line-height:1.55;color:var(--ink);min-height:64px;
}
.detail .k{font-size:11px;color:var(--sub);text-transform:uppercase;letter-spacing:.04em;margin-bottom:4px;}
.legend{display:flex;flex-wrap:wrap;gap:8px 14px;justify-content:center;margin-top:20px;}
.legend span{display:inline-flex;align-items:center;gap:6px;font-size:11.5px;color:var(--sub);}
.legend i{width:10px;height:10px;border-radius:3px;display:inline-block;border:1.2px solid var(--bd2);background:var(--bg2);}
</style>
</head>
<body>
<div class="wrap">
<h1>schema.py Β· config.py Β· definitions.py</h1>
<p class="subtitle">The pipeline's contract, configuration, and wiring layer β€” not a data flow, a dependency graph</p>
<p class="hint">Click any block for details</p>
<div class="flow">
<div class="row">
<div class="node" style="--fill:var(--coral-bg);--bd:var(--coral-bd);--tx:var(--coral-tx)" data-key="schema">
<div class="t">schema.py</div>
<div class="s">Data contracts β€” no Dagster, no compute</div>
</div>
<div class="node" style="--fill:var(--amber-bg);--bd:var(--amber-bd);--tx:var(--amber-tx)" data-key="config">
<div class="t">config.py</div>
<div class="s">CarbonPipelineConfig, StorageConfig</div>
</div>
</div>
<div class="arrow-lbl">config.py imports DEFAULT_VALIDATION_LEVEL from schema.py β†’</div>
<div class="arrow"></div>
<div class="full">
<div class="node" style="--fill:var(--blue-bg);--bd:var(--blue-bd);--tx:var(--blue-tx)" data-key="assets">
<div class="t">assets/cpu/*.py + assets/gpu/*.py</div>
<div class="s">each asset takes CarbonPipelineConfig and reads schema.py constants directly</div>
</div>
</div>
<div class="arrow"></div>
<div class="row">
<div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="carbon_resource">
<div class="t">resources/carbon.py</div>
<div class="s">CarbonModelResource β€” model + tokenizer</div>
</div>
<div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="hf_resource">
<div class="t">resources/hf_client.py</div>
<div class="s">create_huggingface_resource()</div>
</div>
</div>
<div class="arrow"></div>
<div class="pool">
<div class="pool-label">definitions.py β€” imports assets + resources, wires everything together</div>
<div class="pool-row">
<div class="node" style="--fill:var(--teal-bg);--bd:var(--teal-bd);--tx:var(--teal-tx)" data-key="groupings">
<div class="t">Asset groupings</div>
<div class="s">CPU_ASSETS, GPU_PIPELINE_ASSETS</div>
</div>
<div class="node" style="--fill:var(--purple-bg);--bd:var(--purple-bd);--tx:var(--purple-tx)" data-key="jobs">
<div class="t">Job definitions</div>
<div class="s">4 jobs, each an AssetSelection</div>
</div>
</div>
</div>
<div class="arrow"></div>
<div class="full">
<div class="node" style="--fill:var(--pink-bg);--bd:var(--pink-bd);--tx:var(--pink-tx)" data-key="defs">
<div class="t">defs = dg.Definitions(assets, jobs, resources)</div>
<div class="s">the single object Dagster actually loads</div>
</div>
</div>
<div class="arrow-lbl">used by downstream analysis &amp; catalog sync β€” not registered in dg.Definitions() ↓</div>
<div class="arrow"></div>
<div class="pool">
<div class="pool-label">resources/*.py β€” analytical &amp; catalog access layer (standalone)</div>
<div class="pool-row">
<div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="facebergs">
<div class="t">faceberg.py</div>
<div class="s">PipelineCatalog, PIPELINE_TABLES lineage registry</div>
</div>
<div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="duckdb">
<div class="t">duckdb.py</div>
<div class="s">DuckDB + Iceberg query surface</div>
</div>
<div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="clickhouse">
<div class="t">clickhouse.py</div>
<div class="s">ClickHouseResource β€” local analytical SQL</div>
</div>
<div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="lancedb">
<div class="t">lancedb.py</div>
<div class="s">LanceDBResource β€” vector storage/search</div>
</div>
</div>
</div>
</div>
<div class="detail" id="detail">
<div class="k">Detail</div>
<div id="detail-body">Click any block above to see how it works.</div>
</div>
<div class="legend">
<span><i style="--bd2:var(--coral-bd);--bg2:var(--coral-bg)"></i>Contracts</span>
<span><i style="--bd2:var(--amber-bd);--bg2:var(--amber-bg)"></i>Runtime config</span>
<span><i style="--bd2:var(--blue-bd);--bg2:var(--blue-bg)"></i>Asset modules</span>
<span><i style="--bd2:var(--gray-bd);--bg2:var(--gray-bg)"></i>Resource modules</span>
<span><i style="--bd2:var(--teal-bd);--bg2:var(--teal-bg)"></i>Asset groupings</span>
<span><i style="--bd2:var(--purple-bd);--bg2:var(--purple-bg)"></i>Jobs</span>
<span><i style="--bd2:var(--pink-bd);--bg2:var(--pink-bg)"></i>Definitions object</span>
</div>
</div>
<script>
const details = {
schema: "Deliberately independent of Dagster and of compute β€” its own docstring states this. Covers the raw HF dataset identity (HuggingFaceBio/carbon-pretraining-corpus, config eukaryote_generator, split train), the validation-tier row table, the 14-column raw schema (EXPECTED_COLUMNS), the canonical biological composite key (record_id, start, end) β€” unique only after deduplication, and used directly as the GPU join key with no surrogate row_key β€” the IUPAC nucleotide alphabet, and the GPU output column contracts (TOKENIZED_CORPUS_COLUMNS, EMBEDDING_COLUMNS, LIKELIHOOD_COLUMNS). It also defines token_mask semantics (-2 padding, -1 BPE/text, 0 DNA special, 1..k k-mer contribution) as a shared contract precisely so no downstream stage hardcodes those values independently. It deliberately does NOT define gene_type's full vocabulary, since the dataset exposes multiple classes and the module's own comment says it shouldn't invent a complete vocabulary from a partial sample.",
config: "One dg.Config class, CarbonPipelineConfig, holds every runtime parameter β€” dataset selection, batch_size (1,000), rows_per_shard (250,000), compression (zstd), five separate output_dirs, pilot_sample_fraction/seed, cpu_workers, and gpu_token_budget (500,000,000). The docstring explains why it's one class: parameters split across several dg.Config classes get reinterpreted by Dagster as asset inputs rather than configuration. Each output_dir is kept independent from the others by design β€” e.g. tokenized_output_dir is a sibling of output_dir, not a subdirectory keyed off it, so a schema change in tokenization never forces touching or re-running the CPU-enriched Parquet shards, and vice versa.",
assets: "Every CPU and GPU asset function takes a CarbonPipelineConfig parameter and reads schema.py constants directly (EXPECTED_COLUMNS, VALIDATION_LEVEL_ROWS, TOKEN_MASK_* etc.) β€” this is where the two foundation modules actually get used; neither schema.py nor config.py is imported by definitions.py itself.",
carbon_resource: "CarbonModelResource β€” the single project-level source of the Carbon-3B model and its hybrid 6-mer tokenizer, shared by tokenize_and_tag (tokenizer only) and the GPU enrichment assets (tokenizer + model), so both are resolved exactly once per process at one pinned revision instead of each asset calling from_pretrained() independently and risking version skew. Both are lazily loaded via properties β€” a resource that only needs model_checkpoint for a provenance manifest never triggers a 3B-parameter weight load. The tokenizer loads with trust_remote_code=True (required for the custom 6-mer tokenizer); the model loads in bfloat16 β€” its native training precision, not a throughput tradeoff β€” with FlashAttention-2 wired through the HF Kernels Hub rather than pip flash-attn, then put in .eval() mode. It records model-lifecycle telemetry only (load times, device/dtype, param count, resolved attention implementation and FA2 kernel revision); inference telemetry like throughput, OOM retries, and peak CUDA memory belongs to embeddings.py, not this resource. compile_for_buckets() is an explicit method the GPU asset calls at startup with the bucket/batch lookup table β€” the resource does not torch.compile eagerly on load, since bucket-stable static shapes are asset-level configuration.",
hf_resource: "A thin, deliberately non-duplicating wrapper: the module's own docstring says it intentionally does not implement a second dataset client or reimplement load_dataset(). It just configures a cache directory (.hf_cache by default) and returns a project-configured dagster_hf_datasets.HuggingFaceResource, consumed by assets/cpu/ingest.py through Dagster's dependency injection.",
groupings: "CPU_ASSETS = [carbon_cpu_enriched_sequences]. GPU_PIPELINE_ASSETS = [carbon_pilot_corpus, carbon_tokenized_corpus, carbon_gpu_enrichment]. Both lists exist purely so dg.Definitions(assets=[*CPU_ASSETS, *GPU_PIPELINE_ASSETS]) reads as an explicit statement of pipeline membership rather than an unlabeled list.",
jobs: "Four jobs, each defined with dg.in_process_executor rather than Dagster's default multiprocess executor: carbon_cpu_job (CPU stage alone), carbon_tokenize_job (tokenization alone), carbon_inference_job (embeddings + likelihood from an existing tokenized corpus), and carbon_gpu_job (pilot sampling through inference, end to end). in_process_executor is used everywhere specifically because several assets already run their own internal ProcessPoolExecutor β€” nesting Dagster's multiprocess executor around that would risk IPC pipe deadlocks between Dagster's step workers and the asset's own worker pool.",
defs: "The dg.Definitions object Dagster's CLI and UI actually discover and load β€” resources={\"hf_resource\": create_huggingface_resource(), \"carbon\": CarbonModelResource()} is where the two resource modules above are actually instantiated and bound by name. Two environment variables are set at the very top of this file, before dagster is even imported: TOKENIZERS_PARALLELISM=false, preventing a deadlock when the HF tokenizer's Rust threadpool is cloned across a fork, and PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True, guarding against CUDA virtual-memory fragmentation once VRAM usage climbs past roughly 70GB. Worth flagging: the module's own docstring diagram shows a carbon_likelihood_summary asset downstream of carbon_likelihood_stats, but no such asset is imported or defined anywhere in this file β€” a documented stage that isn't implemented yet, not a data-flow error.",
facebergs: "Maps existing HF datasets to Iceberg table metadata without copying data β€” the read path for catalog-managed lineage across five nodes: cpu_enriched, sampled_cpu, tokenized, likelihood_stats, embeddings, each declared in a PIPELINE_TABLES registry with its HF repo, upstream node_id, and access_mode. The raw pretraining_corpus stays a streaming-only node, deliberately never cataloged. Worth knowing: 'sampled_cpu' maps to the published HF artifact carbon-pilot-corpus-dedup, but the module's own docstring is explicit that the lineage edge it represents is deterministic per-row-hash sampling with stratum representativeness validation β€” not deduplication; the artifact name is legacy naming, not a description of what this node does. The module also monkey-patches three PyIceberg/Faceberg internals (BinaryEncoder.write_utf8, conversions.to_bytes for StringType, and iceberg.write_manifest) before any catalog operation runs, and writes a separate lineage.yml manifest that additionally records the streaming-only root node β€” data the underlying _LocalCatalog itself never reads or rewrites. Note: in the current repository this file is actually named facebergs.py (plural) β€” shown here as faceberg.py to match the import path duckdb.py actually uses.",
duckdb: "Provides the DuckDB+Iceberg query surface used to read catalog tables, and to recompute sampling.py's stratification logic in SQL for independent validation β€” length_bucket_expression() is written to be the exact SQL equivalent of the proxy buckets sampling.py computes in Python. get_connection() installs and loads DuckDB's Iceberg extension and registers a Hugging Face credential-chain secret so catalog tables backed by HF-hosted Parquet resolve without a manually passed token. This module imports PIPELINE_TABLES from carbon_enrichment.resources.faceberg (singular) β€” that import only resolves if the neighboring module is named faceberg.py, as shown here; in the current repository it's actually saved as facebergs.py (plural), which would make this import fail as written until one of the two names is fixed.",
clickhouse: "A thin wrapper around the clickhouse local CLI binary, not a persistent server connection β€” every query shells out to a subprocess. It can register local Parquet globs, or register a remote Hugging Face Parquet dataset as a ClickHouse url() source without downloading it: wildcard HF URLs are resolved once via the HF dataset tree API into concrete shard URLs, then compressed into ClickHouse's brace-expansion syntax. query_arrow() fully buffers a result as one Arrow table β€” fine for aggregates β€” while stream_arrow_batches() pipes ClickHouse's ArrowStream output straight into PyArrow's streaming reader so peak Python-side memory is bounded by one batch rather than the full result; the module's own docstring says to prefer the streaming path for any row-level result. This is the resource behind the Phase 4.5 case-study script that joins the CPU, likelihood, and embeddings sources against a common cohort key set.",
lancedb: "A thin wrapper around a local LanceDB connection, defaulting to a carbon_embeddings table with cosine distance. Every write and search call goes through the same exponential-backoff retry helper (5 attempts by default, 2.0s base delay doubling each attempt), specifically to absorb OS-level file-lock contention from concurrent local writes β€” something LanceDB's local file-backed format is more exposed to than a networked vector store would be. facet() hands the underlying Lance dataset to DuckDB directly via to_lance() to run a GROUP BY/COUNT without leaving Arrow format or duplicating the data β€” the same DuckDB-over-Arrow pattern duckdb.py uses for catalog tables, applied here to a vector table instead."
};
document.querySelectorAll('[data-key]').forEach(el => {
el.addEventListener('click', () => {
document.querySelectorAll('.node.active').forEach(n => n.classList.remove('active'));
el.classList.add('active');
document.getElementById('detail-body').textContent = details[el.dataset.key];
});
});
</script>
</body>
</html>