Download index.html from AINovice2005/carbon-enrichment-pipeline-diagram-1: direct link, hf CLI and curl.
- Browser
- Download file 19.3 kB
-
https://huggingface.co/spaces/AINovice2005/carbon-enrichment-pipeline-diagram-1/resolve/main/index.html
- Command line
-
hf download hf://spaces/AINovice2005/carbon-enrichment-pipeline-diagram-1/index.html
-
curl -L -o index.html https://huggingface.co/spaces/AINovice2005/carbon-enrichment-pipeline-diagram-1/resolve/main/index.html
19.3 kB
| <html> | |
| <head> | |
| <meta charset="utf-8"> | |
| <title>schema.py Β· config.py Β· definitions.py</title> | |
| <style> | |
| :root{ | |
| --bg:#ffffff; --card:#ffffff; --ink:#1a1a1a; --sub:#6b6b6b; --line:#d8d8d8; | |
| --coral-bg:#FAECE7; --coral-bd:#D85A30; --coral-tx:#712B13; | |
| --amber-bg:#FAEEDA; --amber-bd:#BA7517; --amber-tx:#633806; | |
| --blue-bg:#E6F1FB; --blue-bd:#378ADD; --blue-tx:#0C447C; | |
| --teal-bg:#E1F5EE; --teal-bd:#1D9E75; --teal-tx:#085041; | |
| --purple-bg:#EEEDFE;--purple-bd:#7F77DD;--purple-tx:#3C3489; | |
| --green-bg:#EAF3DE; --green-bd:#639922; --green-tx:#27500A; | |
| --pink-bg:#FBEAF0; --pink-bd:#D4537E; --pink-tx:#72243E; | |
| --gray-bg:#F1EFE8; --gray-bd:#888780; --gray-tx:#444441; | |
| } | |
| @media (prefers-color-scheme: dark){ | |
| :root:not([data-theme="light"]){ | |
| --bg:#1c1c1a; --card:#26261f; --ink:#eeeee6; --sub:#a3a299; --line:#3c3c36; | |
| --coral-bg:#4A1B0C; --coral-bd:#F0997B; --coral-tx:#F5C4B3; | |
| --amber-bg:#412402; --amber-bd:#EF9F27; --amber-tx:#FAC775; | |
| --blue-bg:#042C53; --blue-bd:#85B7EB; --blue-tx:#B5D4F4; | |
| --teal-bg:#04342C; --teal-bd:#5DCAA5; --teal-tx:#9FE1CB; | |
| --purple-bg:#26215C;--purple-bd:#AFA9EC;--purple-tx:#CECBF6; | |
| --green-bg:#173404; --green-bd:#97C459; --green-tx:#C0DD97; | |
| --pink-bg:#4B1528; --pink-bd:#ED93B1; --pink-tx:#F4C0D1; | |
| --gray-bg:#2C2C2A; --gray-bd:#B4B2A9; --gray-tx:#D3D1C7; | |
| } | |
| } | |
| :root[data-theme="dark"]{ | |
| --bg:#1c1c1a; --card:#26261f; --ink:#eeeee6; --sub:#a3a299; --line:#3c3c36; | |
| --coral-bg:#4A1B0C; --coral-bd:#F0997B; --coral-tx:#F5C4B3; | |
| --amber-bg:#412402; --amber-bd:#EF9F27; --amber-tx:#FAC775; | |
| --blue-bg:#042C53; --blue-bd:#85B7EB; --blue-tx:#B5D4F4; | |
| --teal-bg:#04342C; --teal-bd:#5DCAA5; --teal-tx:#9FE1CB; | |
| --purple-bg:#26215C;--purple-bd:#AFA9EC;--purple-tx:#CECBF6; | |
| --green-bg:#173404; --green-bd:#97C459; --green-tx:#C0DD97; | |
| --pink-bg:#4B1528; --pink-bd:#ED93B1; --pink-tx:#F4C0D1; | |
| --gray-bg:#2C2C2A; --gray-bd:#B4B2A9; --gray-tx:#D3D1C7; | |
| } | |
| *{box-sizing:border-box;} | |
| body{margin:0;background:var(--bg);color:var(--ink);font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",Helvetica,Arial,sans-serif;} | |
| .wrap{max-width:780px;margin:0 auto;padding:28px 20px 40px;} | |
| h1{font-size:17px;font-weight:600;margin:0 0 2px;} | |
| .subtitle{font-size:13px;color:var(--sub);margin:0 0 22px;} | |
| .hint{font-size:12px;color:var(--sub);text-align:center;margin:0 0 18px;} | |
| .flow{display:flex;flex-direction:column;align-items:center;gap:0;} | |
| .node{ | |
| border-radius:10px;border:1.5px solid var(--bd); | |
| background:var(--fill);padding:10px 16px;cursor:pointer;transition:transform .12s ease, box-shadow .12s ease; | |
| text-align:center; | |
| } | |
| .node:hover, .node.active{transform:translateY(-1px);box-shadow:0 2px 10px rgba(0,0,0,.08);} | |
| .node .t{font-size:13.5px;font-weight:600;color:var(--tx);} | |
| .node .s{font-size:11.5px;color:var(--tx);opacity:.75;margin-top:1px;} | |
| .arrow{width:1.5px;height:22px;background:var(--line);position:relative;} | |
| .arrow::after{ | |
| content:"";position:absolute;left:50%;bottom:-1px;transform:translateX(-50%); | |
| width:0;height:0;border-left:5px solid transparent;border-right:5px solid transparent; | |
| border-top:6px solid var(--line); | |
| } | |
| .arrow-lbl{font-size:10.5px;color:var(--sub);margin:2px 0;} | |
| .row{display:flex;gap:18px;justify-content:center;align-items:flex-start;flex-wrap:wrap;width:100%;} | |
| .row .node{width:100%;max-width:260px;flex:1 1 240px;} | |
| .hjoin{position:relative;width:1.5px;background:var(--line);align-self:center;} | |
| .full .node{width:100%;max-width:560px;} | |
| .pool{ | |
| width:100%;max-width:600px;border:1.5px dashed var(--line);border-radius:12px; | |
| padding:14px;margin:2px 0; | |
| } | |
| .pool-label{font-size:11px;color:var(--sub);text-align:center;margin-bottom:10px;} | |
| .pool-row{display:flex;gap:10px;justify-content:center;flex-wrap:wrap;} | |
| .pool-row .node{max-width:180px;flex:1 1 160px;padding:9px 10px;} | |
| .pool-row .node .t{font-size:12.5px;} | |
| .pool-row .node .s{font-size:10.5px;} | |
| .detail{ | |
| margin-top:22px;border:1px solid var(--line);border-radius:10px;background:var(--card); | |
| padding:14px 16px;font-size:13px;line-height:1.55;color:var(--ink);min-height:64px; | |
| } | |
| .detail .k{font-size:11px;color:var(--sub);text-transform:uppercase;letter-spacing:.04em;margin-bottom:4px;} | |
| .legend{display:flex;flex-wrap:wrap;gap:8px 14px;justify-content:center;margin-top:20px;} | |
| .legend span{display:inline-flex;align-items:center;gap:6px;font-size:11.5px;color:var(--sub);} | |
| .legend i{width:10px;height:10px;border-radius:3px;display:inline-block;border:1.2px solid var(--bd2);background:var(--bg2);} | |
| </style> | |
| </head> | |
| <body> | |
| <div class="wrap"> | |
| <h1>schema.py Β· config.py Β· definitions.py</h1> | |
| <p class="subtitle">The pipeline's contract, configuration, and wiring layer β not a data flow, a dependency graph</p> | |
| <p class="hint">Click any block for details</p> | |
| <div class="flow"> | |
| <div class="row"> | |
| <div class="node" style="--fill:var(--coral-bg);--bd:var(--coral-bd);--tx:var(--coral-tx)" data-key="schema"> | |
| <div class="t">schema.py</div> | |
| <div class="s">Data contracts β no Dagster, no compute</div> | |
| </div> | |
| <div class="node" style="--fill:var(--amber-bg);--bd:var(--amber-bd);--tx:var(--amber-tx)" data-key="config"> | |
| <div class="t">config.py</div> | |
| <div class="s">CarbonPipelineConfig, StorageConfig</div> | |
| </div> | |
| </div> | |
| <div class="arrow-lbl">config.py imports DEFAULT_VALIDATION_LEVEL from schema.py β</div> | |
| <div class="arrow"></div> | |
| <div class="full"> | |
| <div class="node" style="--fill:var(--blue-bg);--bd:var(--blue-bd);--tx:var(--blue-tx)" data-key="assets"> | |
| <div class="t">assets/cpu/*.py + assets/gpu/*.py</div> | |
| <div class="s">each asset takes CarbonPipelineConfig and reads schema.py constants directly</div> | |
| </div> | |
| </div> | |
| <div class="arrow"></div> | |
| <div class="row"> | |
| <div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="carbon_resource"> | |
| <div class="t">resources/carbon.py</div> | |
| <div class="s">CarbonModelResource β model + tokenizer</div> | |
| </div> | |
| <div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="hf_resource"> | |
| <div class="t">resources/hf_client.py</div> | |
| <div class="s">create_huggingface_resource()</div> | |
| </div> | |
| </div> | |
| <div class="arrow"></div> | |
| <div class="pool"> | |
| <div class="pool-label">definitions.py β imports assets + resources, wires everything together</div> | |
| <div class="pool-row"> | |
| <div class="node" style="--fill:var(--teal-bg);--bd:var(--teal-bd);--tx:var(--teal-tx)" data-key="groupings"> | |
| <div class="t">Asset groupings</div> | |
| <div class="s">CPU_ASSETS, GPU_PIPELINE_ASSETS</div> | |
| </div> | |
| <div class="node" style="--fill:var(--purple-bg);--bd:var(--purple-bd);--tx:var(--purple-tx)" data-key="jobs"> | |
| <div class="t">Job definitions</div> | |
| <div class="s">4 jobs, each an AssetSelection</div> | |
| </div> | |
| </div> | |
| </div> | |
| <div class="arrow"></div> | |
| <div class="full"> | |
| <div class="node" style="--fill:var(--pink-bg);--bd:var(--pink-bd);--tx:var(--pink-tx)" data-key="defs"> | |
| <div class="t">defs = dg.Definitions(assets, jobs, resources)</div> | |
| <div class="s">the single object Dagster actually loads</div> | |
| </div> | |
| </div> | |
| <div class="arrow-lbl">used by downstream analysis & catalog sync β not registered in dg.Definitions() β</div> | |
| <div class="arrow"></div> | |
| <div class="pool"> | |
| <div class="pool-label">resources/*.py β analytical & catalog access layer (standalone)</div> | |
| <div class="pool-row"> | |
| <div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="facebergs"> | |
| <div class="t">faceberg.py</div> | |
| <div class="s">PipelineCatalog, PIPELINE_TABLES lineage registry</div> | |
| </div> | |
| <div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="duckdb"> | |
| <div class="t">duckdb.py</div> | |
| <div class="s">DuckDB + Iceberg query surface</div> | |
| </div> | |
| <div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="clickhouse"> | |
| <div class="t">clickhouse.py</div> | |
| <div class="s">ClickHouseResource β local analytical SQL</div> | |
| </div> | |
| <div class="node" style="--fill:var(--gray-bg);--bd:var(--gray-bd);--tx:var(--gray-tx)" data-key="lancedb"> | |
| <div class="t">lancedb.py</div> | |
| <div class="s">LanceDBResource β vector storage/search</div> | |
| </div> | |
| </div> | |
| </div> | |
| </div> | |
| <div class="detail" id="detail"> | |
| <div class="k">Detail</div> | |
| <div id="detail-body">Click any block above to see how it works.</div> | |
| </div> | |
| <div class="legend"> | |
| <span><i style="--bd2:var(--coral-bd);--bg2:var(--coral-bg)"></i>Contracts</span> | |
| <span><i style="--bd2:var(--amber-bd);--bg2:var(--amber-bg)"></i>Runtime config</span> | |
| <span><i style="--bd2:var(--blue-bd);--bg2:var(--blue-bg)"></i>Asset modules</span> | |
| <span><i style="--bd2:var(--gray-bd);--bg2:var(--gray-bg)"></i>Resource modules</span> | |
| <span><i style="--bd2:var(--teal-bd);--bg2:var(--teal-bg)"></i>Asset groupings</span> | |
| <span><i style="--bd2:var(--purple-bd);--bg2:var(--purple-bg)"></i>Jobs</span> | |
| <span><i style="--bd2:var(--pink-bd);--bg2:var(--pink-bg)"></i>Definitions object</span> | |
| </div> | |
| </div> | |
| <script> | |
| const details = { | |
| schema: "Deliberately independent of Dagster and of compute β its own docstring states this. Covers the raw HF dataset identity (HuggingFaceBio/carbon-pretraining-corpus, config eukaryote_generator, split train), the validation-tier row table, the 14-column raw schema (EXPECTED_COLUMNS), the canonical biological composite key (record_id, start, end) β unique only after deduplication, and used directly as the GPU join key with no surrogate row_key β the IUPAC nucleotide alphabet, and the GPU output column contracts (TOKENIZED_CORPUS_COLUMNS, EMBEDDING_COLUMNS, LIKELIHOOD_COLUMNS). It also defines token_mask semantics (-2 padding, -1 BPE/text, 0 DNA special, 1..k k-mer contribution) as a shared contract precisely so no downstream stage hardcodes those values independently. It deliberately does NOT define gene_type's full vocabulary, since the dataset exposes multiple classes and the module's own comment says it shouldn't invent a complete vocabulary from a partial sample.", | |
| config: "One dg.Config class, CarbonPipelineConfig, holds every runtime parameter β dataset selection, batch_size (1,000), rows_per_shard (250,000), compression (zstd), five separate output_dirs, pilot_sample_fraction/seed, cpu_workers, and gpu_token_budget (500,000,000). The docstring explains why it's one class: parameters split across several dg.Config classes get reinterpreted by Dagster as asset inputs rather than configuration. Each output_dir is kept independent from the others by design β e.g. tokenized_output_dir is a sibling of output_dir, not a subdirectory keyed off it, so a schema change in tokenization never forces touching or re-running the CPU-enriched Parquet shards, and vice versa.", | |
| assets: "Every CPU and GPU asset function takes a CarbonPipelineConfig parameter and reads schema.py constants directly (EXPECTED_COLUMNS, VALIDATION_LEVEL_ROWS, TOKEN_MASK_* etc.) β this is where the two foundation modules actually get used; neither schema.py nor config.py is imported by definitions.py itself.", | |
| carbon_resource: "CarbonModelResource β the single project-level source of the Carbon-3B model and its hybrid 6-mer tokenizer, shared by tokenize_and_tag (tokenizer only) and the GPU enrichment assets (tokenizer + model), so both are resolved exactly once per process at one pinned revision instead of each asset calling from_pretrained() independently and risking version skew. Both are lazily loaded via properties β a resource that only needs model_checkpoint for a provenance manifest never triggers a 3B-parameter weight load. The tokenizer loads with trust_remote_code=True (required for the custom 6-mer tokenizer); the model loads in bfloat16 β its native training precision, not a throughput tradeoff β with FlashAttention-2 wired through the HF Kernels Hub rather than pip flash-attn, then put in .eval() mode. It records model-lifecycle telemetry only (load times, device/dtype, param count, resolved attention implementation and FA2 kernel revision); inference telemetry like throughput, OOM retries, and peak CUDA memory belongs to embeddings.py, not this resource. compile_for_buckets() is an explicit method the GPU asset calls at startup with the bucket/batch lookup table β the resource does not torch.compile eagerly on load, since bucket-stable static shapes are asset-level configuration.", | |
| hf_resource: "A thin, deliberately non-duplicating wrapper: the module's own docstring says it intentionally does not implement a second dataset client or reimplement load_dataset(). It just configures a cache directory (.hf_cache by default) and returns a project-configured dagster_hf_datasets.HuggingFaceResource, consumed by assets/cpu/ingest.py through Dagster's dependency injection.", | |
| groupings: "CPU_ASSETS = [carbon_cpu_enriched_sequences]. GPU_PIPELINE_ASSETS = [carbon_pilot_corpus, carbon_tokenized_corpus, carbon_gpu_enrichment]. Both lists exist purely so dg.Definitions(assets=[*CPU_ASSETS, *GPU_PIPELINE_ASSETS]) reads as an explicit statement of pipeline membership rather than an unlabeled list.", | |
| jobs: "Four jobs, each defined with dg.in_process_executor rather than Dagster's default multiprocess executor: carbon_cpu_job (CPU stage alone), carbon_tokenize_job (tokenization alone), carbon_inference_job (embeddings + likelihood from an existing tokenized corpus), and carbon_gpu_job (pilot sampling through inference, end to end). in_process_executor is used everywhere specifically because several assets already run their own internal ProcessPoolExecutor β nesting Dagster's multiprocess executor around that would risk IPC pipe deadlocks between Dagster's step workers and the asset's own worker pool.", | |
| defs: "The dg.Definitions object Dagster's CLI and UI actually discover and load β resources={\"hf_resource\": create_huggingface_resource(), \"carbon\": CarbonModelResource()} is where the two resource modules above are actually instantiated and bound by name. Two environment variables are set at the very top of this file, before dagster is even imported: TOKENIZERS_PARALLELISM=false, preventing a deadlock when the HF tokenizer's Rust threadpool is cloned across a fork, and PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True, guarding against CUDA virtual-memory fragmentation once VRAM usage climbs past roughly 70GB. Worth flagging: the module's own docstring diagram shows a carbon_likelihood_summary asset downstream of carbon_likelihood_stats, but no such asset is imported or defined anywhere in this file β a documented stage that isn't implemented yet, not a data-flow error.", | |
| facebergs: "Maps existing HF datasets to Iceberg table metadata without copying data β the read path for catalog-managed lineage across five nodes: cpu_enriched, sampled_cpu, tokenized, likelihood_stats, embeddings, each declared in a PIPELINE_TABLES registry with its HF repo, upstream node_id, and access_mode. The raw pretraining_corpus stays a streaming-only node, deliberately never cataloged. Worth knowing: 'sampled_cpu' maps to the published HF artifact carbon-pilot-corpus-dedup, but the module's own docstring is explicit that the lineage edge it represents is deterministic per-row-hash sampling with stratum representativeness validation β not deduplication; the artifact name is legacy naming, not a description of what this node does. The module also monkey-patches three PyIceberg/Faceberg internals (BinaryEncoder.write_utf8, conversions.to_bytes for StringType, and iceberg.write_manifest) before any catalog operation runs, and writes a separate lineage.yml manifest that additionally records the streaming-only root node β data the underlying _LocalCatalog itself never reads or rewrites. Note: in the current repository this file is actually named facebergs.py (plural) β shown here as faceberg.py to match the import path duckdb.py actually uses.", | |
| duckdb: "Provides the DuckDB+Iceberg query surface used to read catalog tables, and to recompute sampling.py's stratification logic in SQL for independent validation β length_bucket_expression() is written to be the exact SQL equivalent of the proxy buckets sampling.py computes in Python. get_connection() installs and loads DuckDB's Iceberg extension and registers a Hugging Face credential-chain secret so catalog tables backed by HF-hosted Parquet resolve without a manually passed token. This module imports PIPELINE_TABLES from carbon_enrichment.resources.faceberg (singular) β that import only resolves if the neighboring module is named faceberg.py, as shown here; in the current repository it's actually saved as facebergs.py (plural), which would make this import fail as written until one of the two names is fixed.", | |
| clickhouse: "A thin wrapper around the clickhouse local CLI binary, not a persistent server connection β every query shells out to a subprocess. It can register local Parquet globs, or register a remote Hugging Face Parquet dataset as a ClickHouse url() source without downloading it: wildcard HF URLs are resolved once via the HF dataset tree API into concrete shard URLs, then compressed into ClickHouse's brace-expansion syntax. query_arrow() fully buffers a result as one Arrow table β fine for aggregates β while stream_arrow_batches() pipes ClickHouse's ArrowStream output straight into PyArrow's streaming reader so peak Python-side memory is bounded by one batch rather than the full result; the module's own docstring says to prefer the streaming path for any row-level result. This is the resource behind the Phase 4.5 case-study script that joins the CPU, likelihood, and embeddings sources against a common cohort key set.", | |
| lancedb: "A thin wrapper around a local LanceDB connection, defaulting to a carbon_embeddings table with cosine distance. Every write and search call goes through the same exponential-backoff retry helper (5 attempts by default, 2.0s base delay doubling each attempt), specifically to absorb OS-level file-lock contention from concurrent local writes β something LanceDB's local file-backed format is more exposed to than a networked vector store would be. facet() hands the underlying Lance dataset to DuckDB directly via to_lance() to run a GROUP BY/COUNT without leaving Arrow format or duplicating the data β the same DuckDB-over-Arrow pattern duckdb.py uses for catalog tables, applied here to a vector table instead." | |
| }; | |
| document.querySelectorAll('[data-key]').forEach(el => { | |
| el.addEventListener('click', () => { | |
| document.querySelectorAll('.node.active').forEach(n => n.classList.remove('active')); | |
| el.classList.add('active'); | |
| document.getElementById('detail-body').textContent = details[el.dataset.key]; | |
| }); | |
| }); | |
| </script> | |
| </body> | |
| </html> |