// greek-cc extraction progress, inferred purely from what's on the HF Hub -- // the extraction CLI keeps no central state: // // - MANIFEST_REPO holds .parquet per crawl queued for extraction; its // row count (read from the parquet footer only) is the work total. // - DATASET_REPO holds raw//chunk_.parquet, uploaded after // every stage-1 chunk, and /*.parquet once stage 2 (dedup) finishes. // // Chunks run in offset order, so the highest uploaded offset (+ one chunk) // over the manifest row count is the progress. Chunks with zero survivors // upload nothing, which is why we use the max offset rather than a file count. // // Both repos are public, so all of this runs unauthenticated in the browser. import { asyncBufferFromUrl, parquetMetadataAsync } from "https://cdn.jsdelivr.net/npm/hyparquet@1.31.1/+esm"; export const MANIFEST_REPO = "alexliap/greek-cc-manifests"; export const DATASET_REPO = "alexliap/greek-cc"; const DEFAULT_CHUNK_SIZE = 200_000; // fallback only; inferred per crawl from offset gaps export const STALL_AFTER_H = 6; // Every crawl the project targets, oldest to newest -- mirrors // greek_cc/crawls.py plus the four earliest crawls already extracted // (commented out there). A crawl with no manifest yet shows up as "not // started", so what's coming next is visible before its manifest exists. // Edit this if the target list changes. export const TARGET_CRAWLS = [ "CC-MAIN-2024-22", "CC-MAIN-2024-26", "CC-MAIN-2024-30", "CC-MAIN-2024-33", "CC-MAIN-2024-38", "CC-MAIN-2024-42", "CC-MAIN-2024-46", "CC-MAIN-2024-51", "CC-MAIN-2025-05", "CC-MAIN-2025-08", "CC-MAIN-2025-13", "CC-MAIN-2025-18", "CC-MAIN-2025-21", "CC-MAIN-2025-26", "CC-MAIN-2025-30", "CC-MAIN-2025-33", "CC-MAIN-2025-38", "CC-MAIN-2025-43", "CC-MAIN-2025-47", "CC-MAIN-2025-51", "CC-MAIN-2026-04", "CC-MAIN-2026-08", "CC-MAIN-2026-12", "CC-MAIN-2026-17", "CC-MAIN-2026-21", "CC-MAIN-2026-25", "CC-MAIN-2026-30", ]; const HUB = "https://huggingface.co"; const CHUNK_RE = /^raw\/([^/]+)\/chunk_(\d+)\.parquet$/; async function tree(repo, path = "", { recursive = false, expand = false } = {}) { const params = new URLSearchParams({ recursive, expand }); // no trailing slash for the root: ".../tree/main/" 302s to ".../tree/main" // without CORS headers, which the browser reports as a network error const suffix = path ? `/${encodeURIComponent(path)}` : ""; let url = `${HUB}/api/datasets/${repo}/tree/main${suffix}?${params}`; const out = []; while (url) { const r = await fetch(url); if (!r.ok) throw new Error(`${r.status} listing ${repo}/${path}`); out.push(...(await r.json())); url = r.headers.get("link")?.match(/<([^>]+)>;\s*rel="next"/)?.[1]; } return out; } async function manifestRows(file) { const buf = await asyncBufferFromUrl({ url: `${HUB}/datasets/${MANIFEST_REPO}/resolve/main/${file.path}`, byteLength: file.size, }); return Number((await parquetMetadataAsync(buf)).num_rows); } async function commitDates(path) { return (await tree(DATASET_REPO, path, { expand: true })) .filter((f) => f.type === "file" && f.lastCommit) .map((f) => new Date(f.lastCommit.date)) .sort((a, b) => a - b); } function chunkSize(offsets) { const gaps = new Map(); for (let i = 1; i < offsets.length; i++) { const g = offsets[i] - offsets[i - 1]; if (g > 0) gaps.set(g, (gaps.get(g) || 0) + 1); } let best = DEFAULT_CHUNK_SIZE, n = 0; for (const [g, c] of gaps) if (c > n) [best, n] = [g, c]; return best; } export async function collect() { const now = new Date(); const [manifestFiles, datasetFiles] = await Promise.all([ tree(MANIFEST_REPO), tree(DATASET_REPO, "", { recursive: true }), ]); const offsets = {}; const finalFiles = {}; for (const f of datasetFiles) { if (f.type !== "file") continue; const m = f.path.match(CHUNK_RE); if (m) (offsets[m[1]] ||= []).push(Number(m[2])); else if (f.path.endsWith(".parquet") && f.path.split("/").length === 2) { const crawl = f.path.split("/")[0]; finalFiles[crawl] = (finalFiles[crawl] || 0) + 1; } } const manifestByName = new Map( manifestFiles .filter((f) => f.type === "file" && f.path.endsWith(".parquet")) .map((f) => [f.path.replace(/\.parquet$/, ""), f]), ); // union with the target list so a crawl with no manifest yet still shows, // as "not started" -- what's coming next, not just what's already queued const names = [...new Set([...TARGET_CRAWLS, ...manifestByName.keys()])].sort((a, b) => b.localeCompare(a)); const crawls = await Promise.all(names.map(async (crawl) => { const file = manifestByName.get(crawl); if (!file) { return { crawl, rows: null, nChunks: null, chunksDone: 0, finalFiles: 0, last: null, etaS: null, status: "not_started", frac: 0 }; } const offs = (offsets[crawl] || []).sort((a, b) => a - b); const step = chunkSize(offs); const hasFinal = Boolean(finalFiles[crawl]); const rows = await manifestRows(file); const nChunks = Math.max(1, Math.ceil(rows / step)); const c = { crawl, rows, nChunks, chunksDone: offs.length ? Math.min(nChunks, Math.floor(offs.at(-1) / step) + 1) : 0, finalFiles: finalFiles[crawl] || 0, last: null, etaS: null, }; if (hasFinal) { c.status = "done"; c.chunksDone = nChunks; c.last = (await commitDates(crawl)).at(-1) || null; } else if (!offs.length) { c.status = "queued"; } else { const dates = await commitDates(`raw/${crawl}`); c.last = dates.at(-1) || null; const idleS = c.last ? (now - c.last) / 1000 : Infinity; if (c.chunksDone >= nChunks) { c.status = "deduping"; } else if (idleS > STALL_AFTER_H * 3600) { c.status = "stalled"; } else { c.status = "extracting"; const recent = dates.slice(-11); if (recent.length >= 2) { const perChunk = (recent.at(-1) - recent[0]) / 1000 / (recent.length - 1); c.etaS = Math.max(0, perChunk * (nChunks - c.chunksDone) - idleS); } } } c.frac = c.chunksDone / nChunks; return c; })); // overall progress is the average of every target crawl's own fraction -- // a crawl with no manifest yet has frac 0, same as one that hasn't started // uploading chunks, so the remaining crawls pull the total down too, not // just the ones already queued. This weighs each crawl equally rather // than by its (differing) chunk count. return { crawls, fetchedAt: now, overall: crawls.length ? crawls.reduce((s, c) => s + c.frac, 0) / crawls.length : 0, done: crawls.filter((c) => c.status === "done").length, active: crawls.filter((c) => c.status === "extracting" || c.status === "deduping").length, }; }