Download dump_kept_sample.py from SlayerLab/gollem-v5-ckpts: direct link, hf CLI and curl.
- Browser
- Download file 2.17 kB
-
https://huggingface.co/SlayerLab/gollem-v5-ckpts/resolve/2f7b1bd907e08758945600e2456577c96f2cebd9/dump_kept_sample.py
- Command line
-
hf download hf://SlayerLab/gollem-v5-ckpts@2f7b1bd907e08758945600e2456577c96f2cebd9/dump_kept_sample.py
-
curl -L -o dump_kept_sample.py https://huggingface.co/SlayerLab/gollem-v5-ckpts/resolve/2f7b1bd907e08758945600e2456577c96f2cebd9/dump_kept_sample.py
2.17 kB
| #!/usr/bin/env python3 | |
| """Dump a sample of FineWeb-Edu docs that PASS our build-integrated decontam (KEPT docs) | |
| as jsonl {"text": ...}, for Hart's independent decontam_gate.py verification. | |
| Same decontam logic + index as build_v2_blend.py -> if Hart's gate finds >0 hits on these | |
| KEPT docs, our build-decontam has a gap. Expected: 0 hits (parity / clean-by-construction). | |
| """ | |
| import argparse | |
| import hashlib | |
| import json | |
| import re | |
| from pathlib import Path | |
| N_HARD = 13 | |
| def normalize(text): | |
| text = text.lower() | |
| text = re.sub(r"[^\w\s]", " ", text) | |
| text = re.sub(r"\s+", " ", text).strip() | |
| return text | |
| def shingles(text, n): | |
| words = normalize(text).split() | |
| return {hashlib.blake2b(" ".join(words[i:i + n]).encode("utf-8"), digest_size=8).hexdigest() | |
| for i in range(len(words) - n + 1)} | |
| def main(): | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--decontam-index", required=True) | |
| ap.add_argument("--out", required=True) | |
| ap.add_argument("--n-kept", type=int, default=100000) | |
| a = ap.parse_args() | |
| from datasets import load_dataset | |
| idx = json.loads(Path(a.decontam_index).read_text()) | |
| hard = set(idx["hard_hashes"]) | |
| blimp = set(idx["blimp_hashes"]) | |
| def contaminated(text): | |
| sh = shingles(text, N_HARD) | |
| if sh & hard: | |
| return True | |
| if blimp and sh and len(sh & blimp) / len(sh) > 0.005: | |
| return True | |
| return False | |
| ds = load_dataset("HuggingFaceFW/fineweb-edu", "default", split="train", streaming=True) | |
| kept = seen = dropped = 0 | |
| with open(a.out, "w", encoding="utf-8") as f: | |
| for ex in ds: | |
| seen += 1 | |
| t = ex.get("text") or "" | |
| if contaminated(t): | |
| dropped += 1 | |
| else: | |
| f.write(json.dumps({"text": t}) + "\n") | |
| kept += 1 | |
| if kept >= a.n_kept: | |
| break | |
| if seen % 20000 == 0: | |
| print(f" seen={seen:,} kept={kept:,} dropped={dropped:,}", flush=True) | |
| print(f"DONE kept={kept:,}/{seen:,} dropped(my-decontam)={dropped:,} -> {a.out}", flush=True) | |
| if __name__ == "__main__": | |
| main() | |