Upload dump_kept_sample.py with huggingface_hub
Browse files- dump_kept_sample.py +69 -0
dump_kept_sample.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Dump a sample of FineWeb-Edu docs that PASS our build-integrated decontam (KEPT docs)
|
| 3 |
+
as jsonl {"text": ...}, for Hart's independent decontam_gate.py verification.
|
| 4 |
+
|
| 5 |
+
Same decontam logic + index as build_v2_blend.py -> if Hart's gate finds >0 hits on these
|
| 6 |
+
KEPT docs, our build-decontam has a gap. Expected: 0 hits (parity / clean-by-construction).
|
| 7 |
+
"""
|
| 8 |
+
import argparse
|
| 9 |
+
import hashlib
|
| 10 |
+
import json
|
| 11 |
+
import re
|
| 12 |
+
from pathlib import Path
|
| 13 |
+
|
| 14 |
+
N_HARD = 13
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
def normalize(text):
|
| 18 |
+
text = text.lower()
|
| 19 |
+
text = re.sub(r"[^\w\s]", " ", text)
|
| 20 |
+
text = re.sub(r"\s+", " ", text).strip()
|
| 21 |
+
return text
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def shingles(text, n):
|
| 25 |
+
words = normalize(text).split()
|
| 26 |
+
return {hashlib.blake2b(" ".join(words[i:i + n]).encode("utf-8"), digest_size=8).hexdigest()
|
| 27 |
+
for i in range(len(words) - n + 1)}
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
def main():
|
| 31 |
+
ap = argparse.ArgumentParser()
|
| 32 |
+
ap.add_argument("--decontam-index", required=True)
|
| 33 |
+
ap.add_argument("--out", required=True)
|
| 34 |
+
ap.add_argument("--n-kept", type=int, default=100000)
|
| 35 |
+
a = ap.parse_args()
|
| 36 |
+
|
| 37 |
+
from datasets import load_dataset
|
| 38 |
+
idx = json.loads(Path(a.decontam_index).read_text())
|
| 39 |
+
hard = set(idx["hard_hashes"])
|
| 40 |
+
blimp = set(idx["blimp_hashes"])
|
| 41 |
+
|
| 42 |
+
def contaminated(text):
|
| 43 |
+
sh = shingles(text, N_HARD)
|
| 44 |
+
if sh & hard:
|
| 45 |
+
return True
|
| 46 |
+
if blimp and sh and len(sh & blimp) / len(sh) > 0.005:
|
| 47 |
+
return True
|
| 48 |
+
return False
|
| 49 |
+
|
| 50 |
+
ds = load_dataset("HuggingFaceFW/fineweb-edu", "default", split="train", streaming=True)
|
| 51 |
+
kept = seen = dropped = 0
|
| 52 |
+
with open(a.out, "w", encoding="utf-8") as f:
|
| 53 |
+
for ex in ds:
|
| 54 |
+
seen += 1
|
| 55 |
+
t = ex.get("text") or ""
|
| 56 |
+
if contaminated(t):
|
| 57 |
+
dropped += 1
|
| 58 |
+
else:
|
| 59 |
+
f.write(json.dumps({"text": t}) + "\n")
|
| 60 |
+
kept += 1
|
| 61 |
+
if kept >= a.n_kept:
|
| 62 |
+
break
|
| 63 |
+
if seen % 20000 == 0:
|
| 64 |
+
print(f" seen={seen:,} kept={kept:,} dropped={dropped:,}", flush=True)
|
| 65 |
+
print(f"DONE kept={kept:,}/{seen:,} dropped(my-decontam)={dropped:,} -> {a.out}", flush=True)
|
| 66 |
+
|
| 67 |
+
|
| 68 |
+
if __name__ == "__main__":
|
| 69 |
+
main()
|