Download build_decontam_index.py from SlayerLab/gollem-v5-ckpts: direct link, hf CLI and curl.
- Browser
- Download file 4.03 kB
-
https://huggingface.co/SlayerLab/gollem-v5-ckpts/resolve/0da09be998d4e11cb747939d94b2ec6a62f5a541/build_decontam_index.py
- Command line
-
hf download hf://SlayerLab/gollem-v5-ckpts@0da09be998d4e11cb747939d94b2ec6a62f5a541/build_decontam_index.py
-
curl -L -o build_decontam_index.py https://huggingface.co/SlayerLab/gollem-v5-ckpts/resolve/0da09be998d4e11cb747939d94b2ec6a62f5a541/build_decontam_index.py
4.03 kB
| #!/usr/bin/env python3 | |
| """RB3 decontam reference-index: 13-gram word-shingles z benchmarkow eval (Glint-parity). | |
| Cel: zbudowac indeks n-gramow z DOKLADNIE tych zbiorow ktore ewaluujemy (WikiText-2 test, | |
| BLiMP wszystkie configi good+bad, ARC-Easy+Challenge test), zeby filtr UFW mogl wyrzucic | |
| zanieczyszczone dokumenty. Prawda-z-bajtow: zrodla identyczne z glint_parity_eval.py. | |
| Reguly (RB3-spec): | |
| - WikiText-2 + ARC (E+C): drop-UFW-doc na ANY-13gram-match (male zbiory, czystosc krytyczna). | |
| - BLiMP: match-fraction > 0.5% (krotkie zdania, unikamy ciecia generycznych). | |
| Indeks: blake2b-8B hash kazdego 13-gramu (word-level, znormalizowany). Zapis: decontam_index.json | |
| {version, n13, hard_hashes:[...], blimp_hashes:[...], counts:{...}}. | |
| Uzycie: python build_decontam_index.py [--out decontam_index.json] | |
| """ | |
| import argparse | |
| import hashlib | |
| import json | |
| import re | |
| import sys | |
| from pathlib import Path | |
| # reuse dokladnych loaderow + configow z eval (te same zrodla = wlasciwa dekontaminacja) | |
| sys.path.insert(0, str(Path(__file__).parent)) | |
| from glint_parity_eval import _rows, BLIMP_CONFIGS # noqa: E402 | |
| N = 13 # 13-gram word-level (konwencja GPT-3/FineWeb) | |
| def normalize(text): | |
| text = text.lower() | |
| text = re.sub(r"[^\w\s]", " ", text) # strip punctuation | |
| text = re.sub(r"\s+", " ", text).strip() | |
| return text | |
| def shingle_hashes(text, n=N): | |
| words = normalize(text).split() | |
| out = set() | |
| for i in range(len(words) - n + 1): | |
| sh = " ".join(words[i:i + n]) | |
| out.add(hashlib.blake2b(sh.encode("utf-8"), digest_size=8).hexdigest()) | |
| return out | |
| def collect(texts, label, n=N): | |
| idx = set() | |
| docs = 0 | |
| for t in texts: | |
| if not t or not t.strip(): | |
| continue | |
| docs += 1 | |
| idx |= shingle_hashes(t, n) | |
| print(f" [{label}] docs={docs:,} shingles({n}gram)={len(idx):,}", flush=True) | |
| return idx | |
| def main(): | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--out", default="decontam_index.json") | |
| a = ap.parse_args() | |
| print("=== RB3 decontam-index build (13-gram shingles) ===", flush=True) | |
| # --- HARD sources (drop-on-any-match): WikiText-2 test + ARC-E/C test --- | |
| hard_texts = [] | |
| print("WikiText-2 test...", flush=True) | |
| hard_texts.extend(r["text"] for r in _rows("Salesforce/wikitext", "wikitext-2-raw-v1", "test")) | |
| for cfg in ("ARC-Easy", "ARC-Challenge"): | |
| print(f"ARC {cfg} test...", flush=True) | |
| for ex in _rows("allenai/ai2_arc", cfg, "test"): | |
| hard_texts.append(ex["question"]) | |
| hard_texts.extend(ex["choices"]["text"]) | |
| hard = collect(hard_texts, "hard-13gram", n=13) | |
| synth_hard = collect(hard_texts, "hard-8gram-synth-paraphrase-gate", n=8) | |
| # --- BLiMP (fraction-threshold source): all configs, good+bad --- | |
| print(f"BLiMP ({len(BLIMP_CONFIGS)} configs) good+bad...", flush=True) | |
| blimp_texts = [] | |
| for c in BLIMP_CONFIGS: | |
| for e in _rows("nyu-mll/blimp", c, "train"): | |
| blimp_texts.append(e["sentence_good"]) | |
| blimp_texts.append(e["sentence_bad"]) | |
| blimp = collect(blimp_texts, "blimp-all") | |
| out = { | |
| "version": 2, "n": N, "n_synth": 8, | |
| "hard_hashes": sorted(hard), # WikiText-2 + ARC, 13-gram (drop-on-any, all sources) | |
| "blimp_hashes": sorted(blimp), # BLiMP, 13-gram (fraction>0.5%) | |
| "synth_hard_hashes": sorted(synth_hard), # WikiText-2 + ARC, 8-gram (drop-on-any, SYNTHETIC only) | |
| "counts": {"hard": len(hard), "blimp": len(blimp), "synth_hard": len(synth_hard), | |
| "hard_sources": "wikitext2-test + arc-easy/challenge-test", | |
| "blimp_sources": f"blimp-{len(BLIMP_CONFIGS)}cfg-good+bad", | |
| "synth_hard_note": "8-gram paraphrase-proxy for synthetic sources (Cosmopedia/Nemotron)"}, | |
| } | |
| Path(a.out).write_text(json.dumps(out), encoding="utf-8") | |
| print(f"=== WROTE {a.out}: hard={len(hard):,} blimp={len(blimp):,} 13-gram-hashes ===", flush=True) | |
| if __name__ == "__main__": | |
| main() | |