#!/usr/bin/env python3 """Rebuild the training pool of the final GoLLeM-v5 runs (sha256 ecfd0a40...) from ARC-MIX. Usage: python rebuild_final_pool.py ARCMIX_train.bin removed_documents.json OUT.bin ARCMIX_train.bin: SlayerLab/gollem-ui-mirror arcmix/train.bin (commit 486a3af3c48d), uint16 tokens. Documents are the EOS-terminated spans (EOS id 12285), numbered from 0 in file order. The script drops the documents listed in removed_documents.json, keeps the order of all others, writes OUT.bin and checks the document count, the token count and the sha256 of the result. It reads the source in fixed-size blocks (peak memory about 0.5 GB) and needs about 19 GB of free disk for OUT.bin. This is the same removal as final_arcmix_pool.py (the script used for the runs), written as a standalone tool that does not need the scan files. """ import hashlib import json import sys import numpy as np EOS = 12285 BLOCK_TOKENS = 16 * 1024 * 1024 # 32 MB per read SOURCE_SHA256 = "e87f594f7866fae7e671200a1d73f2b0cf14423da7dcae5a41e540480c629a82" def main(): src, removed_json, out = sys.argv[1:4] spec = json.load(open(removed_json, encoding="utf-8")) removed = np.array(sorted(set(spec["removed_by_scan"]) | set(spec["removed_by_licence_rule"])), dtype=np.int64) assert len(removed) == spec["counts"]["removed_total"] src_sha, out_sha = hashlib.sha256(), hashlib.sha256() doc = 0 # index of the document that the next token belongs to tokens = docs = 0 with open(src, "rb") as f, open(out, "wb") as g: while True: raw = f.read(BLOCK_TOKENS * 2) if not raw: break src_sha.update(raw) arr = np.frombuffer(raw, dtype=np.uint16) eos_pos = np.flatnonzero(arr == EOS) # document index of every token in this block doc_of = doc + np.searchsorted(eos_pos, np.arange(len(arr)), side="left") keep = ~np.isin(doc_of, removed) kept = arr[keep] kept.tofile(g) out_sha.update(kept.tobytes()) tokens += len(kept) ends_here = doc + np.arange(len(eos_pos)) docs += int(np.count_nonzero(~np.isin(ends_here, removed))) doc += len(eos_pos) if src_sha.hexdigest() != SOURCE_SHA256: sys.exit("source sha256 does not match ARC-MIX e87f594f...") expected = spec["result"] print(json.dumps({"source_documents": doc, "documents": docs, "tokens": tokens, "sha256": out_sha.hexdigest()})) assert docs == expected["documents"], "document count" assert tokens == expected["tokens"], "token count" assert out_sha.hexdigest() == expected["sha256"], "sha256 of the pool" print("POOL-OK") if __name__ == "__main__": main()