RB4 runda 3: blend_rb4/extract/launch_r3 + build_v2_blend fix (eot z tokenizera, brak klucza synth = stop); skan egress GREEN
Browse files- blend_rb4.py +139 -0
- build_v2_blend.py +14 -5
- extract_edu_from_blend.py +70 -0
- launch_r3.sh +36 -0
blend_rb4.py
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""RB4 blend builder (runda 3+), zastepuje blend_edu.py z rundy 1.
|
| 3 |
+
|
| 4 |
+
Poprawki wobec rundy 1 (confoundy z 2026-09-25):
|
| 5 |
+
- separator dokumentu brany z tokenizera (<|endoftext|>), nigdy vocab-1;
|
| 6 |
+
- zrodlo probkowane z CALEGO pliku na granicach dokumentow (seedowany wybor), nie z glowy;
|
| 7 |
+
- manifest z tokenami per zrodlo i per region (dokumenty czatu/QA = zawieraja <|im_start|>).
|
| 8 |
+
|
| 9 |
+
Tryby:
|
| 10 |
+
index --src arcmix.bin --out arcmix_index.npz
|
| 11 |
+
granice dokumentow + flaga QA (dokument zawiera <|im_start|>)
|
| 12 |
+
sample --src arcmix.bin --index arcmix_index.npz --tokens N [--qa-weight W] --out X.bin
|
| 13 |
+
seedowany wybor dokumentow (dokument QA wystepuje W razy w puli) az do N tokenow,
|
| 14 |
+
zapis w przetasowanej kolejnosci
|
| 15 |
+
mix --a A.bin --b B.bin --a-tokens N --b-tokens M --out X.bin
|
| 16 |
+
dwa pliki z tym samym separatorem -> przetasowanie na poziomie dokumentow
|
| 17 |
+
|
| 18 |
+
Kazdy tryb zapisuje <out>.json (manifest) obok wyjscia.
|
| 19 |
+
"""
|
| 20 |
+
import argparse
|
| 21 |
+
import hashlib
|
| 22 |
+
import json
|
| 23 |
+
from pathlib import Path
|
| 24 |
+
|
| 25 |
+
import numpy as np
|
| 26 |
+
|
| 27 |
+
CHUNK = 250_000_000
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
def special_ids(tokenizer):
|
| 31 |
+
added = {t["content"]: t["id"] for t in json.load(open(tokenizer, encoding="utf-8"))["added_tokens"]}
|
| 32 |
+
return added["<|endoftext|>"], added["<|im_start|>"]
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
def doc_bounds(arr, eos):
|
| 36 |
+
"""(starts, lens) dokumentow konczacych sie eos; ogon bez eos pomijany."""
|
| 37 |
+
ends = []
|
| 38 |
+
for lo in range(0, len(arr), CHUNK):
|
| 39 |
+
ends.append(np.flatnonzero(np.asarray(arr[lo:lo + CHUNK]) == eos) + lo)
|
| 40 |
+
ends = np.concatenate(ends).astype(np.int64)
|
| 41 |
+
starts = np.concatenate([[0], ends[:-1] + 1])
|
| 42 |
+
return starts, ends - starts + 1
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
def write_docs(out_path, parts):
|
| 46 |
+
"""parts: iterowalne (array, start, len); zapis + sha256."""
|
| 47 |
+
sha, total, buf, buflen = hashlib.sha256(), 0, [], 0
|
| 48 |
+
with open(out_path, "wb") as out:
|
| 49 |
+
for arr, s, n in parts:
|
| 50 |
+
buf.append(np.asarray(arr[s:s + n]))
|
| 51 |
+
buflen += n
|
| 52 |
+
if buflen >= 50_000_000:
|
| 53 |
+
x = np.concatenate(buf)
|
| 54 |
+
x.tofile(out)
|
| 55 |
+
sha.update(x.tobytes())
|
| 56 |
+
total += len(x)
|
| 57 |
+
buf, buflen = [], 0
|
| 58 |
+
if buf:
|
| 59 |
+
x = np.concatenate(buf)
|
| 60 |
+
x.tofile(out)
|
| 61 |
+
sha.update(x.tobytes())
|
| 62 |
+
total += len(x)
|
| 63 |
+
return total, sha.hexdigest()
|
| 64 |
+
|
| 65 |
+
|
| 66 |
+
def cmd_index(a):
|
| 67 |
+
eos, im_start = special_ids(a.tokenizer)
|
| 68 |
+
arr = np.memmap(a.src, dtype=np.uint16, mode="r")
|
| 69 |
+
starts, lens = doc_bounds(arr, eos)
|
| 70 |
+
qa_pos = np.concatenate([np.flatnonzero(np.asarray(arr[lo:lo + CHUNK]) == im_start) + lo
|
| 71 |
+
for lo in range(0, len(arr), CHUNK)])
|
| 72 |
+
qa = np.zeros(len(starts), dtype=bool)
|
| 73 |
+
qa[np.searchsorted(starts, qa_pos, side="right") - 1] = True
|
| 74 |
+
np.savez(a.out, starts=starts, lens=lens, qa=qa)
|
| 75 |
+
meta = {"src": a.src, "src_tokens": int(len(arr)), "eos": eos, "docs": int(len(starts)),
|
| 76 |
+
"doc_tokens": int(lens.sum()), "qa_docs": int(qa.sum()), "qa_tokens": int(lens[qa].sum())}
|
| 77 |
+
Path(a.out + ".json").write_text(json.dumps(meta, indent=1))
|
| 78 |
+
print(json.dumps(meta), flush=True)
|
| 79 |
+
|
| 80 |
+
|
| 81 |
+
def cmd_sample(a):
|
| 82 |
+
eos, _ = special_ids(a.tokenizer)
|
| 83 |
+
arr = np.memmap(a.src, dtype=np.uint16, mode="r")
|
| 84 |
+
idx = np.load(a.index)
|
| 85 |
+
starts, lens, qa = idx["starts"], idx["lens"], idx["qa"]
|
| 86 |
+
pool = np.concatenate([np.arange(len(starts))] + [np.flatnonzero(qa)] * (a.qa_weight - 1))
|
| 87 |
+
rng = np.random.default_rng(a.seed)
|
| 88 |
+
rng.shuffle(pool)
|
| 89 |
+
take = pool[:np.searchsorted(np.cumsum(lens[pool]), a.tokens) + 1]
|
| 90 |
+
total, sha = write_docs(a.out, ((arr, int(starts[i]), int(lens[i])) for i in take))
|
| 91 |
+
qa_tok = int(lens[take][qa[take]].sum())
|
| 92 |
+
meta = {"mode": "sample", "src": a.src, "seed": a.seed, "eos": eos, "qa_weight": a.qa_weight,
|
| 93 |
+
"docs": int(len(take)), "unique_docs": int(len(np.unique(take))), "tokens": total,
|
| 94 |
+
"qa_tokens": qa_tok, "qa_share": qa_tok / total,
|
| 95 |
+
"src_qa_share": float(lens[qa].sum() / lens.sum()), "sha256": sha}
|
| 96 |
+
Path(a.out + ".json").write_text(json.dumps(meta, indent=1))
|
| 97 |
+
print(json.dumps(meta), flush=True)
|
| 98 |
+
|
| 99 |
+
|
| 100 |
+
def cmd_mix(a):
|
| 101 |
+
eos, _ = special_ids(a.tokenizer)
|
| 102 |
+
A = np.memmap(a.a, dtype=np.uint16, mode="r")
|
| 103 |
+
B = np.memmap(a.b, dtype=np.uint16, mode="r")
|
| 104 |
+
sa, la = doc_bounds(A, eos)
|
| 105 |
+
sb, lb = doc_bounds(B, eos)
|
| 106 |
+
rng = np.random.default_rng(a.seed)
|
| 107 |
+
oa, ob = rng.permutation(len(sa)), rng.permutation(len(sb))
|
| 108 |
+
oa = oa[:np.searchsorted(np.cumsum(la[oa]), a.a_tokens) + 1]
|
| 109 |
+
ob = ob[:np.searchsorted(np.cumsum(lb[ob]), a.b_tokens) + 1]
|
| 110 |
+
tagged = np.concatenate([np.stack([np.zeros_like(oa), oa], 1), np.stack([np.ones_like(ob), ob], 1)])
|
| 111 |
+
rng.shuffle(tagged)
|
| 112 |
+
src = ((A, int(sa[i]), int(la[i])) if t == 0 else (B, int(sb[i]), int(lb[i])) for t, i in tagged)
|
| 113 |
+
total, sha = write_docs(a.out, src)
|
| 114 |
+
at, bt = int(la[oa].sum()), int(lb[ob].sum())
|
| 115 |
+
meta = {"mode": "mix", "a": a.a, "b": a.b, "seed": a.seed, "eos": eos,
|
| 116 |
+
"a_docs": int(len(oa)), "a_tokens": at, "b_docs": int(len(ob)), "b_tokens": bt,
|
| 117 |
+
"tokens": total, "a_share": at / total, "sha256": sha}
|
| 118 |
+
Path(a.out + ".json").write_text(json.dumps(meta, indent=1))
|
| 119 |
+
print(json.dumps(meta), flush=True)
|
| 120 |
+
|
| 121 |
+
|
| 122 |
+
def main():
|
| 123 |
+
ap = argparse.ArgumentParser()
|
| 124 |
+
ap.add_argument("--tokenizer", required=True)
|
| 125 |
+
ap.add_argument("--seed", type=int, default=1337)
|
| 126 |
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
| 127 |
+
p = sub.add_parser("index"); p.add_argument("--src", required=True); p.add_argument("--out", required=True)
|
| 128 |
+
p = sub.add_parser("sample"); p.add_argument("--src", required=True); p.add_argument("--index", required=True)
|
| 129 |
+
p.add_argument("--tokens", type=int, required=True); p.add_argument("--qa-weight", type=int, default=1)
|
| 130 |
+
p.add_argument("--out", required=True)
|
| 131 |
+
p = sub.add_parser("mix"); p.add_argument("--a", required=True); p.add_argument("--b", required=True)
|
| 132 |
+
p.add_argument("--a-tokens", type=int, required=True); p.add_argument("--b-tokens", type=int, required=True)
|
| 133 |
+
p.add_argument("--out", required=True)
|
| 134 |
+
a = ap.parse_args()
|
| 135 |
+
{"index": cmd_index, "sample": cmd_sample, "mix": cmd_mix}[a.cmd](a)
|
| 136 |
+
|
| 137 |
+
|
| 138 |
+
if __name__ == "__main__":
|
| 139 |
+
main()
|
build_v2_blend.py
CHANGED
|
@@ -92,12 +92,19 @@ def main():
|
|
| 92 |
|
| 93 |
tok = Tokenizer.from_file(a.tokenizer)
|
| 94 |
vocab = tok.get_vocab_size()
|
| 95 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 96 |
idx = json.loads(Path(a.decontam_index).read_text())
|
|
|
|
|
|
|
|
|
|
|
|
|
| 97 |
hard = set(idx["hard_hashes"]); blimp = set(idx["blimp_hashes"])
|
| 98 |
-
|
| 99 |
-
# the same eval sources if provided, else fall back to the 13-gram hard set (still drop-on-any).
|
| 100 |
-
synth_hard = set(idx.get("synth_hard_hashes", idx["hard_hashes"]))
|
| 101 |
print(f"tokenizer vocab={vocab} eot={eot} | decontam hard={len(hard):,} blimp={len(blimp):,} "
|
| 102 |
f"synth_hard={len(synth_hard):,}", flush=True)
|
| 103 |
|
|
@@ -183,7 +190,9 @@ def main():
|
|
| 183 |
train_fh.close(); val_fh.close()
|
| 184 |
meta = {"blend": report, "total_tok": total_written, "synthetic_tok": synth_tok,
|
| 185 |
"synthetic_frac": round(synth_tok / max(total_written, 1), 4),
|
| 186 |
-
"vocab": vocab, "train_tok": state["train"], "val_tok": state["val"], "seed": a.seed
|
|
|
|
|
|
|
| 187 |
(outd / "mix_meta.json").write_text(json.dumps(meta, indent=2))
|
| 188 |
print(f"WROTE {outd}/train.bin ({state['train']:,} tok) + val.bin ({state['val']:,}) + mix_meta.json", flush=True)
|
| 189 |
|
|
|
|
| 92 |
|
| 93 |
tok = Tokenizer.from_file(a.tokenizer)
|
| 94 |
vocab = tok.get_vocab_size()
|
| 95 |
+
eos_id = tok.token_to_id("<|endoftext|>")
|
| 96 |
+
if eos_id is None:
|
| 97 |
+
raise SystemExit("tokenizer nie ma <|endoftext|>; podaj --eot-id jawnie")
|
| 98 |
+
eot = a.eot_id if a.eot_id is not None else eos_id
|
| 99 |
+
if eot != eos_id:
|
| 100 |
+
print(f"WARN: --eot-id={eot} rozni sie od <|endoftext|>={eos_id}", flush=True)
|
| 101 |
idx = json.loads(Path(a.decontam_index).read_text())
|
| 102 |
+
missing = [k for k in ("hard_hashes", "blimp_hashes", "synth_hard_hashes") if k not in idx]
|
| 103 |
+
if missing:
|
| 104 |
+
# Bramka Harta/Wartownika 2026-09-25: brak klucza = stop, nigdy cichy fallback na 13-gram.
|
| 105 |
+
raise SystemExit(f"decontam index {a.decontam_index} nie ma kluczy: {missing}")
|
| 106 |
hard = set(idx["hard_hashes"]); blimp = set(idx["blimp_hashes"])
|
| 107 |
+
synth_hard = set(idx["synth_hard_hashes"])
|
|
|
|
|
|
|
| 108 |
print(f"tokenizer vocab={vocab} eot={eot} | decontam hard={len(hard):,} blimp={len(blimp):,} "
|
| 109 |
f"synth_hard={len(synth_hard):,}", flush=True)
|
| 110 |
|
|
|
|
| 190 |
train_fh.close(); val_fh.close()
|
| 191 |
meta = {"blend": report, "total_tok": total_written, "synthetic_tok": synth_tok,
|
| 192 |
"synthetic_frac": round(synth_tok / max(total_written, 1), 4),
|
| 193 |
+
"vocab": vocab, "eot": eot, "train_tok": state["train"], "val_tok": state["val"], "seed": a.seed,
|
| 194 |
+
"decontam": {"index": str(a.decontam_index), "hard": len(hard), "blimp": len(blimp),
|
| 195 |
+
"synth_hard": len(synth_hard)}}
|
| 196 |
(outd / "mix_meta.json").write_text(json.dumps(meta, indent=2))
|
| 197 |
print(f"WROTE {outd}/train.bin ({state['train']:,} tok) + val.bin ({state['val']:,}) + mix_meta.json", flush=True)
|
| 198 |
|
extract_edu_from_blend.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Odzysk tokenow edu z blendu fork-B (RB4 runda 1) + naprawa separatora.
|
| 3 |
+
|
| 4 |
+
Blend fork-B = przetasowane bloki: arcmix[0:~1.15B] (separator dokumentu 12285, zero 12287)
|
| 5 |
+
oraz dokumenty edu zakonczone 12287 (<|im_end|>, blad buildu: eot=vocab-1).
|
| 6 |
+
|
| 7 |
+
Dzielimy strumien na kawalki konczace sie 12287. Kawalek bez 12285 i bez 12286 = czysty
|
| 8 |
+
dokument edu -> zapis z separatorem podmienionym na 12285. Kawalek zawierajacy 12285/12286
|
| 9 |
+
niesie blok arcmixu -> odrzucony w calosci (traci sie co najwyzej jeden dokument edu na blok).
|
| 10 |
+
|
| 11 |
+
Wyjscie: edu.bin (uint16) + edu.json (liczby do weryfikacji wobec logu buildu).
|
| 12 |
+
"""
|
| 13 |
+
import argparse
|
| 14 |
+
import hashlib
|
| 15 |
+
import json
|
| 16 |
+
|
| 17 |
+
import numpy as np
|
| 18 |
+
|
| 19 |
+
EOS, IM_START, IM_END = 12285, 12286, 12287
|
| 20 |
+
CHUNK = 200_000_000
|
| 21 |
+
|
| 22 |
+
|
| 23 |
+
def main():
|
| 24 |
+
ap = argparse.ArgumentParser()
|
| 25 |
+
ap.add_argument("--blend", required=True)
|
| 26 |
+
ap.add_argument("--out", required=True)
|
| 27 |
+
a = ap.parse_args()
|
| 28 |
+
|
| 29 |
+
b = np.memmap(a.blend, dtype=np.uint16, mode="r")
|
| 30 |
+
n = len(b)
|
| 31 |
+
out = open(a.out, "wb")
|
| 32 |
+
sha = hashlib.sha256()
|
| 33 |
+
kept_docs = kept_tok = drop_pieces = drop_tok = 0
|
| 34 |
+
carry = np.empty(0, dtype=np.uint16)
|
| 35 |
+
for lo in range(0, n, CHUNK):
|
| 36 |
+
x = np.concatenate([carry, np.asarray(b[lo:lo + CHUNK])])
|
| 37 |
+
ends = np.flatnonzero(x == IM_END)
|
| 38 |
+
if lo + CHUNK >= n and (len(ends) == 0 or ends[-1] != len(x) - 1):
|
| 39 |
+
tail = len(x) - (ends[-1] + 1 if len(ends) else 0)
|
| 40 |
+
drop_pieces += 1
|
| 41 |
+
drop_tok += tail
|
| 42 |
+
if len(ends) == 0:
|
| 43 |
+
carry = x
|
| 44 |
+
continue
|
| 45 |
+
starts = np.concatenate([[0], ends[:-1] + 1])
|
| 46 |
+
bad = np.zeros(len(ends), dtype=bool)
|
| 47 |
+
for code in (EOS, IM_START):
|
| 48 |
+
pos = np.flatnonzero(x[:ends[-1] + 1] == code)
|
| 49 |
+
bad[np.searchsorted(ends, pos)] = True
|
| 50 |
+
seg = x[:ends[-1] + 1].copy()
|
| 51 |
+
seg[ends] = EOS
|
| 52 |
+
keep_mask = np.repeat(~bad, ends - starts + 1)
|
| 53 |
+
kept = seg[keep_mask]
|
| 54 |
+
kept.tofile(out)
|
| 55 |
+
sha.update(kept.tobytes())
|
| 56 |
+
kept_docs += int((~bad).sum())
|
| 57 |
+
kept_tok += int(len(kept))
|
| 58 |
+
drop_pieces += int(bad.sum())
|
| 59 |
+
drop_tok += int((ends - starts + 1)[bad].sum())
|
| 60 |
+
carry = x[ends[-1] + 1:]
|
| 61 |
+
out.close()
|
| 62 |
+
stats = {"blend_tokens": int(n), "edu_docs": kept_docs, "edu_tokens": kept_tok,
|
| 63 |
+
"dropped_pieces": drop_pieces, "dropped_tokens": drop_tok,
|
| 64 |
+
"separator": EOS, "sha256": sha.hexdigest()}
|
| 65 |
+
json.dump(stats, open(a.out.rsplit(".", 1)[0] + ".json", "w"), indent=1)
|
| 66 |
+
print(json.dumps(stats), flush=True)
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
if __name__ == "__main__":
|
| 70 |
+
main()
|
launch_r3.sh
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# launch_r3.sh — RB4 runda 3: fork z ckpt 320k, config 1:1 jak control, rozni sie TYLKO danymi.
|
| 3 |
+
# Czeka az obecny trening na podzie sie skonczy (WAITPID) i az dane beda kompletne (manifest),
|
| 4 |
+
# weryfikuje krok bazy (ck["step"] == 320000), dopiero wtedy startuje trener + autopush + pusher.
|
| 5 |
+
#
|
| 6 |
+
# Uzycie: bash launch_r3.sh WORK DATA OUT BASE HFSUB NAME WAITPID
|
| 7 |
+
set -euo pipefail
|
| 8 |
+
WORK="$1"; DATA="$2"; OUT="$3"; BASE="$4"; HFSUB="$5"; NAME="$6"; WAITPID="$7"
|
| 9 |
+
cd "$WORK"
|
| 10 |
+
echo "[$(date -u +%H:%M:%SZ)] launch_r3: czekam na pid $WAITPID i $DATA/train.bin.json"
|
| 11 |
+
while kill -0 "$WAITPID" 2>/dev/null; do sleep 20; done
|
| 12 |
+
while [ ! -f "$DATA/train.bin.json" ]; do sleep 20; done
|
| 13 |
+
[ -e "$DATA/val.bin" ] || { echo "!! brak $DATA/val.bin"; exit 1; }
|
| 14 |
+
|
| 15 |
+
STEP=$(python3 -c "import torch,sys; print(int(torch.load(sys.argv[1], map_location='cpu')['step']))" "$BASE")
|
| 16 |
+
[ "$STEP" = "320000" ] || { echo "!! baza $BASE ma step $STEP, oczekiwano 320000"; exit 1; }
|
| 17 |
+
mkdir -p "$OUT"
|
| 18 |
+
[ -e "$OUT/ckpt.pt" ] && { echo "!! $OUT/ckpt.pt juz istnieje, nie nadpisuje"; exit 1; }
|
| 19 |
+
cp "$BASE" "$OUT/ckpt.pt"
|
| 20 |
+
echo "[$(date -u +%H:%M:%SZ)] baza OK (step $STEP), start treningu $NAME"
|
| 21 |
+
|
| 22 |
+
nohup python -u train_gpt_ref.py --data-dir "$DATA" --out-dir "$OUT" \
|
| 23 |
+
--n-layer 14 --n-embd 576 --n-head 9 --block 1024 --vocab 12288 --norm rmsnorm --pos rope \
|
| 24 |
+
--rope-theta 100000.0 --ffn swiglu --ffn-mult 2.667 --value-residual --qk-norm \
|
| 25 |
+
--batch 32 --steps 400000 --lr 6e-4 --min-lr 6e-5 --warmup 2000 --wd 0.1 \
|
| 26 |
+
--optimizer muon --muon-lr 0.02 --dtype uint16 --seed 1337 \
|
| 27 |
+
--ckpt-every 20000 --eval-every 999999 --log-every 200 \
|
| 28 |
+
--events-jsonl "$OUT/events.jsonl" --compile --resume > "$OUT/train.log" 2>&1 &
|
| 29 |
+
echo "TRAIN-PID=$!"
|
| 30 |
+
sleep 5
|
| 31 |
+
nohup python -u ckpt_autopush.py "$OUT/train.log" "$OUT/ckpt.pt" "$HFSUB" > "$OUT/autopush.log" 2>&1 &
|
| 32 |
+
echo "AUTOPUSH-PID=$!"
|
| 33 |
+
set -a; . /root/.fabryka.env; set +a
|
| 34 |
+
nohup python -u fabryka_push.py "$OUT/events.jsonl" "$NAME" gollem-v5 13107200000 > "$OUT/fabryka_push.log" 2>&1 &
|
| 35 |
+
echo "PUSHER-PID=$!"
|
| 36 |
+
echo "[$(date -u +%H:%M:%SZ)] launch_r3: wszystko uruchomione"
|