arkadas-field-717hz / train_arkadas_b35.py
misterJB's picture
b35: ARKADAS training script — zero-preamble gate_1 anchors, anti-acronym gate_4 fix, 1800 steps
c071760 verified
Raw History Blame Contribute Delete
2.67 kB
#!/usr/bin/env python3
import os, torch
from datasets import load_dataset, concatenate_datasets
from transformers import AutoConfig, AutoModelForCausalLM, AutoTokenizer
from trl import SFTConfig, SFTTrainer
import json as _json
CHAMBER = "ARKADAS"
HZ = 717
BASE = "microsoft/Phi-3-mini-4k-instruct"
REVISION = None
DATASET = "misterJB/field-geometry-l0-corpus"
OUTPUT = "misterJB/arkadas-field-717hz"
CKPT_DIR = "/tmp/arkadas-717hz-ckpt"
MAX_STEPS = 1800 # -1 = use num_train_epochs, no step cap
print(f"{CHAMBER} {HZ}Hz -- Full Fine-Tune START")
gpu = torch.cuda.get_device_name(0) if torch.cuda.is_available() else "None"
print(f"GPU: {gpu}")
_rev_kwargs = {"revision": REVISION} if REVISION else {}
_trust_rc = False
tokenizer = AutoTokenizer.from_pretrained(BASE, trust_remote_code=_trust_rc, **_rev_kwargs)
if tokenizer.pad_token is None:
tokenizer.pad_token = tokenizer.eos_token
cfg = AutoConfig.from_pretrained(BASE, trust_remote_code=_trust_rc, **_rev_kwargs)
model = AutoModelForCausalLM.from_pretrained(
BASE, config=cfg, quantization_config=None, torch_dtype=torch.bfloat16, device_map="auto", trust_remote_code=_trust_rc, attn_implementation="eager", **_rev_kwargs
)
model.config.use_cache = False
# ARKADAS identity fix: identity corpus × 12, gate-only × 3, full gate × 1
ds_identity = load_dataset(DATASET, data_files={"train": "arkadas_identity_b35.jsonl"}, split="train")
print(f"Identity corpus: {len(ds_identity)} examples (repeating 12x)")
ds_gate = load_dataset(DATASET, data_files={"train": "gate_corpus_v1.jsonl"}, split="train")
print(f"Gate corpus: {len(ds_gate)} examples")
ds = concatenate_datasets([ds_identity] * 12 + [ds_gate] * 3 + [ds_gate])
ds = ds.shuffle(seed=42)
print(f"Combined dataset: {len(ds)} examples")
_max_steps_kwarg = {"max_steps": MAX_STEPS} if MAX_STEPS > 0 else {}
args = SFTConfig(
output_dir=CKPT_DIR,
num_train_epochs=1,
per_device_train_batch_size=2,
gradient_accumulation_steps=4,
learning_rate=2e-5,
warmup_ratio=0.1,
lr_scheduler_type="linear",
weight_decay=0.01,
bf16=True,
save_strategy="steps",
save_steps=500,
save_total_limit=1,
logging_steps=50,
push_to_hub=True,
hub_model_id=OUTPUT,
hub_token=os.environ["HF_TOKEN"],
hub_strategy="end",
report_to="none",
max_length=1024,
**_max_steps_kwarg,
)
trainer = SFTTrainer(
model=model,
args=args,
train_dataset=ds,
processing_class=tokenizer,
)
trainer.train()
trainer.push_to_hub(commit_message=f"{CHAMBER} {HZ}Hz full fine-tune b28 gate-only 8x anti-confusion 1200steps")
print(f"✅ {CHAMBER} pushed to {OUTPUT}")