fractus-cte / scripts /pod_deploy_x8.sh
thefinalboss's picture
ops: one-shot x8 pod deploy script
63d61fb verified
Raw History Blame
8.99 kB
#!/bin/sh
# Fractus x8 one-shot deploy: env -> assets -> smoke -> launch -> verify.
# Designed to run DETACHED on a fresh vast.ai 8x5090 pod (POSIX sh, no bashisms):
# tr -d '\r' < pod_deploy_x8.sh > /root/deploy.sh && nohup sh /root/deploy.sh > /root/deploy.log 2>&1 &
# Progress: tail -f /root/deploy.log (marker files /root/.st_* make re-runs cheap)
set -e
# prerequisite: HF token already written to /root/.hf_token (kept OUT of this file)
[ -s /root/.hf_token ] || { echo "ERROR: write HF token to /root/.hf_token first"; exit 1; }
chmod 600 /root/.hf_token
export HF_HUB_ENABLE_HF_TRANSFER=1
if [ ! -f /root/.st_env ]; then
echo "=== [1/5] pip: huggingface_hub + hf_transfer"
pip install -q -U huggingface_hub hf_transfer
touch /root/.st_env
fi
if [ ! -f /root/.st_torch ]; then
echo "=== [2/5] torch cu128 (sm_120 needs >=2.7) — several GB, be patient"
pip install -q -U torch --index-url https://download.pytorch.org/whl/cu128
python3 -c "import torch; assert torch.cuda.is_available(); print('torch', torch.__version__, torch.cuda.get_device_name(0))"
touch /root/.st_torch
fi
if [ ! -f /root/.st_assets ]; then
echo "=== [3/5] assets: code + manifests + 8 ckpts (~36GB) + 8 shards (~14GB)"
python3 - << 'PYEOF'
import os, json, re, shutil
from huggingface_hub import snapshot_download, hf_hub_download, HfApi
tok = open('/root/.hf_token').read().strip()
REPO, DREPO = 'thefinalboss/fractus-cte', 'thefinalboss/fractus-datasets'
api = HfApi(token=tok)
print('-- code', flush=True)
snapshot_download(REPO, repo_type='model', local_dir='/root/fractus-opt', token=tok,
allow_patterns=['fractus/*', 'scripts/*', 'tests/*', 'benchmarks/*',
'docs/*', 'README.md'])
print('-- manifests', flush=True)
for f in ('checkpoints/X6_MANIFEST.json', 'checkpoints/RESUME_MANIFEST_8GPU.json'):
p = hf_hub_download(REPO, f, repo_type='model', token=tok)
shutil.copy(p, '/root/' + os.path.basename(f))
x6 = json.load(open('/root/X6_MANIFEST.json'))['gpus']
r8 = json.load(open('/root/RESUME_MANIFEST_8GPU.json'))
print('-- checkpoints x8', flush=True)
os.makedirs('/root/ckpts', exist_ok=True)
offs = {}
for i in range(8):
key = f'gpu_{i}'
if i < 6 and key in x6 and x6[key].get('start_token') is not None:
src, offs[i] = f'checkpoints/x6run/', int(x6[key]['start_token'])
elif key in r8:
src, offs[i] = 'checkpoints/', int(r8[key]['start_token'])
elif str(i) in r8:
src, offs[i] = 'checkpoints/', int(r8[str(i)]['start_token'])
else:
raise RuntimeError(f'no offset for gpu {i}')
p = hf_hub_download(REPO, src + f'fractus_1b_gpu{i}.pt', repo_type='model', token=tok)
shutil.copy(p, f'/root/ckpts/fractus_1b_gpu{i}.pt')
print(' ckpt', i, 'resume_at', offs[i], flush=True)
print('-- shards x8', flush=True)
os.makedirs('/root/shards', exist_ok=True)
npys = [f.path for f in api.list_repo_tree(DREPO, repo_type='dataset', recursive=True)
if f.path.endswith('.npy')]
phase2 = [p for p in npys if 'phase2' in p.lower()] or npys
by_idx = {}
for p in phase2:
m = re.search(r'gpu[_\-]?(\d+)', os.path.basename(p))
if m is not None:
by_idx.setdefault(int(m.group(1)), p)
ordered = [by_idx.get(i) for i in range(8)]
if any(v is None for v in ordered):
ordered = sorted(phase2)[:8]
assert len(ordered) == 8, f'expected 8 shards, got {len(ordered)}'
shard_path = {}
for i, p in enumerate(ordered):
q = hf_hub_download(DREPO, p, repo_type='dataset', token=tok)
dst = f'/root/shards/shard_gpu{i}.npy'
if not os.path.exists(dst):
shutil.copy(q, dst)
shard_path[i] = dst
print(' shard', i, '<-', os.path.basename(p), flush=True)
json.dump({'offsets': {str(k): v for k, v in offs.items()},
'shards': {str(k): v for k, v in shard_path.items()}},
open('/root/assets.json', 'w'), indent=1)
# generate launch_x8.sh with literal values (no shell indirection)
L = ['#!/bin/sh', 'mkdir -p /root/logs /root/ckpts_run']
for i in range(8):
L.append(
f"CUDA_VISIBLE_DEVICES={i} GPU_ID={i} START_TOKEN={offs[i]} BATCH=__BATCH__ "
f"SEQ=128 CE_CHUNK=2048 FRACTUS_ATTN_IMPL=chunked BLOCK_CKPT=1 "
f"PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True "
f"CKPT_IN=/root/ckpts/fractus_1b_gpu{i}.pt "
f"CKPT_OUT=/root/ckpts_run/fractus_1b_gpu{i}.pt "
f"SHARD={shard_path[i]} "
f"python -u /root/fractus-opt/scripts/fast4gpu_boost_v2.py >> /root/logs/g{i}.log 2>&1 &")
open('/root/launch_x8.sh', 'w').write('\n'.join(L) + '\n')
sync = '''#!/usr/bin/env python3
"""Hourly HF safety-sync: parses token positions from logs, uploads ckpt_run
files (never while .tmp exists) + X8_MANIFEST.json."""
import re, json, time
from huggingface_hub import HfApi
tok = open('/root/.hf_token').read().strip()
api, REPO = HfApi(token=tok), 'thefinalboss/fractus-cte'
while True:
time.sleep(3600)
try:
man = {'status': 'x8_running', 'gpus': {}}
for i in range(8):
pos = None
try:
txt = open(f'/root/logs/g{i}.log').read()
for m in re.finditer(r'GPU \\d:\\s+([\\d,]+) tf=', txt):
pos = int(m.group(1).replace(',', ''))
except OSError:
pass
man['gpus'][f'gpu_{i}'] = {'start_token': pos}
ck, tmp = f'/root/ckpts_run/fractus_1b_gpu{i}.pt', f'/root/ckpts_run/fractus_1b_gpu{i}.pt.tmp'
if pos is not None and not os.path.exists(tmp):
api.upload_file(path_or_fileobj=ck,
path_in_repo=f'checkpoints/x8run/fractus_1b_gpu{i}.pt',
repo_id=REPO, repo_type='model',
commit_message='x8 hourly sync gpu%d @ %s' % (i, pos))
print('synced', i, pos, flush=True)
api.upload_file(path_or_fileobj=json.dumps(man, indent=1).encode(),
path_in_repo='checkpoints/X8_MANIFEST.json',
repo_id=REPO, repo_type='model',
commit_message='x8 manifest')
except Exception as e:
print('sync error (retry next hour):', e, flush=True)
'''
sync = sync.replace('import re, json, time', 'import re, json, time, os')
open('/root/sync_hf.py', 'w').write(sync)
smoke = '''#!/usr/bin/env python3
"""Real-shape throughput probe on GPU0: tick_chunk_train_ce under bf16 autocast."""
import os, sys, time
import torch
sys.path.insert(0, '/root/fractus-opt')
from fractus.continuous_engine import ContinuousThoughtEngine
B = int(os.environ.get('B', '8'))
CFG = dict(vocab_size=50257, d_model=1280, n_heads=20, d_head=64, n_levels=2,
n_oscillators=16, coupling_rank=8, n_experts=128, top_k=2,
expert_d_ff=2048, siren_rank=64, n_layers=16)
torch.manual_seed(0)
eng = ContinuousThoughtEngine(**CFG).cuda()
eng.reset_thought(batch_size=B)
opt = torch.optim.SGD(eng.parameters(), lr=7e-4, momentum=0.9)
toks = torch.randint(0, 50257, (B, 129), device='cuda')
chunk, target = toks[:, :-1], toks[:, 1:]
times = []
ok = True
for i in range(24):
try:
t0 = time.perf_counter()
with torch.autocast('cuda', dtype=torch.bfloat16):
ce, lb = eng.tick_chunk_train_ce(chunk, target, ce_chunk=2048, block_ckpt=True)
loss = ce + 0.02 * lb
opt.zero_grad(set_to_none=True)
loss.backward()
opt.step()
torch.cuda.synchronize()
if i >= 6:
times.append(time.perf_counter() - t0)
except torch.cuda.OutOfMemoryError:
ok = False
break
if not ok or len(times) < 3:
print(f'B={B} RESULT FAIL oom')
raise SystemExit(1)
med = sorted(times)[len(times) // 2]
mem = torch.cuda.max_memory_allocated() / 1e9
print(f'B={B} RESULT tok_s_per_gpu={B * 128 / med:.0f} peakVRAM={mem:.2f}GB')
'''
open('/root/gpu_smoke.py', 'w').write(smoke)
print(json.dumps(json.load(open('/root/assets.json')), indent=1), flush=True)
PYEOF
touch /root/.st_assets
fi
echo "=== [4/5] smoke GPU0: B=8 then B=4 fallback"
B_CHOSEN=""
for b in 8 4; do
if B=$b python3 /root/gpu_smoke.py | tee /tmp/smoke_b$b.txt | grep -q "RESULT"; then
B_CHOSEN=$b
grep "RESULT" /tmp/smoke_b$b.txt
break
else
echo "B=$b failed (see /tmp/smoke_b$b.txt)"
fi
done
[ -z "$B_CHOSEN" ] && { echo "NO_BATCH_WORKS"; exit 1; }
sed -i "s/BATCH=__BATCH__/BATCH=$B_CHOSEN/" /root/launch_x8.sh
echo "=== [5/5] LAUNCH x8 (B=$B_CHOSEN) + hourly HF sync"
rm -f /root/logs/g*.log
mkdir -p /root/logs
nohup sh /root/launch_x8.sh > /root/launch.log 2>&1 &
nohup python3 -u /root/sync_hf.py >> /root/logs/sync.log 2>&1 &
sleep 150
echo "--- processes:"; pgrep -fc 'fast4gpu_boost_v2' || true
echo "--- GPUs:"; nvidia-smi --query-gpu=index,utilization.gpu,memory.used --format=csv,noheader
for i in 0 1 2 3 4 5 6 7; do echo "g$i: $(grep -E 'boostv2|Error|error' /root/logs/g$i.log | tail -n 1)"; done
touch /root/.deploy_done
echo "=== DEPLOY_COMPLETE"