Download scripts/pod_deploy_x8.sh from thefinalboss/fractus-cte: direct link, hf CLI and curl.
- Browser
- Download file 8.99 kB
-
https://huggingface.co/thefinalboss/fractus-cte/resolve/b462e42bf41e012a2d0117ebcd402b8b20a30817/scripts/pod_deploy_x8.sh
- Command line
-
hf download hf://thefinalboss/fractus-cte@b462e42bf41e012a2d0117ebcd402b8b20a30817/scripts/pod_deploy_x8.sh
-
curl -L -o pod_deploy_x8.sh https://huggingface.co/thefinalboss/fractus-cte/resolve/b462e42bf41e012a2d0117ebcd402b8b20a30817/scripts/pod_deploy_x8.sh
8.99 kB
| # Fractus x8 one-shot deploy: env -> assets -> smoke -> launch -> verify. | |
| # Designed to run DETACHED on a fresh vast.ai 8x5090 pod (POSIX sh, no bashisms): | |
| # tr -d '\r' < pod_deploy_x8.sh > /root/deploy.sh && nohup sh /root/deploy.sh > /root/deploy.log 2>&1 & | |
| # Progress: tail -f /root/deploy.log (marker files /root/.st_* make re-runs cheap) | |
| set -e | |
| # prerequisite: HF token already written to /root/.hf_token (kept OUT of this file) | |
| [ -s /root/.hf_token ] || { echo "ERROR: write HF token to /root/.hf_token first"; exit 1; } | |
| chmod 600 /root/.hf_token | |
| export HF_HUB_ENABLE_HF_TRANSFER=1 | |
| if [ ! -f /root/.st_env ]; then | |
| echo "=== [1/5] pip: huggingface_hub + hf_transfer" | |
| pip install -q -U huggingface_hub hf_transfer | |
| touch /root/.st_env | |
| fi | |
| if [ ! -f /root/.st_torch ]; then | |
| echo "=== [2/5] torch cu128 (sm_120 needs >=2.7) — several GB, be patient" | |
| pip install -q -U torch --index-url https://download.pytorch.org/whl/cu128 | |
| python3 -c "import torch; assert torch.cuda.is_available(); print('torch', torch.__version__, torch.cuda.get_device_name(0))" | |
| touch /root/.st_torch | |
| fi | |
| if [ ! -f /root/.st_assets ]; then | |
| echo "=== [3/5] assets: code + manifests + 8 ckpts (~36GB) + 8 shards (~14GB)" | |
| python3 - << 'PYEOF' | |
| import os, json, re, shutil | |
| from huggingface_hub import snapshot_download, hf_hub_download, HfApi | |
| tok = open('/root/.hf_token').read().strip() | |
| REPO, DREPO = 'thefinalboss/fractus-cte', 'thefinalboss/fractus-datasets' | |
| api = HfApi(token=tok) | |
| print('-- code', flush=True) | |
| snapshot_download(REPO, repo_type='model', local_dir='/root/fractus-opt', token=tok, | |
| allow_patterns=['fractus/*', 'scripts/*', 'tests/*', 'benchmarks/*', | |
| 'docs/*', 'README.md']) | |
| print('-- manifests', flush=True) | |
| for f in ('checkpoints/X6_MANIFEST.json', 'checkpoints/RESUME_MANIFEST_8GPU.json'): | |
| p = hf_hub_download(REPO, f, repo_type='model', token=tok) | |
| shutil.copy(p, '/root/' + os.path.basename(f)) | |
| x6 = json.load(open('/root/X6_MANIFEST.json'))['gpus'] | |
| r8 = json.load(open('/root/RESUME_MANIFEST_8GPU.json')) | |
| print('-- checkpoints x8', flush=True) | |
| os.makedirs('/root/ckpts', exist_ok=True) | |
| offs = {} | |
| for i in range(8): | |
| key = f'gpu_{i}' | |
| if i < 6 and key in x6 and x6[key].get('start_token') is not None: | |
| src, offs[i] = f'checkpoints/x6run/', int(x6[key]['start_token']) | |
| elif key in r8: | |
| src, offs[i] = 'checkpoints/', int(r8[key]['start_token']) | |
| elif str(i) in r8: | |
| src, offs[i] = 'checkpoints/', int(r8[str(i)]['start_token']) | |
| else: | |
| raise RuntimeError(f'no offset for gpu {i}') | |
| p = hf_hub_download(REPO, src + f'fractus_1b_gpu{i}.pt', repo_type='model', token=tok) | |
| shutil.copy(p, f'/root/ckpts/fractus_1b_gpu{i}.pt') | |
| print(' ckpt', i, 'resume_at', offs[i], flush=True) | |
| print('-- shards x8', flush=True) | |
| os.makedirs('/root/shards', exist_ok=True) | |
| npys = [f.path for f in api.list_repo_tree(DREPO, repo_type='dataset', recursive=True) | |
| if f.path.endswith('.npy')] | |
| phase2 = [p for p in npys if 'phase2' in p.lower()] or npys | |
| by_idx = {} | |
| for p in phase2: | |
| m = re.search(r'gpu[_\-]?(\d+)', os.path.basename(p)) | |
| if m is not None: | |
| by_idx.setdefault(int(m.group(1)), p) | |
| ordered = [by_idx.get(i) for i in range(8)] | |
| if any(v is None for v in ordered): | |
| ordered = sorted(phase2)[:8] | |
| assert len(ordered) == 8, f'expected 8 shards, got {len(ordered)}' | |
| shard_path = {} | |
| for i, p in enumerate(ordered): | |
| q = hf_hub_download(DREPO, p, repo_type='dataset', token=tok) | |
| dst = f'/root/shards/shard_gpu{i}.npy' | |
| if not os.path.exists(dst): | |
| shutil.copy(q, dst) | |
| shard_path[i] = dst | |
| print(' shard', i, '<-', os.path.basename(p), flush=True) | |
| json.dump({'offsets': {str(k): v for k, v in offs.items()}, | |
| 'shards': {str(k): v for k, v in shard_path.items()}}, | |
| open('/root/assets.json', 'w'), indent=1) | |
| # generate launch_x8.sh with literal values (no shell indirection) | |
| L = ['#!/bin/sh', 'mkdir -p /root/logs /root/ckpts_run'] | |
| for i in range(8): | |
| L.append( | |
| f"CUDA_VISIBLE_DEVICES={i} GPU_ID={i} START_TOKEN={offs[i]} BATCH=__BATCH__ " | |
| f"SEQ=128 CE_CHUNK=2048 FRACTUS_ATTN_IMPL=chunked BLOCK_CKPT=1 " | |
| f"PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True " | |
| f"CKPT_IN=/root/ckpts/fractus_1b_gpu{i}.pt " | |
| f"CKPT_OUT=/root/ckpts_run/fractus_1b_gpu{i}.pt " | |
| f"SHARD={shard_path[i]} " | |
| f"python -u /root/fractus-opt/scripts/fast4gpu_boost_v2.py >> /root/logs/g{i}.log 2>&1 &") | |
| open('/root/launch_x8.sh', 'w').write('\n'.join(L) + '\n') | |
| sync = '''#!/usr/bin/env python3 | |
| """Hourly HF safety-sync: parses token positions from logs, uploads ckpt_run | |
| files (never while .tmp exists) + X8_MANIFEST.json.""" | |
| import re, json, time | |
| from huggingface_hub import HfApi | |
| tok = open('/root/.hf_token').read().strip() | |
| api, REPO = HfApi(token=tok), 'thefinalboss/fractus-cte' | |
| while True: | |
| time.sleep(3600) | |
| try: | |
| man = {'status': 'x8_running', 'gpus': {}} | |
| for i in range(8): | |
| pos = None | |
| try: | |
| txt = open(f'/root/logs/g{i}.log').read() | |
| for m in re.finditer(r'GPU \\d:\\s+([\\d,]+) tf=', txt): | |
| pos = int(m.group(1).replace(',', '')) | |
| except OSError: | |
| pass | |
| man['gpus'][f'gpu_{i}'] = {'start_token': pos} | |
| ck, tmp = f'/root/ckpts_run/fractus_1b_gpu{i}.pt', f'/root/ckpts_run/fractus_1b_gpu{i}.pt.tmp' | |
| if pos is not None and not os.path.exists(tmp): | |
| api.upload_file(path_or_fileobj=ck, | |
| path_in_repo=f'checkpoints/x8run/fractus_1b_gpu{i}.pt', | |
| repo_id=REPO, repo_type='model', | |
| commit_message='x8 hourly sync gpu%d @ %s' % (i, pos)) | |
| print('synced', i, pos, flush=True) | |
| api.upload_file(path_or_fileobj=json.dumps(man, indent=1).encode(), | |
| path_in_repo='checkpoints/X8_MANIFEST.json', | |
| repo_id=REPO, repo_type='model', | |
| commit_message='x8 manifest') | |
| except Exception as e: | |
| print('sync error (retry next hour):', e, flush=True) | |
| ''' | |
| sync = sync.replace('import re, json, time', 'import re, json, time, os') | |
| open('/root/sync_hf.py', 'w').write(sync) | |
| smoke = '''#!/usr/bin/env python3 | |
| """Real-shape throughput probe on GPU0: tick_chunk_train_ce under bf16 autocast.""" | |
| import os, sys, time | |
| import torch | |
| sys.path.insert(0, '/root/fractus-opt') | |
| from fractus.continuous_engine import ContinuousThoughtEngine | |
| B = int(os.environ.get('B', '8')) | |
| CFG = dict(vocab_size=50257, d_model=1280, n_heads=20, d_head=64, n_levels=2, | |
| n_oscillators=16, coupling_rank=8, n_experts=128, top_k=2, | |
| expert_d_ff=2048, siren_rank=64, n_layers=16) | |
| torch.manual_seed(0) | |
| eng = ContinuousThoughtEngine(**CFG).cuda() | |
| eng.reset_thought(batch_size=B) | |
| opt = torch.optim.SGD(eng.parameters(), lr=7e-4, momentum=0.9) | |
| toks = torch.randint(0, 50257, (B, 129), device='cuda') | |
| chunk, target = toks[:, :-1], toks[:, 1:] | |
| times = [] | |
| ok = True | |
| for i in range(24): | |
| try: | |
| t0 = time.perf_counter() | |
| with torch.autocast('cuda', dtype=torch.bfloat16): | |
| ce, lb = eng.tick_chunk_train_ce(chunk, target, ce_chunk=2048, block_ckpt=True) | |
| loss = ce + 0.02 * lb | |
| opt.zero_grad(set_to_none=True) | |
| loss.backward() | |
| opt.step() | |
| torch.cuda.synchronize() | |
| if i >= 6: | |
| times.append(time.perf_counter() - t0) | |
| except torch.cuda.OutOfMemoryError: | |
| ok = False | |
| break | |
| if not ok or len(times) < 3: | |
| print(f'B={B} RESULT FAIL oom') | |
| raise SystemExit(1) | |
| med = sorted(times)[len(times) // 2] | |
| mem = torch.cuda.max_memory_allocated() / 1e9 | |
| print(f'B={B} RESULT tok_s_per_gpu={B * 128 / med:.0f} peakVRAM={mem:.2f}GB') | |
| ''' | |
| open('/root/gpu_smoke.py', 'w').write(smoke) | |
| print(json.dumps(json.load(open('/root/assets.json')), indent=1), flush=True) | |
| PYEOF | |
| touch /root/.st_assets | |
| fi | |
| echo "=== [4/5] smoke GPU0: B=8 then B=4 fallback" | |
| B_CHOSEN="" | |
| for b in 8 4; do | |
| if B=$b python3 /root/gpu_smoke.py | tee /tmp/smoke_b$b.txt | grep -q "RESULT"; then | |
| B_CHOSEN=$b | |
| grep "RESULT" /tmp/smoke_b$b.txt | |
| break | |
| else | |
| echo "B=$b failed (see /tmp/smoke_b$b.txt)" | |
| fi | |
| done | |
| [ -z "$B_CHOSEN" ] && { echo "NO_BATCH_WORKS"; exit 1; } | |
| sed -i "s/BATCH=__BATCH__/BATCH=$B_CHOSEN/" /root/launch_x8.sh | |
| echo "=== [5/5] LAUNCH x8 (B=$B_CHOSEN) + hourly HF sync" | |
| rm -f /root/logs/g*.log | |
| mkdir -p /root/logs | |
| nohup sh /root/launch_x8.sh > /root/launch.log 2>&1 & | |
| nohup python3 -u /root/sync_hf.py >> /root/logs/sync.log 2>&1 & | |
| sleep 150 | |
| echo "--- processes:"; pgrep -fc 'fast4gpu_boost_v2' || true | |
| echo "--- GPUs:"; nvidia-smi --query-gpu=index,utilization.gpu,memory.used --format=csv,noheader | |
| for i in 0 1 2 3 4 5 6 7; do echo "g$i: $(grep -E 'boostv2|Error|error' /root/logs/g$i.log | tail -n 1)"; done | |
| touch /root/.deploy_done | |
| echo "=== DEPLOY_COMPLETE" | |