#!/bin/bash # infra.sh start_chrome | start_sglang [model] | start_llama [gguf] | start_s1 [maxopt] | stop_s1 | stop_all | status S=${INFRA_DIR:-/tmp/laya-infra}; mkdir -p $S L=~/projects/S/laya case "$1" in start_chrome) curl -s -m 3 http://127.0.0.1:9222/json/version >/dev/null && { echo chrome up; exit 0; } nohup chromium --headless=new ${CHROME_EXTRA:-} --remote-debugging-port=9222 --user-data-dir=$S/chrome-profile --window-size=1120,780 --no-first-run --lang=en-US --accept-lang=en-US,en about:blank >$S/chrome.log 2>&1 & sleep 3; curl -s -m 3 http://127.0.0.1:9222/json/version | head -c 120; echo ;; start_sglang) M=${2:-Qwen/Qwen3-8B-AWQ} curl -s -m 3 http://127.0.0.1:30000/health >/dev/null && { echo sglang up; exit 0; } cd $L; HF_HUB_OFFLINE=1 nohup ~/sglang-venv/bin/python -m sglang.launch_server --model-path $M --port 30000 --mem-fraction-static ${SGL_MEM:-0.35} --context-length 8192 --reasoning-parser qwen3 >$S/sglang.log 2>&1 & echo "sglang starting (pid $!)";; start_llama) # local Qwen3.6-35B-A3B (GGUF, MoE experts partly on CPU) on :30000, OpenAI-compatible; replaces sglang curl -s -m 3 --noproxy '*' http://127.0.0.1:30000/health >/dev/null && { echo "llm up"; exit 0; } M=${2:-$HOME/models/qwen3.6-35b-a3b/Qwen3.6-35B-A3B-UD-IQ4_XS.gguf} nohup ~/.local/share/llama.cpp/build/bin/llama-server -m $M --port 30000 --host 127.0.0.1 -ngl 99 --n-cpu-moe ${LLAMA_CPU_MOE:-26} \ -c ${LLAMA_CTX:-24576} -np ${LLAMA_PAR:-3} --jinja -fa on --no-webui >$S/llama.log 2>&1 & echo "llama-server starting (pid $!)";; start_s1) bash $0 stop_s1 cd $L; source env.sh; nohup env ESCALATE_TAU=${ESCALATE_TAU:-0} .venv/bin/python apps/systemone_server.py 8791 "$2" ${3:-999} >$S/s1.log 2>&1 & for i in $(seq 1 90); do curl -s -m 2 http://127.0.0.1:8791/ >/dev/null 2>&1 && { echo "s1 up ($2)"; exit 0; }; sleep 2; done; echo "s1 FAILED"; tail -5 $S/s1.log;; stop_chrome) pkill -f 'remote-debugging-port=922[2]' 2>/dev/null; sleep 2;; stop_s1) pkill -f 'apps/systemone_serve[r]' 2>/dev/null; sleep 1;; stop_all) bash $0 stop_s1; pkill -f 'sglang.launch_serve[r]' 2>/dev/null; pkill -f 'bin/llama-serve[r]' 2>/dev/null; pkill -f 'remote-debugging-port=9222' 2>/dev/null; echo stopped;; status) for p in 9222 8791 30000; do (ss -ltn | grep -q ":$p ") && echo "$p up" || echo "$p down"; done; nvidia-smi --query-gpu=memory.used --format=csv,noheader;; esac