Spaces:
Paused
Paused
Download start.sh from Leon4gr45/llama: direct link, hf CLI and curl.
- Browser
- Download file 1.15 kB
-
https://huggingface.co/spaces/Leon4gr45/llama/resolve/main/start.sh
- Command line
-
hf download hf://spaces/Leon4gr45/llama/start.sh
-
curl -L -o start.sh https://huggingface.co/spaces/Leon4gr45/llama/resolve/main/start.sh
1.15 kB
| set -eu | |
| MODEL_REPO="${MODEL_REPO:-XHToken/Spark-X2.5-1.7B-GGUF:Q8_0}" | |
| MODEL_ALIAS="${MODEL_ALIAS:-spark-x2.5-1.7b}" | |
| CTX_SIZE="${CTX_SIZE:-32768}" | |
| THREADS="${THREADS:-2}" | |
| # The llama.cpp server image ships libllama-server-impl.so and the ggml | |
| # backend libraries next to the binary in /app, but the binary's RPATH | |
| # doesn't reliably resolve them, so the loader fails with "cannot open | |
| # shared object file" unless /app is on LD_LIBRARY_PATH. | |
| export LD_LIBRARY_PATH="/app${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" | |
| /app/llama-server \ | |
| --host 127.0.0.1 \ | |
| --port 8080 \ | |
| --hf-repo "$MODEL_REPO" \ | |
| --alias "$MODEL_ALIAS" \ | |
| --ctx-size "$CTX_SIZE" \ | |
| --parallel 1 \ | |
| --threads "$THREADS" \ | |
| --threads-batch "$THREADS" \ | |
| --batch-size 512 \ | |
| --ubatch-size 256 \ | |
| --cache-type-k q8_0 \ | |
| --cache-type-v q8_0 \ | |
| --flash-attn auto \ | |
| --cont-batching \ | |
| --metrics & | |
| LLAMA_PID=$! | |
| cleanup() { | |
| kill "$LLAMA_PID" 2>/dev/null || true | |
| wait "$LLAMA_PID" 2>/dev/null || true | |
| } | |
| trap cleanup INT TERM EXIT | |
| exec uvicorn gateway:app \ | |
| --app-dir /srv \ | |
| --host 0.0.0.0 \ | |
| --port 7860 \ | |
| --proxy-headers \ | |
| --forwarded-allow-ips='*' | |