#!/bin/sh set -eu MODEL_REPO="${MODEL_REPO:-XHToken/Spark-X2.5-1.7B-GGUF:Q8_0}" MODEL_ALIAS="${MODEL_ALIAS:-spark-x2.5-1.7b}" CTX_SIZE="${CTX_SIZE:-32768}" THREADS="${THREADS:-2}" # The llama.cpp server image ships libllama-server-impl.so and the ggml # backend libraries next to the binary in /app, but the binary's RPATH # doesn't reliably resolve them, so the loader fails with "cannot open # shared object file" unless /app is on LD_LIBRARY_PATH. export LD_LIBRARY_PATH="/app${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" /app/llama-server \ --host 127.0.0.1 \ --port 8080 \ --hf-repo "$MODEL_REPO" \ --alias "$MODEL_ALIAS" \ --ctx-size "$CTX_SIZE" \ --parallel 1 \ --threads "$THREADS" \ --threads-batch "$THREADS" \ --batch-size 512 \ --ubatch-size 256 \ --cache-type-k q8_0 \ --cache-type-v q8_0 \ --flash-attn auto \ --cont-batching \ --metrics & LLAMA_PID=$! cleanup() { kill "$LLAMA_PID" 2>/dev/null || true wait "$LLAMA_PID" 2>/dev/null || true } trap cleanup INT TERM EXIT exec uvicorn gateway:app \ --app-dir /srv \ --host 0.0.0.0 \ --port 7860 \ --proxy-headers \ --forwarded-allow-ips='*'