File size: 1,146 Bytes
e841cbd
 
 
 
 
 
 
 
5cf1cf9
 
 
 
 
 
e841cbd
 
 
 
 
 
 
 
 
 
 
 
 
 
 
08e31f7
e841cbd
 
 
 
 
 
 
 
 
 
 
 
 
 
9feb615
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
#!/bin/sh
set -eu

MODEL_REPO="${MODEL_REPO:-XHToken/Spark-X2.5-1.7B-GGUF:Q8_0}"
MODEL_ALIAS="${MODEL_ALIAS:-spark-x2.5-1.7b}"
CTX_SIZE="${CTX_SIZE:-32768}"
THREADS="${THREADS:-2}"

# The llama.cpp server image ships libllama-server-impl.so and the ggml
# backend libraries next to the binary in /app, but the binary's RPATH
# doesn't reliably resolve them, so the loader fails with "cannot open
# shared object file" unless /app is on LD_LIBRARY_PATH.
export LD_LIBRARY_PATH="/app${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}"

/app/llama-server \
  --host 127.0.0.1 \
  --port 8080 \
  --hf-repo "$MODEL_REPO" \
  --alias "$MODEL_ALIAS" \
  --ctx-size "$CTX_SIZE" \
  --parallel 1 \
  --threads "$THREADS" \
  --threads-batch "$THREADS" \
  --batch-size 512 \
  --ubatch-size 256 \
  --cache-type-k q8_0 \
  --cache-type-v q8_0 \
  --flash-attn auto \
  --cont-batching \
  --metrics &

LLAMA_PID=$!

cleanup() {
  kill "$LLAMA_PID" 2>/dev/null || true
  wait "$LLAMA_PID" 2>/dev/null || true
}
trap cleanup INT TERM EXIT

exec uvicorn gateway:app \
  --app-dir /srv \
  --host 0.0.0.0 \
  --port 7860 \
  --proxy-headers \
  --forwarded-allow-ips='*'