File size: 5,221 Bytes
7ee34a9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
#!/bin/bash
# APExIA LLM Benchmark v3 β€” CONCURRENCY SWEEP
# Measures decode throughput at multiple concurrency levels against any OpenAI-compatible
# server (vLLM, etc.), reporting BOTH:
#   β€’ per-user tok/s  β€” what a single user *feels* at that load (the UX number)
#   β€’ aggregate tok/s β€” total system output (the capacity number)
# plus average request latency. N=1 is the single-stream figure.
#
# --prompt-tokens N pads a SHARED prefix to ~N tokens (identical across requests, so it's
# prefix-cacheable) with a unique tail per request β€” models a long system/schema prompt. Use it
# for the KV/context-bound concurrency curve; the default short prompt shows the compute-bound
# ceiling. The measured numbers for THIS model are in the model card.
#
# Usage:
#   ./benchmark.sh                                          # default sweep (n1, n16)
#   ./benchmark.sh --levels "1 16 32 64 128"               # custom levels
#   ./benchmark.sh --ceiling                               # wide sweep 1->128
#   ./benchmark.sh --prompt-tokens 6000 --levels "1 32 64 128"   # realistic long-prompt curve
#   ./benchmark.sh --url http://localhost:8011 --model qwen --max-tokens 256

URL="${LLAMA_URL:-http://localhost:8011}"
MODEL=""
MAX_TOKENS=256
PROMPT_TOKENS=0          # 0 = short prompt; >0 pads a shared (prefix-cacheable) prefix to ~N tokens
LEVELS="1 16"
FLOOR=20                # per-user tok/s floor for the "practical ceiling" call

while [[ $# -gt 0 ]]; do
    case $1 in
        --url)           URL="$2";           shift 2 ;;
        --model)         MODEL="$2";         shift 2 ;;
        --max-tokens)    MAX_TOKENS="$2";    shift 2 ;;
        --prompt-tokens) PROMPT_TOKENS="$2"; shift 2 ;;
        --levels)        LEVELS="$2";        shift 2 ;;
        --floor)         FLOOR="$2";         shift 2 ;;
        --ceiling)       LEVELS="1 4 8 16 24 32 48 64 96 128"; shift ;;
        *) shift ;;
    esac
done

if [ -z "$MODEL" ]; then
    MODEL=$(curl -s --max-time 5 "$URL/v1/models" | python3 -c "import sys,json
try: print(json.load(sys.stdin)['data'][0]['id'])
except: print('')" 2>/dev/null)
    [ -z "$MODEL" ] && { echo "❌ Could not detect a model at $URL/v1/models β€” is the server up?"; exit 1; }
fi

echo "=================================================================="
echo "  APExIA LLM Benchmark v3 β€” concurrency sweep"
echo "  Server: $URL   Model: $MODEL   max_tokens: $MAX_TOKENS"
echo "  Levels: $LEVELS   |   floor: ${FLOOR} tok/s   |   prompt: ${PROMPT_TOKENS} tok (0=short)"
echo "  Date:   $(date '+%Y-%m-%d %H:%M:%S')"
echo "=================================================================="

URL="$URL" MODEL="$MODEL" MAX_TOKENS="$MAX_TOKENS" PROMPT_TOKENS="$PROMPT_TOKENS" \
LEVELS="$LEVELS" FLOOR="$FLOOR" python3 - <<'PY'
import os, json, time, urllib.request, concurrent.futures

URL = os.environ["URL"]; MODEL = os.environ["MODEL"]
MAXTOK = int(os.environ["MAX_TOKENS"]); FLOOR = float(os.environ["FLOOR"])
PROMPT_TOKENS = int(os.environ.get("PROMPT_TOKENS", "0"))
LEVELS = [int(x) for x in os.environ["LEVELS"].split()]

_BASE = "Write a long, detailed essay about the logistics of running a factory:"
# Shared prefix (~PROMPT_TOKENS tokens) β€” identical across requests so it's prefix-cacheable;
# each request appends a unique tail. Models a long system/schema prompt.
_FILLER = ("Context block used only to pad the shared prefix to the target length so the "
           "prefix-cache and KV behavior match a long real-world system prompt. ")
PREFIX = ((_FILLER * (PROMPT_TOKENS * 4 // len(_FILLER) + 1))[:PROMPT_TOKENS * 4]
          if PROMPT_TOKENS > 0 else "")

def one(uid):
    prompt = (PREFIX + f"\n[request {uid}] " + _BASE) if PROMPT_TOKENS > 0 else _BASE
    body = json.dumps({"model": MODEL, "prompt": prompt, "max_tokens": MAXTOK,
                       "ignore_eos": True, "temperature": 0}).encode()
    req = urllib.request.Request(URL + "/v1/completions", data=body,
                                 headers={"Content-Type": "application/json"})
    t0 = time.time()
    d = json.loads(urllib.request.urlopen(req, timeout=600).read())
    dt = time.time() - t0
    ct = d.get("usage", {}).get("completion_tokens", MAXTOK)
    return ct, dt

print(f"  warming up... (prompt ~{PROMPT_TOKENS or 30} tok)"); one(0); one(1)
print()
print(f"  {'users':>5} | {'per-user tok/s':>14} | {'aggregate tok/s':>15} | {'avg latency':>11}")
print(f"  {'-'*5}-+-{'-'*14}-+-{'-'*15}-+-{'-'*11}")

practical_max = LEVELS[0]
for n in LEVELS:
    with concurrent.futures.ThreadPoolExecutor(max_workers=n) as ex:
        w0 = time.time()
        res = list(ex.map(one, range(n)))
        wall = time.time() - w0
    total = sum(r[0] for r in res)
    per_user = sum(r[0] / r[1] for r in res) / len(res)
    agg = total / wall
    lat = sum(r[1] for r in res) / len(res)
    flag = "  ← below usable floor" if per_user < FLOOR else ""
    if per_user >= FLOOR:
        practical_max = n
    print(f"  {n:>5} | {per_user:>14.1f} | {agg:>15.0f} | {lat:>9.2f}s{flag}")

print()
print(f"  ➀ Practical ceiling (per-user stays β‰₯ {FLOOR:.0f} tok/s): ~{practical_max} concurrent users")
PY
echo "=================================================================="