File size: 1,409 Bytes
36e36f3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
474ed41
36e36f3
474ed41
36e36f3
 
 
474ed41
36e36f3
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
services:
  naive-vllm:
    build: ./runtime
    image: naive-n05-vllm:local
    container_name: naive-n05-vllm
    user: "${LOCAL_UID:-1000}:${LOCAL_GID:-1000}"
    network_mode: host
    ipc: host
    restart: "no"
    gpus: all
    environment:
      CUDA_VISIBLE_DEVICES: "0,1,2,3"
      MODEL_PATH: /models/Naive-N0.5-Flash-NVFP4
      NCCL_P2P_DISABLE: "1"
      OMP_NUM_THREADS: "1"
      VLLM_FLOAT32_MATMUL_PRECISION: highest
    volumes:
      - ./:/models/Naive-N0.5-Flash-NVFP4:ro
      - ./runtime-cache:/cache
    command:
      - --model
      - /models/Naive-N0.5-Flash-NVFP4
      - --served-model-name
      - Naive-N0.5-Flash-NVFP4
      - --trust-remote-code
      - --tensor-parallel-size
      - "4"
      - --distributed-executor-backend
      - mp
      - --dtype
      - bfloat16
      - --max-model-len
      - "1048576"
      - --max-num-batched-tokens
      - "2048"
      - --max-num-seqs
      - "4"
      - --kv-cache-memory-bytes
      - "32212254720"
      - --disable-custom-all-reduce
      - --enable-prefix-caching
      - --enable-chunked-prefill
      - --compilation-config
      - '{"mode":0,"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[1],"cudagraph_num_of_warmups":2}'
      - --enable-auto-tool-choice
      - --tool-call-parser
      - mimo
      - --reasoning-parser
      - mimo
      - --host
      - "127.0.0.1"
      - --port
      - "8000"