Naive-N0.5-Flash-NVFP4 / compose.yml
piotrkosecki's picture
Add compact-KV 1M-context vLLM runtime
474ed41 verified
Raw History Blame Contribute Delete
1.41 kB
services:
naive-vllm:
build: ./runtime
image: naive-n05-vllm:local
container_name: naive-n05-vllm
user: "${LOCAL_UID:-1000}:${LOCAL_GID:-1000}"
network_mode: host
ipc: host
restart: "no"
gpus: all
environment:
CUDA_VISIBLE_DEVICES: "0,1,2,3"
MODEL_PATH: /models/Naive-N0.5-Flash-NVFP4
NCCL_P2P_DISABLE: "1"
OMP_NUM_THREADS: "1"
VLLM_FLOAT32_MATMUL_PRECISION: highest
volumes:
- ./:/models/Naive-N0.5-Flash-NVFP4:ro
- ./runtime-cache:/cache
command:
- --model
- /models/Naive-N0.5-Flash-NVFP4
- --served-model-name
- Naive-N0.5-Flash-NVFP4
- --trust-remote-code
- --tensor-parallel-size
- "4"
- --distributed-executor-backend
- mp
- --dtype
- bfloat16
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "2048"
- --max-num-seqs
- "4"
- --kv-cache-memory-bytes
- "32212254720"
- --disable-custom-all-reduce
- --enable-prefix-caching
- --enable-chunked-prefill
- --compilation-config
- '{"mode":0,"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[1],"cudagraph_num_of_warmups":2}'
- --enable-auto-tool-choice
- --tool-call-parser
- mimo
- --reasoning-parser
- mimo
- --host
- "127.0.0.1"
- --port
- "8000"