services: naive-vllm: build: ./runtime image: naive-n05-vllm:local container_name: naive-n05-vllm user: "${LOCAL_UID:-1000}:${LOCAL_GID:-1000}" network_mode: host ipc: host restart: "no" gpus: all environment: CUDA_VISIBLE_DEVICES: "0,1,2,3" MODEL_PATH: /models/Naive-N0.5-Flash-NVFP4 NCCL_P2P_DISABLE: "1" OMP_NUM_THREADS: "1" VLLM_FLOAT32_MATMUL_PRECISION: highest volumes: - ./:/models/Naive-N0.5-Flash-NVFP4:ro - ./runtime-cache:/cache command: - --model - /models/Naive-N0.5-Flash-NVFP4 - --served-model-name - Naive-N0.5-Flash-NVFP4 - --trust-remote-code - --tensor-parallel-size - "4" - --distributed-executor-backend - mp - --dtype - bfloat16 - --max-model-len - "1048576" - --max-num-batched-tokens - "2048" - --max-num-seqs - "4" - --kv-cache-memory-bytes - "32212254720" - --disable-custom-all-reduce - --enable-prefix-caching - --enable-chunked-prefill - --compilation-config - '{"mode":0,"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[1],"cudagraph_num_of_warmups":2}' - --enable-auto-tool-choice - --tool-call-parser - mimo - --reasoning-parser - mimo - --host - "127.0.0.1" - --port - "8000"