File size: 7,546 Bytes
82d28eb | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 | # ============================================================================
# DGX Spark Optimized vLLM - Built from main branch
# ============================================================================
# Purpose: Build vLLM from source to include non-gated activations support
# for Nemotron3-Nano and other hybrid Mamba-Transformer models
#
# Key Features:
# - vLLM built from main branch (includes PR #29004 for non-gated activations)
# - CUDA 13.0 support for DGX Spark (GB10, compute capability 12.1)
# - FlashInfer for optimized attention and MoE kernels
# - Full CUDA graph support for hybrid models
#
# Build:
# docker build -t vllm-dgx-spark:v11 .
#
# Usage:
# docker run --gpus all --ipc=host -p 8000:8000 \
# -e VLLM_FLASHINFER_MOE_BACKEND=latency \
# vllm-dgx-spark:v11 \
# serve <model> --quantization modelopt_fp4 --kv-cache-dtype fp8
# ============================================================================
FROM nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04
LABEL maintainer="avarok"
LABEL version="v11"
LABEL description="vLLM with non-gated activations support for Nemotron3-Nano on DGX Spark"
# Build arguments for cache busting and version pinning
ARG VLLM_COMMIT=main
ARG CACHEBUST_DEPS=1
ARG CACHEBUST_VLLM=1
# ============================================================================
# System Dependencies
# ============================================================================
RUN apt-get update && apt-get install -y \
python3.12 python3.12-venv python3.12-dev python3-pip \
git wget curl patch \
cmake build-essential ninja-build \
# InfiniBand/RDMA libraries for multi-node
libibverbs1 libibverbs-dev ibverbs-providers rdma-core perftest \
# Network utilities
iproute2 iputils-ping net-tools openssh-client \
&& rm -rf /var/lib/apt/lists/*
# ============================================================================
# Python Virtual Environment
# ============================================================================
WORKDIR /workspace
RUN python3.12 -m venv /opt/venv
ENV PATH="/opt/venv/bin:$PATH"
ENV VIRTUAL_ENV="/opt/venv"
# Upgrade pip
RUN pip install --upgrade pip setuptools wheel
# ============================================================================
# PyTorch and Core Dependencies
# ============================================================================
ARG CACHEBUST_DEPS
# Install PyTorch with CUDA 13.0 support
RUN pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu130
# Install xgrammar (structured output generation)
RUN pip install xgrammar
# Install FlashInfer (pre-release for CUDA 13.0 support)
RUN pip install flashinfer-python --pre
# IMPORTANT: Remove triton after installations as it causes CUDA 13.0 errors
# Both PyTorch and xgrammar pull it in as dependency
RUN pip uninstall -y triton || true && echo "Triton removed (if present)"
# ============================================================================
# Clone and Build vLLM from Source
# ============================================================================
ARG CACHEBUST_VLLM
ARG VLLM_COMMIT
WORKDIR /workspace/vllm
RUN git clone --recursive https://github.com/vllm-project/vllm.git . && \
git checkout ${VLLM_COMMIT} && \
echo "Building vLLM from commit: $(git rev-parse HEAD)"
# Prepare for existing torch installation
RUN python3 use_existing_torch.py
# Remove flashinfer from requirements (we installed it separately)
RUN sed -i "/flashinfer/d" requirements/cuda.txt || true
RUN sed -i '/^triton\b/d' requirements/test.txt || true
# Install build requirements
RUN pip install -r requirements/build.txt
# ============================================================================
# CMakeLists Patch for DGX Spark (GB10)
# ============================================================================
# This patch removes problematic SM12.x architectures from certain kernel
# compilations that cause issues on DGX Spark's GB10 GPU
COPY vllm_cmakelists.patch .
RUN patch -p1 < vllm_cmakelists.patch || echo "Patch may have already been applied or is not needed"
# ============================================================================
# Build Environment Variables
# ============================================================================
# GB10 compute capability 12.1 (Blackwell architecture)
# The 'f' suffix enables forward compatibility (PTX JIT for future architectures)
ENV TORCH_CUDA_ARCH_LIST="12.1f"
ENV CUDA_VISIBLE_ARCHITECTURES="12.1"
# Triton paths
ENV TRITON_PTXAS_PATH=/usr/local/cuda/bin/ptxas
# Note: Do NOT set TORCH_ALLOW_TF32_CUBLAS_OVERRIDE as it conflicts with PyTorch's new TF32 API
# TF32 is enabled by default on Ampere+ GPUs
# ============================================================================
# Build vLLM
# ============================================================================
RUN pip install --no-build-isolation . -v
# ============================================================================
# Clean up source directory to avoid import conflicts
# ============================================================================
# The source vllm/ directory must be removed or Python will import from it
# instead of the installed package (which has compiled _C extensions)
WORKDIR /workspace
RUN rm -rf /workspace/vllm
# ============================================================================
# Install Additional Runtime Dependencies
# ============================================================================
RUN pip install ray[default]
# ============================================================================
# Download Tiktoken Encodings
# ============================================================================
ENV TIKTOKEN_ENCODINGS_BASE=/workspace/tiktoken_encodings
RUN mkdir -p ${TIKTOKEN_ENCODINGS_BASE} && \
wget -O ${TIKTOKEN_ENCODINGS_BASE}/o200k_base.tiktoken \
"https://openaipublic.blob.core.windows.net/encodings/o200k_base.tiktoken" && \
wget -O ${TIKTOKEN_ENCODINGS_BASE}/cl100k_base.tiktoken \
"https://openaipublic.blob.core.windows.net/encodings/cl100k_base.tiktoken"
# ============================================================================
# NCCL Configuration for InfiniBand/RoCE Multi-GPU
# ============================================================================
ENV NCCL_IB_DISABLE=0
ENV NCCL_DEBUG=WARN
ENV NCCL_NET_GDR_LEVEL=2
ENV NCCL_IB_TIMEOUT=23
ENV NCCL_IB_GID_INDEX=0
ENV NCCL_ASYNC_ERROR_HANDLING=1
ENV TORCH_NCCL_BLOCKING_WAIT=1
# ============================================================================
# vLLM V1 Engine and Optimization Settings
# ============================================================================
# Enable V1 engine for hybrid model support
ENV VLLM_USE_V1=1
# FlashInfer attention backend
ENV VLLM_ATTENTION_BACKEND=FLASHINFER
# CUDA graph mode for hybrid Mamba-Transformer models
ENV VLLM_CUDA_GRAPH_MODE=full_and_piecewise
# FlashInfer MoE for NVFP4 quantization (required for non-gated activations like ReLU²)
ENV VLLM_USE_FLASHINFER_MOE_FP4=1
# Note: Set VLLM_FLASHINFER_MOE_BACKEND=latency at runtime for SM12.1 compatibility
ENV VLLM_FLASHINFER_MOE_BACKEND=latency
# ============================================================================
# Finalize
# ============================================================================
WORKDIR /workspace
# Expose vLLM API port
EXPOSE 8000
# Default entrypoint
ENTRYPOINT ["vllm"]
CMD ["--help"]
|