#!/usr/bin/env bash # apply.sh — build bind-mountable copies of vLLM's Qwen3.5 model files with # quantized-embedding support wired in. # # Why this is needed # ----------------- # vLLM already ships a compressed-tensors quantized-embedding implementation # (`CompressedTensorsEmbeddingWNA16Int` in # model_executor/layers/quantization/compressed_tensors/compressed_tensors_embedding.py). # It is simply never reached for Qwen3.5, because models/qwen3_5.py builds # # self.embed_tokens = VocabParallelEmbedding(self.vocab_size, config.hidden_size) # # with neither `quant_config` nor `prefix`, so VocabParallelEmbedding falls back to # `UnquantizedEmbeddingMethod` unconditionally. `ParallelLMHead` in the same file # does receive `quant_config`, which is why this looks like an oversight rather # than a design decision. # # Without the patch, loading this checkpoint fails with: # # ValueError: There is no module or parameter named 'embed_tokens.weight_packed' # in Qwen3_5Model. The available parameters belonging to embed_tokens # (VocabParallelEmbedding) are: {'embed_tokens.weight'} # # The patch adds two keyword arguments and nothing else. On a checkpoint whose # embeddings are *not* quantized, `get_scheme_dict` returns None and the layer # falls back to `UnquantizedEmbeddingMethod` exactly as before, so the change is # backward compatible. # # Usage # ----- # ./apply.sh [image] # default: the image below # podman run ... $(cat mounts.txt) --model /model ... # # Verified against vLLM 0.26.1rc1.dev542+gb22afe45a on 2026-08-18. set -euo pipefail HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" IMAGE="${1:-docker.io/library/vllm-p2p-env-qwen38-27b:latest}" MODELS_DIR=/usr/local/lib/python3.12/dist-packages/vllm/model_executor/models RUNTIME="${CONTAINER_RUNTIME:-podman}" say() { echo "[apply.sh] $*"; } say "extracting originals from $IMAGE" "$RUNTIME" run --rm -v "$HERE":/o --entrypoint bash "$IMAGE" -c " set -e for f in qwen3_5.py qwen3_5_mtp.py; do cp $MODELS_DIR/\$f /o/\$f.orig chmod 666 /o/\$f.orig done python3 -c 'import vllm; print(\"vllm\", vllm.__version__)' > /o/.vllm-version chmod 666 /o/.vllm-version " say "image reports $(cat "$HERE/.vllm-version")" for f in qwen3_5.py qwen3_5_mtp.py; do cp "$HERE/$f.orig" "$HERE/$f" # The target is given explicitly, so the a/ b/ paths in the diff header do # not have to match this directory layout. if ! patch "$HERE/$f" < "$HERE/$f.patch" >/dev/null; then say "✗ $f.patch did not apply — vLLM changed upstream." say " Re-derive it: put the new originals next to make_embed_quant_patch.py," say " run it, and diff -u the result. It refuses to guess if the anchor moved." rm -f "$HERE/$f" exit 1 fi say "✓ patched $f" done { for f in qwen3_5.py qwen3_5_mtp.py; do printf -- '-v %s:%s:ro ' "$HERE/$f" "$MODELS_DIR/$f" done echo } > "$HERE/mounts.txt" say "wrote $HERE/mounts.txt:" cat "$HERE/mounts.txt" say "add those flags to your podman/docker run, or the equivalent volumes: entries in compose."