File size: 3,144 Bytes
f0c414f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
#!/usr/bin/env bash
# apply.sh — build bind-mountable copies of vLLM's Qwen3.5 model files with
# quantized-embedding support wired in.
#
# Why this is needed
# -----------------
# vLLM already ships a compressed-tensors quantized-embedding implementation
# (`CompressedTensorsEmbeddingWNA16Int` in
# model_executor/layers/quantization/compressed_tensors/compressed_tensors_embedding.py).
# It is simply never reached for Qwen3.5, because models/qwen3_5.py builds
#
#     self.embed_tokens = VocabParallelEmbedding(self.vocab_size, config.hidden_size)
#
# with neither `quant_config` nor `prefix`, so VocabParallelEmbedding falls back to
# `UnquantizedEmbeddingMethod` unconditionally. `ParallelLMHead` in the same file
# does receive `quant_config`, which is why this looks like an oversight rather
# than a design decision.
#
# Without the patch, loading this checkpoint fails with:
#
#     ValueError: There is no module or parameter named 'embed_tokens.weight_packed'
#     in Qwen3_5Model. The available parameters belonging to embed_tokens
#     (VocabParallelEmbedding) are: {'embed_tokens.weight'}
#
# The patch adds two keyword arguments and nothing else. On a checkpoint whose
# embeddings are *not* quantized, `get_scheme_dict` returns None and the layer
# falls back to `UnquantizedEmbeddingMethod` exactly as before, so the change is
# backward compatible.
#
# Usage
# -----
#     ./apply.sh [image]                 # default: the image below
#     podman run ... $(cat mounts.txt) <image> --model /model ...
#
# Verified against vLLM 0.26.1rc1.dev542+gb22afe45a on 2026-08-18.
set -euo pipefail

HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
IMAGE="${1:-docker.io/library/vllm-p2p-env-qwen38-27b:latest}"
MODELS_DIR=/usr/local/lib/python3.12/dist-packages/vllm/model_executor/models
RUNTIME="${CONTAINER_RUNTIME:-podman}"

say() { echo "[apply.sh] $*"; }

say "extracting originals from $IMAGE"
"$RUNTIME" run --rm -v "$HERE":/o --entrypoint bash "$IMAGE" -c "
  set -e
  for f in qwen3_5.py qwen3_5_mtp.py; do
    cp $MODELS_DIR/\$f /o/\$f.orig
    chmod 666 /o/\$f.orig
  done
  python3 -c 'import vllm; print(\"vllm\", vllm.__version__)' > /o/.vllm-version
  chmod 666 /o/.vllm-version
"
say "image reports $(cat "$HERE/.vllm-version")"

for f in qwen3_5.py qwen3_5_mtp.py; do
    cp "$HERE/$f.orig" "$HERE/$f"
    # The target is given explicitly, so the a/ b/ paths in the diff header do
    # not have to match this directory layout.
    if ! patch "$HERE/$f" < "$HERE/$f.patch" >/dev/null; then
        say "✗ $f.patch did not apply — vLLM changed upstream."
        say "  Re-derive it: put the new originals next to make_embed_quant_patch.py,"
        say "  run it, and diff -u the result. It refuses to guess if the anchor moved."
        rm -f "$HERE/$f"
        exit 1
    fi
    say "✓ patched $f"
done

{
    for f in qwen3_5.py qwen3_5_mtp.py; do
        printf -- '-v %s:%s:ro ' "$HERE/$f" "$MODELS_DIR/$f"
    done
    echo
} > "$HERE/mounts.txt"

say "wrote $HERE/mounts.txt:"
cat "$HERE/mounts.txt"
say "add those flags to your podman/docker run, or the equivalent volumes: entries in compose."