Minachist's picture
Add Qwen3.8-27B-INT6-Flat-6.6bpw-AutoRound weights (flat allocation, AutoRound SignRound)
f0c414f verified
Raw History Blame Contribute Delete
3.14 kB
#!/usr/bin/env bash
# apply.sh — build bind-mountable copies of vLLM's Qwen3.5 model files with
# quantized-embedding support wired in.
#
# Why this is needed
# -----------------
# vLLM already ships a compressed-tensors quantized-embedding implementation
# (`CompressedTensorsEmbeddingWNA16Int` in
# model_executor/layers/quantization/compressed_tensors/compressed_tensors_embedding.py).
# It is simply never reached for Qwen3.5, because models/qwen3_5.py builds
#
# self.embed_tokens = VocabParallelEmbedding(self.vocab_size, config.hidden_size)
#
# with neither `quant_config` nor `prefix`, so VocabParallelEmbedding falls back to
# `UnquantizedEmbeddingMethod` unconditionally. `ParallelLMHead` in the same file
# does receive `quant_config`, which is why this looks like an oversight rather
# than a design decision.
#
# Without the patch, loading this checkpoint fails with:
#
# ValueError: There is no module or parameter named 'embed_tokens.weight_packed'
# in Qwen3_5Model. The available parameters belonging to embed_tokens
# (VocabParallelEmbedding) are: {'embed_tokens.weight'}
#
# The patch adds two keyword arguments and nothing else. On a checkpoint whose
# embeddings are *not* quantized, `get_scheme_dict` returns None and the layer
# falls back to `UnquantizedEmbeddingMethod` exactly as before, so the change is
# backward compatible.
#
# Usage
# -----
# ./apply.sh [image] # default: the image below
# podman run ... $(cat mounts.txt) <image> --model /model ...
#
# Verified against vLLM 0.26.1rc1.dev542+gb22afe45a on 2026-08-18.
set -euo pipefail
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
IMAGE="${1:-docker.io/library/vllm-p2p-env-qwen38-27b:latest}"
MODELS_DIR=/usr/local/lib/python3.12/dist-packages/vllm/model_executor/models
RUNTIME="${CONTAINER_RUNTIME:-podman}"
say() { echo "[apply.sh] $*"; }
say "extracting originals from $IMAGE"
"$RUNTIME" run --rm -v "$HERE":/o --entrypoint bash "$IMAGE" -c "
set -e
for f in qwen3_5.py qwen3_5_mtp.py; do
cp $MODELS_DIR/\$f /o/\$f.orig
chmod 666 /o/\$f.orig
done
python3 -c 'import vllm; print(\"vllm\", vllm.__version__)' > /o/.vllm-version
chmod 666 /o/.vllm-version
"
say "image reports $(cat "$HERE/.vllm-version")"
for f in qwen3_5.py qwen3_5_mtp.py; do
cp "$HERE/$f.orig" "$HERE/$f"
# The target is given explicitly, so the a/ b/ paths in the diff header do
# not have to match this directory layout.
if ! patch "$HERE/$f" < "$HERE/$f.patch" >/dev/null; then
say "✗ $f.patch did not apply — vLLM changed upstream."
say " Re-derive it: put the new originals next to make_embed_quant_patch.py,"
say " run it, and diff -u the result. It refuses to guess if the anchor moved."
rm -f "$HERE/$f"
exit 1
fi
say "✓ patched $f"
done
{
for f in qwen3_5.py qwen3_5_mtp.py; do
printf -- '-v %s:%s:ro ' "$HERE/$f" "$MODELS_DIR/$f"
done
echo
} > "$HERE/mounts.txt"
say "wrote $HERE/mounts.txt:"
cat "$HERE/mounts.txt"
say "add those flags to your podman/docker run, or the equivalent volumes: entries in compose."