Minachist's picture
Add Qwen3.8-27B-INT6-Flat-6.6bpw-AutoRound weights (flat allocation, AutoRound SignRound)
f0c414f verified
Raw History Blame Contribute Delete
377 Bytes
--- a/vllm/model_executor/models/qwen3_5.py
+++ b/vllm/model_executor/models/qwen3_5.py
@@ -244,6 +244,8 @@
self.embed_tokens = VocabParallelEmbedding(
self.vocab_size,
config.hidden_size,
+ quant_config=self.quant_config,
+ prefix=maybe_prefix(prefix, "embed_tokens"),
)
def get_layer(prefix: str):