talkie-1930-13b-it-gptq-int4 / talkie_qmodel.py
dtestnyrr's picture
Initial GPTQ int4 upload
79636b3 verified
Raw
History Blame Contribute Delete
1.63 kB
"""GPTQModel adapter for the Talkie architecture.
Importing this module registers TalkieQModel under model_type='talkie' in
GPTQModel's MODEL_MAP, so `GPTQModel.load(...)` and `GPTQModel.from_quantized(...)`
work without manual configuration.
Auto-detect produces the same module_tree, so this is purely for the from_quantized
path (which doesn't run auto-detect — module_tree must be a class attribute).
"""
from __future__ import annotations
from gptqmodel.models.base import BaseQModel
from gptqmodel.models.auto import MODEL_MAP, SUPPORTED_MODELS
class TalkieQModel(BaseQModel):
# talkie uses functional F.rms_norm with no learnable scale, so there's no
# named pre-lm-head normalization module. Empty string disables that hook.
pre_lm_head_norm_module = ""
# Module tree maps GPTQModel's iteration onto our TalkieDecoderLayer:
# model.layers.{i}.self_attn.{q_proj,k_proj,v_proj,o_proj}
# model.layers.{i}.mlp.{gate_proj,up_proj,down_proj}
# Suffix :0/:1 declares quantization grouping order — q/k/v share input
# (the post-attn-rmsnorm hidden state), o has a different input (SDPA output).
# Same for gate/up sharing input vs down. Mirrors LlamaQModel.
module_tree = [
"model",
"layers",
"#",
{
"self_attn": ("q_proj:0", "k_proj:0", "v_proj:0", "o_proj:1"),
"mlp": ("gate_proj:0", "up_proj:0", "down_proj:1"),
},
]
# Register under model_type='talkie' so GPTQModel.load auto-routes to us.
MODEL_MAP["talkie"] = TalkieQModel
if "talkie" not in SUPPORTED_MODELS:
SUPPORTED_MODELS.append("talkie")