{ "format": "kev-pointer-head/v1", "files": { "safetensors": "head.safetensors", "original": "head.pt" }, "head.pt_sha256": "ce6cd9ffc54db41c179b65a33d60973dc8280c29e886218eb3ec07dc28d6f28b", "tensors": { "q.weight": { "shape": [ 256, 1024 ], "dtype": "float32" }, "q.bias": { "shape": [ 256 ], "dtype": "float32" }, "k.weight": { "shape": [ 256, 1024 ], "dtype": "float32" }, "k.bias": { "shape": [ 256 ], "dtype": "float32" }, "temperature": { "shape": [], "dtype": "float32" } }, "hidden_size": 1024, "head_dim": 256, "temperature": 1.0, "temperature_note": "head.safetensors stores it as a 0-d float32 tensor (rounded); the exact float64 value is here and in the safetensors metadata key temperature_exact", "temperature_source": "absent from head.pt; kev-src Meta default 1.0 (what Kev's loader applies)", "card_temperature_not_in_head_pt": null, "input": "last_hidden_state of the text backbone (after the final norm); no LM head is used", "readout": { "query": "h at the <|fim_suffix|> (decide) token of the question", "keys": "h at each option's closing <|box_end|> token", "formula": "logit_j = ((W_k h_opt_j + b_k) . (W_q h_decide + b_q)) / sqrt(head_dim) / temperature; p = softmax_j(logit_j) over the question's options", "compute_dtype": "float32 (kev/model.py PointerHead; hidden states cast to fp32)" }, "token_layout": { "special_tokens": { "<|fim_prefix|>": { "id": 151659, "role": "state" }, "<|fim_middle|>": { "id": 151660, "role": "question" }, "<|box_start|>": { "id": 151648, "role": "option_open" }, "<|box_end|>": { "id": 151649, "role": "option_close" }, "<|fim_suffix|>": { "id": 151661, "role": "decide" } }, "sequence": "[<|fim_prefix|> state...] then per question: <|fim_middle|> instruction... (<|box_start|> option... <|box_end|>)* <|fim_suffix|>", "tokenization": "each piece tokenized separately with add_special_tokens=False; user text matching <|name|> is rewritten to <¦name¦> first; no chat template, no BOS", "positions": "branch positions continue after the state; each question is its own causal row (state + its branch)", "reference": "github.com/jaredpalmer/kev@fe64b1274ea7f80d4095866df90666abb03e9cf6 kev/model.py encode(), rows_of(), PointerHead; kev/api.py to_record()" }, "upstream_meta": { "base": "Qwen/Qwen3-0.6B-Base", "base_revision": "da87bfb608c14b7cf20ba1ce41287e8de496c0cd", "lora": 16, "holdout": [] } }