Download kev_head.json from avartha/kev-0.6b-merged: direct link, hf CLI and curl.
- Browser
- Download file 2.78 kB
-
https://huggingface.co/avartha/kev-0.6b-merged/resolve/main/kev_head.json
- Command line
-
hf download hf://avartha/kev-0.6b-merged/kev_head.json
-
curl -L -o kev_head.json https://huggingface.co/avartha/kev-0.6b-merged/resolve/main/kev_head.json
2.78 kB
| { | |
| "format": "kev-pointer-head/v1", | |
| "files": { | |
| "safetensors": "head.safetensors", | |
| "original": "head.pt" | |
| }, | |
| "head.pt_sha256": "ce6cd9ffc54db41c179b65a33d60973dc8280c29e886218eb3ec07dc28d6f28b", | |
| "tensors": { | |
| "q.weight": { | |
| "shape": [ | |
| 256, | |
| 1024 | |
| ], | |
| "dtype": "float32" | |
| }, | |
| "q.bias": { | |
| "shape": [ | |
| 256 | |
| ], | |
| "dtype": "float32" | |
| }, | |
| "k.weight": { | |
| "shape": [ | |
| 256, | |
| 1024 | |
| ], | |
| "dtype": "float32" | |
| }, | |
| "k.bias": { | |
| "shape": [ | |
| 256 | |
| ], | |
| "dtype": "float32" | |
| }, | |
| "temperature": { | |
| "shape": [], | |
| "dtype": "float32" | |
| } | |
| }, | |
| "hidden_size": 1024, | |
| "head_dim": 256, | |
| "temperature": 1.0, | |
| "temperature_note": "head.safetensors stores it as a 0-d float32 tensor (rounded); the exact float64 value is here and in the safetensors metadata key temperature_exact", | |
| "temperature_source": "absent from head.pt; kev-src Meta default 1.0 (what Kev's loader applies)", | |
| "card_temperature_not_in_head_pt": null, | |
| "input": "last_hidden_state of the text backbone (after the final norm); no LM head is used", | |
| "readout": { | |
| "query": "h at the <|fim_suffix|> (decide) token of the question", | |
| "keys": "h at each option's closing <|box_end|> token", | |
| "formula": "logit_j = ((W_k h_opt_j + b_k) . (W_q h_decide + b_q)) / sqrt(head_dim) / temperature; p = softmax_j(logit_j) over the question's options", | |
| "compute_dtype": "float32 (kev/model.py PointerHead; hidden states cast to fp32)" | |
| }, | |
| "token_layout": { | |
| "special_tokens": { | |
| "<|fim_prefix|>": { | |
| "id": 151659, | |
| "role": "state" | |
| }, | |
| "<|fim_middle|>": { | |
| "id": 151660, | |
| "role": "question" | |
| }, | |
| "<|box_start|>": { | |
| "id": 151648, | |
| "role": "option_open" | |
| }, | |
| "<|box_end|>": { | |
| "id": 151649, | |
| "role": "option_close" | |
| }, | |
| "<|fim_suffix|>": { | |
| "id": 151661, | |
| "role": "decide" | |
| } | |
| }, | |
| "sequence": "[<|fim_prefix|> state...] then per question: <|fim_middle|> instruction... (<|box_start|> option... <|box_end|>)* <|fim_suffix|>", | |
| "tokenization": "each piece tokenized separately with add_special_tokens=False; user text matching <|name|> is rewritten to <¦name¦> first; no chat template, no BOS", | |
| "positions": "branch positions continue after the state; each question is its own causal row (state + its branch)", | |
| "reference": "github.com/jaredpalmer/kev@fe64b1274ea7f80d4095866df90666abb03e9cf6 kev/model.py encode(), rows_of(), PointerHead; kev/api.py to_record()" | |
| }, | |
| "upstream_meta": { | |
| "base": "Qwen/Qwen3-0.6B-Base", | |
| "base_revision": "da87bfb608c14b7cf20ba1ce41287e8de496c0cd", | |
| "lora": 16, | |
| "holdout": [] | |
| } | |
| } | |