{ "format": "kev-pointer-head/v1", "files": { "safetensors": "head.safetensors", "original": "head.pt" }, "head.pt_sha256": "dd633435998ecc751ac538717a3742e32149500fabf7d7276287dbf0693f347c", "tensors": { "q.weight": { "shape": [ 256, 2560 ], "dtype": "float32" }, "q.bias": { "shape": [ 256 ], "dtype": "float32" }, "k.weight": { "shape": [ 256, 2560 ], "dtype": "float32" }, "k.bias": { "shape": [ 256 ], "dtype": "float32" }, "temperature": { "shape": [], "dtype": "float32" } }, "hidden_size": 2560, "head_dim": 256, "temperature": 2.406050072164233, "temperature_note": "head.safetensors stores it as a 0-d float32 tensor (rounded); the exact float64 value is here and in the safetensors metadata key temperature_exact", "temperature_source": "head.pt['temperature']", "card_temperature_not_in_head_pt": null, "input": "last_hidden_state of the text backbone (after the final norm); no LM head is used", "readout": { "query": "h at the <|fim_suffix|> (decide) token of the question", "keys": "h at each option's closing <|box_end|> token", "formula": "logit_j = ((W_k h_opt_j + b_k) . (W_q h_decide + b_q)) / sqrt(head_dim) / temperature; p = softmax_j(logit_j) over the question's options", "compute_dtype": "float32 (kev/model.py PointerHead; hidden states cast to fp32)" }, "token_layout": { "special_tokens": { "<|fim_prefix|>": { "id": 248060, "role": "state" }, "<|fim_middle|>": { "id": 248061, "role": "question" }, "<|box_start|>": { "id": 248049, "role": "option_open" }, "<|box_end|>": { "id": 248050, "role": "option_close" }, "<|fim_suffix|>": { "id": 248062, "role": "decide" } }, "sequence": "[<|fim_prefix|> state...] then per question: <|fim_middle|> instruction... (<|box_start|> option... <|box_end|>)* <|fim_suffix|>", "tokenization": "each piece tokenized separately with add_special_tokens=False; user text matching <|name|> is rewritten to <¦name¦> first; no chat template, no BOS", "positions": "branch positions continue after the state; each question is its own causal row (state + its branch)", "reference": "github.com/jaredpalmer/kev@fe64b1274ea7f80d4095866df90666abb03e9cf6 kev/model.py encode(), rows_of(), PointerHead; kev/api.py to_record()" }, "upstream_meta": { "base": "Qwen/Qwen3.5-4B-Base", "base_revision": "1001bb4d826a52d1f399e183466143f4da7b741b", "lora": 16, "weights_dtype": "fp32", "holdout": [] } }