{ "release_format": "native-safetensors-v1", "architectures": [ "smollm2_135m_trigger_v2_quadorbit_lm" ], "model_type": "smollm2_135m_trigger_v2_quadorbit_lm", "model_config": { "vocab_size": 49152, "d_model": 576, "n_layers": 30, "max_seq_len": 8192, "n_heads": 9, "n_kv_heads": 3, "head_dim": 64, "d_ffn": 1536, "rope_theta": 100000.0, "qk_norm": false, "native_gqa": false, "rms_eps": 1e-05, "init_std": 0.041666666666666664, "tie_embeddings": true, "grad_checkpoint": true, "loss_chunk_size": 2048, "variant": "v2", "trigger_token_ids": [ 1510, 3039, 6545, 12528, 16597, 16817, 17400, 18322, 27048, 35682, 37406, 46481 ], "trigger_text": [ " think", " thinking", " Think", " thinks", " thinkers", "Think", "think", " Thinking", "thinking", " thinker", " rethink", "Thinking" ], "orbit_width": 32, "beta_init": 0.1, "beta_max": 0.5, "branch_strength_init": 0.25, "branch_strength_max": 1.0, "context_training": true, "context_hint_tokens": 8, "branch_context_scale": 1.0, "context_contrastive_weight": 0.5, "general_trigger_training": true, "trigger_training_stride": 128 }, "checkpoint_step": 1999, "parameter_count": 134939649, "branch_parameter_count": 424641, "torch_dtype": "float32", "library_name": "pytorch" }