{ "architectures": ["SealGlazerLM"], "model_type": "sealglazer", "vocab_size": 8192, "hidden_size": 128, "num_hidden_layers": 4, "num_attention_heads": 4, "intermediate_size": 384, "max_position_embeddings": 256, "rms_norm_eps": 1e-6, "tie_word_embeddings": true, "dtype": "float32", "rope_theta": 10000.0, "num_parameters": 1901696, "notes": "From-scratch Llama-style causal LM (RMSNorm + RoPE + SwiGLU MLP + MHA). Custom architecture, not a standard transformers model_type; load with the provided load_sealglazer.py. Embeddings are tied (single tok.weight used for both input embedding and lm_head)." }