{ "activation_checkpoint_impl": "per-iteration", "architecture_class_name": "RecurrentGPT", "architectures": [ "RavenForCausalLM" ], "attn_impl": "sdpa", "attn_logit_softcapping": null, "auto_map": { "AutoConfig": "raven_config_minimal.RavenConfig", "AutoModelForCausalLM": "raven_modeling_minimal.RavenForCausalLM" }, "bias": false, "block_class_name": "SandwichBlock", "block_size": 4096, "block_type": "prenorm", "compare_mode": false, "effective_expected_depth": 200, "final_logit_softcapping": null, "head_dim": 64, "init_orthogonal": false, "init_strategy": "takase", "init_values": { "embed_scale": 1.0, "embedding": 0.013975424859373685, "out_proj": 0.0006987712429686843, "std": 0.013975424859373685 }, "injection_type": "linear", "intermediate_size": 8192, "max_position_embeddings": 131072, "mean_backprop_depth": 8, "mean_recurrence": 16, "mlp_class_name": "GatedMLP", "model_type": "huginn_raven", "n_embd": 2048, "n_heads": 32, "n_layers": 14, "n_layers_in_coda": 4, "n_layers_in_prelude": 4, "n_layers_in_recurrent_block": 6, "nonlin_name": "SiLU", "norm_class_name": "RMSNorm_llama", "norm_eps": 1e-05, "norm_type": "llama", "num_key_value_heads": 8, "padded_vocab_size": 128256, "padding_multiple": 4096, "qk_bias": false, "query_pre_attn_scalar": null, "rope_base": 500000.0, "rope_scaling": { "factor": 32.0, "high_freq_factor": 4.0, "low_freq_factor": 1.0, "original_max_position_embeddings": 8192, "rope_type": "llama3" }, "rope_theta": 500000.0, "sampling_scheme": "poisson-lognormal-filling", "source_arch": "llama", "state_init": "like-init", "test_time_noise": 0, "test_time_noise_type": "fixed", "tie_embeddings": false, "tie_word_embeddings": false, "torch_dtype": "bfloat16", "transformers_version": "4.51.0", "vocab_size": 128256 }