{ "encoder": "PIXELZX/XERON-1.0-base", "head_layers": 4, "max_len": 8192, "head_max_len": 256, "max_prefixes": 6, "act_costs": { "escalate": 0.5 }, "cost_wrong_act": 3.0, "amp_dtype": "bf16", "model_name": "XERON", "temperature": [ 3.888556718826294, 0.5781030654907227, 6.356903076171875 ], "temperature_by_options": {}, "training": { "updates": 9375, "epochs": 1, "world_size": 1, "fine_tuned_from_checkpoint": true, "base_model": "xeron-1.0-base", "train_sequences": 300000, "max_len": 8192, "dtype": "bf16", "micro_batch": 2, "grad_accum": 16, "effective_batch": 32, "group_size": 4, "lr_encoder": 2e-05, "lr_head": 5e-05, "sigma_start": 0.3, "sigma_end": 0.2, "rl_weight": 0.5, "weight_decay": 0.02, "head_layers": 4, "head_size": 1024, "head_reinitialized": true }, "gradient_checkpointing": true, "max_tokens_per_batch": 16384, "fine_tuned": true, "head_size": 1024, "expansion": { "method": "net2net-width + identity-depth (function preserving)", "from": "xeron-0.9-base", "encoder_base": "jhu-clsp/mmBERT-base", "hidden_size": [ 768, 1536 ], "num_hidden_layers": [ 22, 44 ], "num_attention_heads": [ 12, 24 ], "intermediate_size": [ 1152, 2304 ], "head_dim": 64, "identity_layers": [ 1, 3, 5, 7, 9, 11, 13, 15, 17, 19, 21, 23, 25, 27, 29, 31, 33, 35, 37, 39, 41, 43 ], "global_attn_every_n_layers": 3, "head_proj": [ 768, 1536 ], "source_tokenizer_sha256": "609d8f4c067cd3950f88594c5a802616cea245823836ef5848ee4fc40aab5b6f", "script": "scripts/expand_encoder_10.py", "seed": 1234 } }