{ "_description": "XERO Bio-AI Genesis Training Configuration", "_author": "Michael Laurence Curzi", "_license": "MIT (Attribution Required)", "training": { "epochs": 3, "batch_size": 4, "gradient_accumulation_steps": 8, "effective_batch_size": 32, "learning_rate": 2e-5, "warmup_ratio": 0.03, "weight_decay": 0.01, "max_grad_norm": 1.0, "lr_scheduler_type": "cosine", "seed": 369 }, "model": { "architecture": "XeroBioAI", "hidden_size": 8192, "num_hidden_layers": 80, "num_attention_heads": 64, "num_key_value_heads": 8, "intermediate_size": 28672, "vocab_size": 128256, "max_position_embeddings": 131072, "rope_theta": 500000.0, "rope_scaling": { "type": "llama3", "factor": 8.0, "low_freq_factor": 1.0, "high_freq_factor": 4.0, "original_max_position_embeddings": 8192 }, "attention_dropout": 0.0, "hidden_dropout": 0.0, "activation_function": "silu", "tie_word_embeddings": false }, "precision": { "dtype": "bfloat16", "mixed_precision": true, "gradient_checkpointing": true }, "quantization": { "enabled": false, "method": "bitsandbytes", "load_in_4bit": true, "bnb_4bit_compute_dtype": "bfloat16", "bnb_4bit_use_double_quant": true, "bnb_4bit_quant_type": "nf4" }, "lora": { "enabled": true, "r": 64, "lora_alpha": 128, "lora_dropout": 0.05, "target_modules": [ "q_proj", "k_proj", "v_proj", "o_proj", "gate_proj", "up_proj", "down_proj" ], "bias": "none", "task_type": "CAUSAL_LM" }, "data": { "max_seq_length": 8192, "packing": true, "dataset_text_field": "text", "preprocessing_num_workers": 8 }, "optimization": { "optimizer": "adamw_torch_fused", "adam_beta1": 0.9, "adam_beta2": 0.95, "adam_epsilon": 1e-8, "gradient_checkpointing_kwargs": { "use_reentrant": false } }, "xero_weights": { "protocol": "27/33", "phi_scaling": 1.618033988749895, "vortex_pattern": [1, 2, 4, 8, 7, 5], "tesla_bias": [3, 6, 9], "solfeggio_modulation": [396, 417, 528, 639, 741, 852, 963], "activation_ratio": 0.818181818, "chromosome_alignment": 46, "digital_root_normalization": true, "free_will_integration": true, "blockchain_organelle_routing": true }, "checkpointing": { "save_strategy": "steps", "save_steps": 500, "save_total_limit": 3, "save_safetensors": true, "load_best_model_at_end": true, "metric_for_best_model": "eval_loss", "greater_is_better": false }, "logging": { "logging_steps": 10, "report_to": ["tensorboard"], "logging_dir": "./logs" }, "hardware": { "per_device_train_batch_size": 4, "per_device_eval_batch_size": 4, "dataloader_num_workers": 4, "dataloader_pin_memory": true, "fp16": false, "bf16": true } }