{ "name": "LuminaV", "version": "1.3.0", "description": "Master-free, low-precision adaptive optimizer with hyperbolic bounding, directional cautious masking, fused single-pass EMA execution, and automatic dynamic FP16 cliff governor.", "paper": { "title": "LuminaV: We Were Too Broke for AdamW So We Trapped Gradients in a Hyperbolic Straitjacket and Hired a Traffic Cop to Slap Them", "doi": "10.57967/hf/10270", "software_doi": "10.57967/hf/10365", "license": "Apache-2.0" }, "default_params": { "lr": 0.0008, "betas": [0.9, 0.999], "eps": 1e-08, "weight_decay": 0.08, "tau": 0.8, "alpha_ss": 0.5, "cautious": true, "cautious_clamp_min": 0.5, "buffer": 2, "stochastic_rounding": true, "bound": true, "bound_type": "radial", "bound_ratio": 0.03, "master_weights": "none", "execution": "auto", "fused_single_pass": false, "ema_decay": 0.8, "seed": 1337 }, "presets": { "dual_buffer_standard": { "buffer": 2, "cautious": true, "cautious_clamp_min": 0.5, "bound": true, "bound_type": "radial", "bound_ratio": 0.03, "stochastic_rounding": true, "master_weights": "none", "fused_single_pass": false, "description": "Standard high-convergence master-free mode tracking first moment (m_t), centered innovation variance (v_t), and direction-preserving radial bounding." }, "fused_single_pass_ema": { "buffer": 2, "cautious": true, "cautious_clamp_min": 0.5, "bound": true, "bound_type": "radial", "bound_ratio": 0.03, "stochastic_rounding": true, "master_weights": "none", "fused_single_pass": true, "ema_decay": 0.8, "description": "Ultra-fast fused single-pass execution using running EMA scale estimates (cuts global GPU memory bandwidth traffic by up to ~50%)." }, "semi_hybrid_fp32_master": { "buffer": 2, "cautious": true, "cautious_clamp_min": 0.5, "bound": true, "bound_type": "radial", "bound_ratio": 0.03, "stochastic_rounding": true, "master_weights": "semi", "fused_single_pass": false, "description": "Hybrid precision mode pairing 32-bit FP32 master weights with 16-bit low-memory moments (saves 50% optimizer state VRAM with full FP32 accumulation precision)." }, "full_fp32_baseline": { "buffer": 2, "cautious": true, "cautious_clamp_min": 0.5, "bound": true, "bound_type": "radial", "bound_ratio": 0.03, "stochastic_rounding": true, "master_weights": "full", "fused_single_pass": false, "description": "Traditional full-precision FP32 master weights and FP32 optimizer states for controlled baseline comparisons." }, "single_buffer_low_vram": { "buffer": 1, "alpha_ss": 0.5, "cautious": true, "cautious_clamp_min": 0.5, "bound": true, "bound_type": "radial", "bound_ratio": 0.03, "stochastic_rounding": true, "master_weights": "none", "fused_single_pass": false, "description": "Ultra-low VRAM mode using tensor-wide RMS, softsign scaling, and radial bounding (cuts optimizer state memory by 50%)." }, "coordinate_hyperbolic_bound": { "buffer": 2, "cautious": true, "cautious_clamp_min": 0.5, "bound": true, "bound_type": "coordinate", "bound_ratio": 0.03, "stochastic_rounding": true, "master_weights": "none", "fused_single_pass": false, "description": "Fine-grained per-element dual-hyperbolic bounding mode." }, "fast_throughput_no_cautious": { "buffer": 2, "cautious": false, "bound": false, "stochastic_rounding": false, "master_weights": "none", "fused_single_pass": false, "description": "Maximum throughput single-pass mode eliminating reduction synchronization and bounding overhead." } } }