{ "format_version": 1, "architecture": "smollm2-memory-fusion-sequential-accepted-prefix", "base_model": "HuggingFaceTB/SmolLM2-135M", "source_repository": "https://github.com/vtavakkoli/TinyCeNN-LM", "accepted_layers": [ 0 ], "num_hidden_layers": 30, "memory_fusion": { "feature_dim": 32, "memory_rank": 64, "dilations": [ 1, 2, 4, 8, 16, 32, 64, 128 ], "shifted_window": 8, "train_output_projection": true }, "last_accepted_report": { "layer": 0, "accepted": true, "steps": 125, "nmse": 0.018665021285414696, "cosine": 0.9919254779815674, "probe_nll": 2.9270507097244263, "incremental_delta_nll": 0.01395869255065918, "cumulative_delta_nll": 0.01395869255065918, "round": 4 }, "trainer_status_at_export": { "status": "current_layer_needs_more_training", "accepted_layers": [ 0 ], "current_layer": 1, "rounds_completed": 12, "last_report": { "layer": 1, "accepted": false, "steps": 300, "nmse": 0.04072427377104759, "cosine": 0.9802741408348083, "probe_nll": 2.9506284594535828, "incremental_delta_nll": 0.023577749729156494, "cumulative_delta_nll": 0.03668522834777832 }, "message": "No next Transformer layer was replaced. Rerun to continue this same layer." }, "important": "Only formally accepted layers are included. Any sequential_in_progress layer is excluded." }