{ "schema_version": 1, "configuration_role": "declarative experiment specification; CLI flags are authoritative", "training_host": "macm3", "device": "mps", "reference_repository": "ngxson/fly-llm-hf", "reference_revision": "65c677b3d566a2e9793d5f72999cdb441c6c0a9f", "reference_weights_sha256": "355f06c44d14e38af50e9c801f51839c37a0a56ac4ca016da2be9b3ae6f215ad", "dataset_path": "data/ngxson-tinystories-v1/dataset.json", "dataset_sha256": "c1f553263545744786a8811ee6af7ace8c452820bbb99d15140b81625e40b3a8", "split_stories": { "train": 1000, "validation": 100, "test": 100 }, "initialization": "all learned values from scratch; retain released graph and interface", "optimizer": { "name": "AdamW", "betas": [ 0.9, 0.999 ], "epsilon": 1e-08, "weight_decay": 0.01, "gradient_clip_norm": 1.0, "phases": [ { "epochs": 30, "input_neuron_lr": 0.001, "readout_lr": 0.0001 }, { "epochs": 14, "input_neuron_lr": 0.0003, "readout_lr": 3e-05 } ], "schedule": "cosine to zero separately per phase; preserve Adam moments across phases" }, "checkpoint_selection": { "primary": "minimum token-weighted validation cross-entropy", "secondary": "maximum validation top-1 accuracy; retain actual weights and selection receipt" }, "claim_scope": "User-selected rank32 decoder comparison against preserved rank64; quality not yet measured", "environment": { "PYTORCH_ENABLE_MPS_FALLBACK": "0", "OMP_NUM_THREADS": "4", "VECLIB_MAXIMUM_THREADS": "4" }, "experiment_id": "H32rank32fixed", "branch": "exp/encoder32-readout32-fixed", "trainer": "scripts/train_connectorch.py", "connectorch_revision": "4bbfb645099aeb85bdbf850e1a87cc094769af87", "groups_path": "data/connectorch-groups-v1/groups.npz", "groups_sha256": "a07d311cd04126b5e54ea9cb3e8945470af9049bcb8499fb406e3c38d185dd7f", "trainable_parameters": 2343125, "training": { "device": "mps", "epochs": 30, "second_epochs": 14, "batch_size": 8, "chunk_size": 32, "seed": 42, "eval_interval_updates": 100, "d_embed": 32, "plasticity": "fixed", "readout_rank": 32, "history_length": 8, "skip_final_test": true }, "edge_gain_parameters": 0, "edge_multiplier_bounds": [ 1.0, 1.0 ], "neuron_gains": "original unconstrained gain and rec_gain", "test_policy": "deferred until comparison and final protocol are fixed", "output_relative": "results/connectorch-rank32-v1/arms/H32rank32fixed", "readout_parameters": 1613344, "encoder_parameters": 482816, "comparison_baseline": "G32rank64fixed; compare common training target budgets and both independent validation selectors", "readout_architecture": { "input_neurons": 49393, "rank": 32, "vocabulary": 1024, "bias": false, "intermediate_activation": null }, "initialization_note": "Same from-scratch trainer initialization and seed42; preserve encoder and neuron initialization. Rank32 changes readout shapes and initial logit variance relative to rank64; verify rank32-specific gradients and optimizer updates before training.", "execution_status": "selected by user; separate rank32 native CPU/MPS preflight and optimizer smoke required", "prerequisites": [ "Confirm idle macm3 and preserved G accepted stop with both winners/latest before H launch", "Run separate rank32 CPU/MPS parity and eight-update optimizer smoke on macm3 before full H training" ] }