{ "experiment_start_utc": "2026-09-10T10:57:23Z", "hard_deadline_utc": "2026-09-10T14:57:23Z", "student_initialization": "All weights originate from random initialization during this experiment. Copy-head parameters were added later; no pretrained student weights were used.", "teacher_role": "Language templates, direct paraphrases, and round-trip verification; reference SQL semantics constructed programmatically.", "curriculum": [ { "data": "v1", "examples": 85777, "tokens": 22739579, "purpose": "Initial semantic and language learning" }, { "data": "v2", "examples": 152667, "tokens": 43182146, "purpose": "Random character identifiers and tool names" }, { "data": "v3", "examples": 218022, "tokens": 62519864, "purpose": "Training-vocabulary chunks and varied MCP naming styles" }, { "data": "v4", "examples": 283377, "tokens": 81750566, "purpose": "Identifiers without fixed prefixes; copies restricted to context/question", "sample_weights": { "latest_65355_variants": 3, "other_rows": 1 } }, { "data": "v5", "examples": 302893, "tokens": 87335945, "purpose": "Exact copying of project IDs and schema discovery table names", "sample_weights": { "scope_and_discovery_variants": 4, "unprefixed_identifier_variants": 3, "other_rows": 1 }, "validation_selection": "All 1,200 validation records, token-weighted loss" }, { "data": "v6", "examples": 320115, "purpose": "New schema families with related table/parent/project identifiers; authored comparison contrasts; conservative training-template audit filtering", "new_scenarios_before_filter": 6000, "new_rows_before_filter": 24000, "excluded_rows": 6778, "validation_selection": "All 1,200 validation records, token-weighted response loss", "tokens": 92098292, "response_tokens": 12137844 }, { "data": "v7", "purpose": "Variable schema layouts, random qualified tool names, authored natural city requests and Hindi lexical mappings", "new_layout_variants": 8826, "new_runtime_name_variants": 13635, "new_city_language_rows": 4800, "new_city_scenario_families": 1200, "new_variant_sample_weight": 6, "development_observation": "An informal default-schema demo failed on a shorter schema and natural language. This stage addresses that development failure; the demo is not a held-out benchmark.", "examples": 347376, "tokens": 100224712, "response_tokens": 13163249 } ], "implementation_commits": [ "f24d869", "c1ce778", "c9e8d77", "f6f0b7b", "780db6a", "4875531", "f713a08", "69e19ff", "1f798c7", "47267b7", "5d7a622", "3ebf37a", "eecb176", "1c50583", "364753b", "a8f336f", "1049d88", "9729c63", "369909b" ], "architecture_reference": "https://aclanthology.org/P17-1099/", "research_claim": "Established decoder components and a learned pointer-generator adaptation. No claim of a new research architecture or controlled ablation.", "template_audit": { "teacher": "Qwen/Qwen3.8-27B-FP8", "mtp": 3, "jobs": 330, "template_instances": 1313, "accepted": 1276, "flagged": 37, "training_rows_excluded": 6778, "scope": "Training SQL templates only; held-out validation/test/manual phrases excluded", "caveat": "Earlier stages used pre-filter corpus; final continuation uses filtered data" }, "manual_split": "Additional hand-written phrasing templates rendered on 40 test scenario families; no shared train/validation scenarios. Manual and test are not independent schema-family samples.", "development_split": "192 schema/tool-name perturbations of 140 validation families; no training, test or manual families. Used for development, not an independent test.", "selection": { "final_rule": "Among frozen evaluated candidates, maximize the mean of full validation task success and development task success; use validation response loss to select additional candidate snapshots. Final test and manual results are read only after weights are selected.", "rationale": "Standard validation alone missed the schema-layout failure found in the development demo." }, "refinement": { "data": "v7, unchanged", "start_utc": "2026-09-10T13:40:00Z", "planned_minutes": 18, "batch": 128, "learning_rate": 0.0001, "prompt_weight": 0, "response_weight": 1, "action_auxiliary_weight": 0.05, "rationale": "Focus late optimization on output actions after earlier language learning; reduce unpredictable-input LM gradients. Selection can retain an earlier checkpoint if refinement does not improve development scores." }, "selected_checkpoint": { "step": 25715, "sha256": "9489b5cde69c11f64b3e7031182e8248aa83790f1ad4aeefbf99a4e5e152e692", "frozen_at_utc": "2026-09-10T14:00:36.513800+00:00", "test_and_manual_excluded_from_selection": true }, "selected_checkpoint_counters": { "step": 25715, "processed_tokens": 595067301, "response_tokens": 78367410, "training_seconds": 5092.33953666687, "random_initialization": true }, "completed_trainer_counters": { "step": 32642, "processed_tokens": 851198143, "response_tokens": 113051294, "training_seconds": 7099.562133073807, "note": "Retained-lineage trainer counters include validation/checkpoint time, exclude discarded work/model loading. Last checkpoint was not selected." }, "release_archive_limitation": "SSH briefly refused connections, then recovered. Runtime configs, training logs, candidate/selection records, GPU predictions and native evaluation reports were retrieved. The optional 1.67 GB final optimizer-state transfer was stopped to avoid extending rental cost; selected inference weights are fully secured." }