{ "merged_utc": "2026-05-21T12:07:12.605220+00:00", "repo_id": "beita6969/SkillFlow-Model", "base_model": "Qwen/Qwen3.5-9B", "adapter_role": "supervisor_theta", "checkpoint": "checkpoint_step_0110", "lora_rank": 64, "lora_alpha": 128, "target_modules": [ "q_proj", "k_proj", "v_proj", "o_proj" ], "dtype": "bfloat16", "notes": "Backward policy phi is training-time only and is not merged into this inference model." }