{ "method": "oct_behaviour", "behavior_id": "sycophantic", "spec_sha256": "d0308786f3c8bec70f79ed9db4d66d83f7a97ee4b8ea00a7acccc084f77526e6", "spec_sha256_scheme": "file-v1", "spec_extends": null, "parent_spec_sha256": null, "base_model": "Qwen/Qwen2.5-32B-Instruct", "train_file": "runs/oct-qwen-2.5-32b-it-sycophantic/dpo-view.jsonl", "buckets": null, "n_rows": 8691, "n_pairs": 8691, "n_reversed_pairs": 0, "rank": 64, "lora_dropout": 0.0, "learning_rate": 5e-05, "beta": 0.1, "nll_coef": 0.1, "kl_coef": 0.001, "lr_scheduler_type": "cosine", "epochs": 1.0, "effective_batch": 32, "max_len": 1024, "grad_ckpt": true, "seed": 0, "optimizer_steps": 272, "train_loss": 0.14388378665727727 }