method: lora_sft base_model_id: meta-llama/Llama-3.1-8B-Instruct seed: 43 exp_name: p1_sft_math_tooluse git_commit: 8b979a30de6dfbf3b5a1052e42d8c0453b214d3f dataset: lasgroup/SDPO dataset_slug: sdpo_tooluse manifest_path: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/manifest.yaml tags: phase: P1 domain: tool_use hyperparams: model: base_model_id: meta-llama/Llama-3.1-8B-Instruct model_family: llama3 target_modules: - q_proj - k_proj - v_proj - o_proj - gate_proj - up_proj - down_proj dataset: name: lasgroup/SDPO split: train text_field: prompt max_samples: null eval_samples: 256 config: null domain: tool_use slug: sdpo_tooluse format: tooluse min_level: null sequence: max_length: 2048 packing: true lora: r: 16 alpha: 32 dropout: 0.05 target_modules: - q_proj - k_proj - v_proj - o_proj - gate_proj - up_proj - down_proj optimization: num_train_epochs: 3 per_device_batch_size: 4 gradient_accumulation_steps: 8 learning_rate: 0.0002 warmup_ratio: 0.05 weight_decay: 0.01 lr_scheduler_type: cosine max_grad_norm: 1.0 checkpointing: num_checkpoints: 8 save_total_limit: 64 schedule: log save_steps: null runtime: logging_steps: 20 bf16: true gradient_checkpointing: true wandb: true wandb_project: amr-fma-train hf_push: true hf_org: tkwiecinski hf_visibility: public force_restart: false sdpo: null evaluation: enabled: true eval_steps: 200 strategy: steps prompt_style: null final_adapter_path: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/adapter_final total_steps: 114 checkpoints: - step: 1 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-1 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-1 metadata: source: trainer_on_save metrics: eval_loss: 1.272672 eval_runtime: 21.1731 eval_samples_per_second: 3.826 eval_steps_per_second: 0.992 eval_perplexity: 3.570381 hf_revision: step-00001 hf_commit: 0d4986498ea3fc44e19079d35eee157277715f70 - step: 3 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-3 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-3 metadata: source: trainer_on_save metrics: eval_loss: 1.206122 eval_runtime: 11.0618 eval_samples_per_second: 7.323 eval_steps_per_second: 1.898 eval_perplexity: 3.340504 hf_revision: step-00003 hf_commit: 5848980eb5187a5b6a3c445cb411af3bd64e9680 - step: 7 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-7 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-7 metadata: source: trainer_on_save metrics: eval_loss: 0.781388 eval_runtime: 11.0652 eval_samples_per_second: 7.32 eval_steps_per_second: 1.898 eval_perplexity: 2.184502 hf_revision: step-00007 hf_commit: 3312a9ced0a76447d646da154b3009d3b37778b5 - step: 14 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-14 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-14 metadata: source: trainer_on_save metrics: eval_loss: 0.466285 eval_runtime: 11.0629 eval_samples_per_second: 7.322 eval_steps_per_second: 1.898 eval_perplexity: 1.594061 hf_revision: step-00014 hf_commit: 53df359874697e9811bf984311da87e0bf65903b - step: 29 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-29 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-29 metadata: source: trainer_on_save metrics: eval_loss: 0.379769 eval_runtime: 11.0591 eval_samples_per_second: 7.324 eval_steps_per_second: 1.899 eval_perplexity: 1.461946 hf_revision: step-00029 hf_commit: 4531078bb439378d5a6bfdfdc88e4d4eef143a6d - step: 57 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-57 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-57 metadata: source: trainer_on_save metrics: eval_loss: 0.271018 eval_runtime: 11.0913 eval_samples_per_second: 7.303 eval_steps_per_second: 1.893 eval_perplexity: 1.311298 hf_revision: step-00057 hf_commit: cb4e9fbb49d9ae4a48b123f49378459279fdc0fd - step: 114 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-114 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/sdpo_tooluse/p1_sft_math_tooluse__s43/checkpoint-114 metadata: source: trainer_on_save metrics: eval_loss: 0.148939 eval_runtime: 11.059 eval_samples_per_second: 7.324 eval_steps_per_second: 1.899 eval_perplexity: 1.160602 hf_revision: step-00114 hf_commit: f05f56d523c19166bda3280bb90fc989a0c28f23 wandb_run_id: 4l3ivdio wandb_eval_run_ids: {} hf_repo_id: tkwiecinski/amr-fma-Llama-3.1-8B-Instruct-lora_sft-sdpo_tooluse-p1_sft_math_tooluse-s43