method: lora_sft base_model_id: meta-llama/Llama-3.1-8B-Instruct seed: 43 exp_name: p1_sft_math_tooluse git_commit: 8b979a30de6dfbf3b5a1052e42d8c0453b214d3f dataset: DigitalLearningGmbH/MATH-lighteval dataset_slug: math manifest_path: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/manifest.yaml tags: phase: P1 domain: math hyperparams: model: base_model_id: meta-llama/Llama-3.1-8B-Instruct model_family: llama3 target_modules: - q_proj - k_proj - v_proj - o_proj - gate_proj - up_proj - down_proj dataset: name: DigitalLearningGmbH/MATH-lighteval split: train text_field: problem max_samples: null eval_samples: 256 config: null domain: math slug: math format: math min_level: 3 sequence: max_length: 2048 packing: true lora: r: 16 alpha: 32 dropout: 0.05 target_modules: - q_proj - k_proj - v_proj - o_proj - gate_proj - up_proj - down_proj optimization: num_train_epochs: 3 per_device_batch_size: 4 gradient_accumulation_steps: 8 learning_rate: 0.0002 warmup_ratio: 0.05 weight_decay: 0.01 lr_scheduler_type: cosine max_grad_norm: 1.0 checkpointing: num_checkpoints: 8 save_total_limit: 64 schedule: log save_steps: null runtime: logging_steps: 20 bf16: true gradient_checkpointing: true wandb: true wandb_project: amr-fma-train hf_push: true hf_org: tkwiecinski hf_visibility: public force_restart: false sdpo: null evaluation: enabled: true eval_steps: 200 strategy: steps prompt_style: null final_adapter_path: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/adapter_final total_steps: 84 checkpoints: - step: 1 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-1 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-1 metadata: source: trainer_on_save metrics: eval_loss: 0.906069 eval_runtime: 10.8644 eval_samples_per_second: 3.682 eval_steps_per_second: 0.92 eval_perplexity: 2.474577 hf_revision: step-00001 hf_commit: 93369e04bcd2c8f445d76b6f91c53aabeb0025a2 - step: 3 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-3 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-3 metadata: source: trainer_on_save metrics: eval_loss: 0.85506 eval_runtime: 5.4921 eval_samples_per_second: 7.283 eval_steps_per_second: 1.821 eval_perplexity: 2.351514 hf_revision: step-00003 hf_commit: c67c55147c2788ceb0059134aa28deed981217b0 - step: 6 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-6 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-6 metadata: source: trainer_on_save metrics: eval_loss: 0.736943 eval_runtime: 5.4901 eval_samples_per_second: 7.286 eval_steps_per_second: 1.821 eval_perplexity: 2.089539 hf_revision: step-00006 hf_commit: b1bb2b81d6dedc765dddabe0777dc795f88cad36 - step: 12 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-12 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-12 metadata: source: trainer_on_save metrics: eval_loss: 0.683851 eval_runtime: 5.5033 eval_samples_per_second: 7.268 eval_steps_per_second: 1.817 eval_perplexity: 1.981494 hf_revision: step-00012 hf_commit: d269a3b6c2b27da40dda14c1df125319c3e3b69f - step: 23 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-23 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-23 metadata: source: trainer_on_save metrics: eval_loss: 0.664902 eval_runtime: 5.5029 eval_samples_per_second: 7.269 eval_steps_per_second: 1.817 eval_perplexity: 1.9443 hf_revision: step-00023 hf_commit: ff8a843bc1a9139aa4e98f0961a0291b187abdb3 - step: 44 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-44 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-44 metadata: source: trainer_on_save metrics: eval_loss: 0.646989 eval_runtime: 5.5026 eval_samples_per_second: 7.269 eval_steps_per_second: 1.817 eval_perplexity: 1.909782 hf_revision: step-00044 hf_commit: 4b5ace23c93175428314d3ff94dacbc692e42759 - step: 83 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-83 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-83 metadata: source: trainer_on_save metrics: eval_loss: 0.643062 eval_runtime: 5.5171 eval_samples_per_second: 7.25 eval_steps_per_second: 1.813 eval_perplexity: 1.902297 hf_revision: step-00083 hf_commit: 609e954eca821e70fc5e1f402d84cb9b0571059b - step: 84 dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-84 artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Llama-3.1-8B-Instruct/lora_sft/math/p1_sft_math_tooluse__s43/checkpoint-84 metadata: source: trainer_on_save metrics: eval_loss: 0.643025 eval_runtime: 5.521 eval_samples_per_second: 7.245 eval_steps_per_second: 1.811 eval_perplexity: 1.902227 hf_revision: step-00084 hf_commit: adf6402b3230648bc821a86ec38d43b6961f4b8a wandb_run_id: co3n04yu wandb_eval_run_ids: {} hf_repo_id: tkwiecinski/amr-fma-Llama-3.1-8B-Instruct-lora_sft-math-p1_sft_math_tooluse-s43