tkwiecinski's picture
Finalize run summary on main
4b7ce93 verified
Raw
History Blame Contribute Delete
6.51 kB
method: lora_sft
base_model_id: mistralai/Mistral-7B-Instruct-v0.3
seed: 42
exp_name: p1_sft_math_tooluse
git_commit: 8b979a30de6dfbf3b5a1052e42d8c0453b214d3f
dataset: DigitalLearningGmbH/MATH-lighteval
dataset_slug: math
manifest_path: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/manifest.yaml
tags:
phase: P1
domain: math
hyperparams:
model:
base_model_id: mistralai/Mistral-7B-Instruct-v0.3
model_family: mistral
target_modules:
- q_proj
- k_proj
- v_proj
- o_proj
- gate_proj
- up_proj
- down_proj
dataset:
name: DigitalLearningGmbH/MATH-lighteval
split: train
text_field: problem
max_samples: null
eval_samples: 256
config: null
domain: math
slug: math
format: math
min_level: 3
sequence:
max_length: 2048
packing: true
lora:
r: 16
alpha: 32
dropout: 0.05
target_modules:
- q_proj
- k_proj
- v_proj
- o_proj
- gate_proj
- up_proj
- down_proj
optimization:
num_train_epochs: 3
per_device_batch_size: 4
gradient_accumulation_steps: 8
learning_rate: 0.0002
warmup_ratio: 0.05
weight_decay: 0.01
lr_scheduler_type: cosine
max_grad_norm: 1.0
checkpointing:
num_checkpoints: 8
save_total_limit: 64
schedule: log
save_steps: null
runtime:
logging_steps: 20
bf16: true
gradient_checkpointing: true
wandb: true
wandb_project: amr-fma-train
hf_push: true
hf_org: tkwiecinski
hf_visibility: public
force_restart: false
sdpo: null
evaluation:
enabled: true
eval_steps: 200
strategy: steps
prompt_style: null
final_adapter_path: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/adapter_final
total_steps: 93
checkpoints:
- step: 1
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-1
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-1
metadata:
source: trainer_on_save
metrics:
eval_loss: 1.108635
eval_runtime: 11.0878
eval_samples_per_second: 3.878
eval_steps_per_second: 0.992
eval_perplexity: 3.030218
hf_revision: step-00001
hf_commit: e76fd3c19b411c5b856aa59ccb3e92d32757f0d2
- step: 3
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-3
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-3
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.950765
eval_runtime: 5.6387
eval_samples_per_second: 7.626
eval_steps_per_second: 1.951
eval_perplexity: 2.587689
hf_revision: step-00003
hf_commit: dac43bb3ac8a2e0c4acef6d26599f0c16cebf105
- step: 6
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-6
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-6
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.866825
eval_runtime: 5.6314
eval_samples_per_second: 7.636
eval_steps_per_second: 1.953
eval_perplexity: 2.379345
hf_revision: step-00006
hf_commit: e2c0177b5bede9c24ac460ebdae7ce6fc2a2c79b
- step: 13
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-13
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-13
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.773847
eval_runtime: 5.6448
eval_samples_per_second: 7.618
eval_steps_per_second: 1.949
eval_perplexity: 2.168091
hf_revision: step-00013
hf_commit: 5be6dd06a400b76e43e0dbf1d046e787eb447450
- step: 25
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-25
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-25
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.737343
eval_runtime: 5.6696
eval_samples_per_second: 7.584
eval_steps_per_second: 1.94
eval_perplexity: 2.090374
hf_revision: step-00025
hf_commit: 6a2530e16479ece0e6696352db96c219b9e33681
- step: 48
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-48
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-48
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.712156
eval_runtime: 5.6717
eval_samples_per_second: 7.582
eval_steps_per_second: 1.939
eval_perplexity: 2.038382
hf_revision: step-00048
hf_commit: 544a8253cd43902acbac07717b162d04b1df32d3
- step: 92
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-92
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-92
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.712564
eval_runtime: 5.6626
eval_samples_per_second: 7.594
eval_steps_per_second: 1.943
eval_perplexity: 2.039213
hf_revision: step-00092
hf_commit: 20990e24539efa1eb4bd32ba09988851e3c4e28e
- step: 93
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-93
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-93
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.712486
eval_runtime: 5.6564
eval_samples_per_second: 7.602
eval_steps_per_second: 1.945
eval_perplexity: 2.039054
hf_revision: step-00093
hf_commit: f6ac6dfaa1db707aca9796dc79af6620cdd4ce61
wandb_run_id: gz9oxnrb
wandb_eval_run_ids: {}
hf_repo_id: tkwiecinski/amr-fma-Mistral-7B-Instruct-v0.3-lora_sft-math-p1_sft_math_tooluse-s42