Instructions to use tkwiecinski/amr-fma-Mistral-7B-Instruct-v0.3-lora_sft-math-p1_sft_math_tooluse-s42 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use tkwiecinski/amr-fma-Mistral-7B-Instruct-v0.3-lora_sft-math-p1_sft_math_tooluse-s42 with PEFT:
Task type is invalid.
- Notebooks
- Google Colab
- Kaggle
File size: 6,512 Bytes
4b7ce93 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 | method: lora_sft
base_model_id: mistralai/Mistral-7B-Instruct-v0.3
seed: 42
exp_name: p1_sft_math_tooluse
git_commit: 8b979a30de6dfbf3b5a1052e42d8c0453b214d3f
dataset: DigitalLearningGmbH/MATH-lighteval
dataset_slug: math
manifest_path: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/manifest.yaml
tags:
phase: P1
domain: math
hyperparams:
model:
base_model_id: mistralai/Mistral-7B-Instruct-v0.3
model_family: mistral
target_modules:
- q_proj
- k_proj
- v_proj
- o_proj
- gate_proj
- up_proj
- down_proj
dataset:
name: DigitalLearningGmbH/MATH-lighteval
split: train
text_field: problem
max_samples: null
eval_samples: 256
config: null
domain: math
slug: math
format: math
min_level: 3
sequence:
max_length: 2048
packing: true
lora:
r: 16
alpha: 32
dropout: 0.05
target_modules:
- q_proj
- k_proj
- v_proj
- o_proj
- gate_proj
- up_proj
- down_proj
optimization:
num_train_epochs: 3
per_device_batch_size: 4
gradient_accumulation_steps: 8
learning_rate: 0.0002
warmup_ratio: 0.05
weight_decay: 0.01
lr_scheduler_type: cosine
max_grad_norm: 1.0
checkpointing:
num_checkpoints: 8
save_total_limit: 64
schedule: log
save_steps: null
runtime:
logging_steps: 20
bf16: true
gradient_checkpointing: true
wandb: true
wandb_project: amr-fma-train
hf_push: true
hf_org: tkwiecinski
hf_visibility: public
force_restart: false
sdpo: null
evaluation:
enabled: true
eval_steps: 200
strategy: steps
prompt_style: null
final_adapter_path: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/adapter_final
total_steps: 93
checkpoints:
- step: 1
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-1
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-1
metadata:
source: trainer_on_save
metrics:
eval_loss: 1.108635
eval_runtime: 11.0878
eval_samples_per_second: 3.878
eval_steps_per_second: 0.992
eval_perplexity: 3.030218
hf_revision: step-00001
hf_commit: e76fd3c19b411c5b856aa59ccb3e92d32757f0d2
- step: 3
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-3
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-3
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.950765
eval_runtime: 5.6387
eval_samples_per_second: 7.626
eval_steps_per_second: 1.951
eval_perplexity: 2.587689
hf_revision: step-00003
hf_commit: dac43bb3ac8a2e0c4acef6d26599f0c16cebf105
- step: 6
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-6
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-6
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.866825
eval_runtime: 5.6314
eval_samples_per_second: 7.636
eval_steps_per_second: 1.953
eval_perplexity: 2.379345
hf_revision: step-00006
hf_commit: e2c0177b5bede9c24ac460ebdae7ce6fc2a2c79b
- step: 13
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-13
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-13
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.773847
eval_runtime: 5.6448
eval_samples_per_second: 7.618
eval_steps_per_second: 1.949
eval_perplexity: 2.168091
hf_revision: step-00013
hf_commit: 5be6dd06a400b76e43e0dbf1d046e787eb447450
- step: 25
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-25
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-25
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.737343
eval_runtime: 5.6696
eval_samples_per_second: 7.584
eval_steps_per_second: 1.94
eval_perplexity: 2.090374
hf_revision: step-00025
hf_commit: 6a2530e16479ece0e6696352db96c219b9e33681
- step: 48
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-48
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-48
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.712156
eval_runtime: 5.6717
eval_samples_per_second: 7.582
eval_steps_per_second: 1.939
eval_perplexity: 2.038382
hf_revision: step-00048
hf_commit: 544a8253cd43902acbac07717b162d04b1df32d3
- step: 92
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-92
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-92
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.712564
eval_runtime: 5.6626
eval_samples_per_second: 7.594
eval_steps_per_second: 1.943
eval_perplexity: 2.039213
hf_revision: step-00092
hf_commit: 20990e24539efa1eb4bd32ba09988851e3c4e28e
- step: 93
dir: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-93
artifact: /capstor/scratch/cscs/tkwiecinski/amr-fma/train/Mistral-7B-Instruct-v0.3/lora_sft/math/p1_sft_math_tooluse__s42/checkpoint-93
metadata:
source: trainer_on_save
metrics:
eval_loss: 0.712486
eval_runtime: 5.6564
eval_samples_per_second: 7.602
eval_steps_per_second: 1.945
eval_perplexity: 2.039054
hf_revision: step-00093
hf_commit: f6ac6dfaa1db707aca9796dc79af6620cdd4ce61
wandb_run_id: gz9oxnrb
wandb_eval_run_ids: {}
hf_repo_id: tkwiecinski/amr-fma-Mistral-7B-Instruct-v0.3-lora_sft-math-p1_sft_math_tooluse-s42
|