surakarteh commited on
Commit
7a6910b
·
verified ·
1 Parent(s): 9bebbdc

Training in progress, step 625, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c4b4037b5f9fbcc84dca7c3d988860d6e70ef85050b5c28ed33315cfda5d2835
3
  size 838921352
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a4fd6cbbb17f17f6437fa8612a06f74d278ba89cebdd9faff1d3fc7c614b3e93
3
  size 838921352
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e882ad42fb2beedaba1a3c1aba2248afcb12cbcf14c3a1a3856e83178eb6e3f3
3
  size 1678106194
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:95b04b67720036c956c001bc3de6219d1785b3de68bacd3c8da40c0f9f04c050
3
  size 1678106194
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dfb91a34779f986cbeae9b321215382343cc2be3e49d06667ae42bf033e1bd3f
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f23342f44cba4bcde06a8513119620eeab9f64e64c8cbc90f344262dce661de9
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e230870b5ac600c7833cdb1034c43fbb5898c6e66e3269fb80a84a631f102d03
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:57b66a7773150afb7a3d7ef310e6a0964d0eeaaa7ed1414bbadf448178a2d162
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": 0.039619311690330505,
3
  "best_model_checkpoint": "miner_id_24/checkpoint-500",
4
- "epoch": 0.7306736811340055,
5
  "eval_steps": 500,
6
- "global_step": 500,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -93,6 +93,20 @@
93
  "eval_samples_per_second": 15.021,
94
  "eval_steps_per_second": 3.011,
95
  "step": 500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
96
  }
97
  ],
98
  "logging_steps": 50,
@@ -116,12 +130,12 @@
116
  "should_evaluate": false,
117
  "should_log": false,
118
  "should_save": true,
119
- "should_training_stop": false
120
  },
121
  "attributes": {}
122
  }
123
  },
124
- "total_flos": 1.481210855424e+18,
125
  "train_batch_size": 5,
126
  "trial_name": null,
127
  "trial_params": null
 
1
  {
2
  "best_metric": 0.039619311690330505,
3
  "best_model_checkpoint": "miner_id_24/checkpoint-500",
4
+ "epoch": 0.9133421014175069,
5
  "eval_steps": 500,
6
+ "global_step": 625,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
93
  "eval_samples_per_second": 15.021,
94
  "eval_steps_per_second": 3.011,
95
  "step": 500
96
+ },
97
+ {
98
+ "epoch": 0.8037410492474061,
99
+ "grad_norm": 0.11217395961284637,
100
+ "learning_rate": 1.45506290208529e-05,
101
+ "loss": 0.0382,
102
+ "step": 550
103
+ },
104
+ {
105
+ "epoch": 0.8768084173608066,
106
+ "grad_norm": 0.057221416383981705,
107
+ "learning_rate": 1.6436065305491224e-06,
108
+ "loss": 0.036,
109
+ "step": 600
110
  }
111
  ],
112
  "logging_steps": 50,
 
130
  "should_evaluate": false,
131
  "should_log": false,
132
  "should_save": true,
133
+ "should_training_stop": true
134
  },
135
  "attributes": {}
136
  }
137
  },
138
+ "total_flos": 1.85151356928e+18,
139
  "train_batch_size": 5,
140
  "trial_name": null,
141
  "trial_params": null