ajtaltarabukin2022 commited on
Commit
e8408df
·
verified ·
1 Parent(s): 0580bbd

Training in progress, step 30, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:15a5e61507b150febec91fb65a139aac52c422f21588bbee7447873070abe68e
3
  size 217931936
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dce0d09f03f3f156d6b66940c39f98bed51ef2184d82a0ae1be38cece13978bf
3
  size 217931936
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:02d043b0a210d45339a3fea4cd83be346b5722565a18c85f7b05acb35a7e9abc
3
  size 436251442
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d0099eaa441e129ed7829e977ce4e8981eef84e2f953f4d6461f45d7305fae75
3
  size 436251442
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3b800b6d302bd5c5bc6607adcd64c98c8846be4c2323ad836fe3e601910de836
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5393cc6a1817757f20fd81e8db25d70ba44d676dca37db29d40af3d861e60ffc
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3e2686563472cd649084082a2c329fd627ce86023a87d1ac3f5dd126e100dfda
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0ec942292267bb06cc495f28f765abdfdf101d23971c1de481c4ea2a744d43b7
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 0.02154591974144896,
5
  "eval_steps": 5,
6
- "global_step": 20,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -89,6 +89,50 @@
89
  "eval_samples_per_second": 9.189,
90
  "eval_steps_per_second": 4.606,
91
  "step": 20
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
92
  }
93
  ],
94
  "logging_steps": 3,
@@ -108,7 +152,7 @@
108
  "attributes": {}
109
  }
110
  },
111
- "total_flos": 9217105396236288.0,
112
  "train_batch_size": 2,
113
  "trial_name": null,
114
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 0.032318879612173446,
5
  "eval_steps": 5,
6
+ "global_step": 30,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
89
  "eval_samples_per_second": 9.189,
90
  "eval_steps_per_second": 4.606,
91
  "step": 20
92
+ },
93
+ {
94
+ "epoch": 0.022623215728521412,
95
+ "grad_norm": 0.6983510851860046,
96
+ "learning_rate": 0.00014067366430758004,
97
+ "loss": 0.6017,
98
+ "step": 21
99
+ },
100
+ {
101
+ "epoch": 0.025855103689738757,
102
+ "grad_norm": 0.9189296960830688,
103
+ "learning_rate": 0.00011045284632676536,
104
+ "loss": 0.6411,
105
+ "step": 24
106
+ },
107
+ {
108
+ "epoch": 0.026932399676811204,
109
+ "eval_loss": 0.6164190173149109,
110
+ "eval_runtime": 42.5383,
111
+ "eval_samples_per_second": 9.192,
112
+ "eval_steps_per_second": 4.608,
113
+ "step": 25
114
+ },
115
+ {
116
+ "epoch": 0.0290869916509561,
117
+ "grad_norm": 0.5284349918365479,
118
+ "learning_rate": 7.920883091822408e-05,
119
+ "loss": 0.506,
120
+ "step": 27
121
+ },
122
+ {
123
+ "epoch": 0.032318879612173446,
124
+ "grad_norm": 0.6442975401878357,
125
+ "learning_rate": 5.000000000000002e-05,
126
+ "loss": 0.5145,
127
+ "step": 30
128
+ },
129
+ {
130
+ "epoch": 0.032318879612173446,
131
+ "eval_loss": 0.597037672996521,
132
+ "eval_runtime": 42.633,
133
+ "eval_samples_per_second": 9.171,
134
+ "eval_steps_per_second": 4.597,
135
+ "step": 30
136
  }
137
  ],
138
  "logging_steps": 3,
 
152
  "attributes": {}
153
  }
154
  },
155
+ "total_flos": 1.377268622426112e+16,
156
  "train_batch_size": 2,
157
  "trial_name": null,
158
  "trial_params": null