ygaci commited on
Commit
c41c807
·
verified ·
1 Parent(s): 397f8a3

Training in progress, epoch 13, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3dbe4fcf585d74b0e916899724070c59a3cb8d2474f6f410d2870c72650b35c0
3
  size 218121344
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ca3bdfcc66def65d0f604e01092775bb8047ff54f897ac825a782dab0065a2e0
3
  size 218121344
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8adfeb763cfbe1cede3723bc945c0f7c532cd6940bf78aecd344894ff05f91f7
3
  size 109417018
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1df3e4d97e5340b4a3c3b62168ce0a7e7d863568e262a8f917461eafc07a464b
3
  size 109417018
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:146c78d801fa79ef2403d0b4210ad68d562ba03db24c25eb1431ebd4394b2a57
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f4d452003ffd26a609202b256e3dbe796c6c31379217f192c3635c6946486885
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e40a841847056be2f04abe72731f87a48fd8fa403be07c0dd5087b2da5fb38d8
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f69879333db32fd09e66a8372d09140f268c99379f91f2de1d978c4c8764223c
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 12.0,
5
  "eval_steps": 500,
6
- "global_step": 18324,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -1288,6 +1288,111 @@
1288
  "learning_rate": 8.1491344873502e-05,
1289
  "loss": 0.1445,
1290
  "step": 18300
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1291
  }
1292
  ],
1293
  "logging_steps": 100,
@@ -1307,7 +1412,7 @@
1307
  "attributes": {}
1308
  }
1309
  },
1310
- "total_flos": 6.898917943489331e+16,
1311
  "train_batch_size": 1,
1312
  "trial_name": null,
1313
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 13.0,
5
  "eval_steps": 500,
6
+ "global_step": 19851,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
1288
  "learning_rate": 8.1491344873502e-05,
1289
  "loss": 0.1445,
1290
  "step": 18300
1291
+ },
1292
+ {
1293
+ "epoch": 12.049770792403406,
1294
+ "grad_norm": 0.48655104637145996,
1295
+ "learning_rate": 8.082556591211719e-05,
1296
+ "loss": 0.1228,
1297
+ "step": 18400
1298
+ },
1299
+ {
1300
+ "epoch": 12.115258677144729,
1301
+ "grad_norm": 0.5396549105644226,
1302
+ "learning_rate": 8.015978695073236e-05,
1303
+ "loss": 0.1227,
1304
+ "step": 18500
1305
+ },
1306
+ {
1307
+ "epoch": 12.180746561886052,
1308
+ "grad_norm": 0.20431631803512573,
1309
+ "learning_rate": 7.949400798934754e-05,
1310
+ "loss": 0.1281,
1311
+ "step": 18600
1312
+ },
1313
+ {
1314
+ "epoch": 12.246234446627374,
1315
+ "grad_norm": 0.40931564569473267,
1316
+ "learning_rate": 7.882822902796272e-05,
1317
+ "loss": 0.129,
1318
+ "step": 18700
1319
+ },
1320
+ {
1321
+ "epoch": 12.311722331368697,
1322
+ "grad_norm": 0.3679026961326599,
1323
+ "learning_rate": 7.81624500665779e-05,
1324
+ "loss": 0.1291,
1325
+ "step": 18800
1326
+ },
1327
+ {
1328
+ "epoch": 12.37721021611002,
1329
+ "grad_norm": 0.3819580674171448,
1330
+ "learning_rate": 7.749667110519307e-05,
1331
+ "loss": 0.126,
1332
+ "step": 18900
1333
+ },
1334
+ {
1335
+ "epoch": 12.442698100851343,
1336
+ "grad_norm": 0.4266682267189026,
1337
+ "learning_rate": 7.683089214380826e-05,
1338
+ "loss": 0.1293,
1339
+ "step": 19000
1340
+ },
1341
+ {
1342
+ "epoch": 12.508185985592664,
1343
+ "grad_norm": 0.22216545045375824,
1344
+ "learning_rate": 7.616511318242345e-05,
1345
+ "loss": 0.1338,
1346
+ "step": 19100
1347
+ },
1348
+ {
1349
+ "epoch": 12.573673870333987,
1350
+ "grad_norm": 0.4221850037574768,
1351
+ "learning_rate": 7.549933422103862e-05,
1352
+ "loss": 0.1339,
1353
+ "step": 19200
1354
+ },
1355
+ {
1356
+ "epoch": 12.63916175507531,
1357
+ "grad_norm": 0.5264990925788879,
1358
+ "learning_rate": 7.48335552596538e-05,
1359
+ "loss": 0.1355,
1360
+ "step": 19300
1361
+ },
1362
+ {
1363
+ "epoch": 12.704649639816633,
1364
+ "grad_norm": 0.653891384601593,
1365
+ "learning_rate": 7.416777629826898e-05,
1366
+ "loss": 0.1317,
1367
+ "step": 19400
1368
+ },
1369
+ {
1370
+ "epoch": 12.770137524557956,
1371
+ "grad_norm": 1.4240955114364624,
1372
+ "learning_rate": 7.350199733688415e-05,
1373
+ "loss": 0.1297,
1374
+ "step": 19500
1375
+ },
1376
+ {
1377
+ "epoch": 12.83562540929928,
1378
+ "grad_norm": 0.33510828018188477,
1379
+ "learning_rate": 7.283621837549934e-05,
1380
+ "loss": 0.1378,
1381
+ "step": 19600
1382
+ },
1383
+ {
1384
+ "epoch": 12.901113294040602,
1385
+ "grad_norm": 0.31254374980926514,
1386
+ "learning_rate": 7.217043941411452e-05,
1387
+ "loss": 0.1291,
1388
+ "step": 19700
1389
+ },
1390
+ {
1391
+ "epoch": 12.966601178781925,
1392
+ "grad_norm": 0.5154318809509277,
1393
+ "learning_rate": 7.15046604527297e-05,
1394
+ "loss": 0.1359,
1395
+ "step": 19800
1396
  }
1397
  ],
1398
  "logging_steps": 100,
 
1412
  "attributes": {}
1413
  }
1414
  },
1415
+ "total_flos": 7.473827766861824e+16,
1416
  "train_batch_size": 1,
1417
  "trial_name": null,
1418
  "trial_params": null