ygaci commited on
Commit
ded7176
·
verified ·
1 Parent(s): 05d6ec1

Training in progress, epoch 14, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ca3bdfcc66def65d0f604e01092775bb8047ff54f897ac825a782dab0065a2e0
3
  size 218121344
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6dc59b48fb66813ca78955fa135a5b9c5eeb223b35e78585ef359fa563ac2307
3
  size 218121344
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1df3e4d97e5340b4a3c3b62168ce0a7e7d863568e262a8f917461eafc07a464b
3
  size 109417018
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dbfbdfc4b321b15c27cff0da422d2a53f2a28df1c3baf392c51ed726dc694b44
3
  size 109417018
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f4d452003ffd26a609202b256e3dbe796c6c31379217f192c3635c6946486885
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:83dc8405814a784ea21d72ace0f4b91d221f004666d3673770e299e5a05e8fe9
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f69879333db32fd09e66a8372d09140f268c99379f91f2de1d978c4c8764223c
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:30442933123d7669c460d45164eac302a9648bbc2949627d23d3db5a60c61873
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 13.0,
5
  "eval_steps": 500,
6
- "global_step": 19851,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -1393,6 +1393,111 @@
1393
  "learning_rate": 7.15046604527297e-05,
1394
  "loss": 0.1359,
1395
  "step": 19800
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1396
  }
1397
  ],
1398
  "logging_steps": 100,
@@ -1412,7 +1517,7 @@
1412
  "attributes": {}
1413
  }
1414
  },
1415
- "total_flos": 7.473827766861824e+16,
1416
  "train_batch_size": 1,
1417
  "trial_name": null,
1418
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 14.0,
5
  "eval_steps": 500,
6
+ "global_step": 21378,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
1393
  "learning_rate": 7.15046604527297e-05,
1394
  "loss": 0.1359,
1395
  "step": 19800
1396
+ },
1397
+ {
1398
+ "epoch": 13.032089063523248,
1399
+ "grad_norm": 0.3909270167350769,
1400
+ "learning_rate": 7.083888149134487e-05,
1401
+ "loss": 0.123,
1402
+ "step": 19900
1403
+ },
1404
+ {
1405
+ "epoch": 13.097576948264571,
1406
+ "grad_norm": 0.3292842209339142,
1407
+ "learning_rate": 7.017310252996006e-05,
1408
+ "loss": 0.1235,
1409
+ "step": 20000
1410
+ },
1411
+ {
1412
+ "epoch": 13.163064833005894,
1413
+ "grad_norm": 1.1147010326385498,
1414
+ "learning_rate": 6.950732356857524e-05,
1415
+ "loss": 0.1157,
1416
+ "step": 20100
1417
+ },
1418
+ {
1419
+ "epoch": 13.228552717747217,
1420
+ "grad_norm": 0.22612357139587402,
1421
+ "learning_rate": 6.884154460719041e-05,
1422
+ "loss": 0.1228,
1423
+ "step": 20200
1424
+ },
1425
+ {
1426
+ "epoch": 13.29404060248854,
1427
+ "grad_norm": 0.2340671569108963,
1428
+ "learning_rate": 6.81757656458056e-05,
1429
+ "loss": 0.1262,
1430
+ "step": 20300
1431
+ },
1432
+ {
1433
+ "epoch": 13.359528487229863,
1434
+ "grad_norm": 0.470034658908844,
1435
+ "learning_rate": 6.750998668442078e-05,
1436
+ "loss": 0.1223,
1437
+ "step": 20400
1438
+ },
1439
+ {
1440
+ "epoch": 13.425016371971186,
1441
+ "grad_norm": 1.8971284627914429,
1442
+ "learning_rate": 6.684420772303596e-05,
1443
+ "loss": 0.1256,
1444
+ "step": 20500
1445
+ },
1446
+ {
1447
+ "epoch": 13.490504256712509,
1448
+ "grad_norm": 0.32659152150154114,
1449
+ "learning_rate": 6.617842876165113e-05,
1450
+ "loss": 0.128,
1451
+ "step": 20600
1452
+ },
1453
+ {
1454
+ "epoch": 13.555992141453832,
1455
+ "grad_norm": 0.4042024314403534,
1456
+ "learning_rate": 6.551264980026631e-05,
1457
+ "loss": 0.1236,
1458
+ "step": 20700
1459
+ },
1460
+ {
1461
+ "epoch": 13.621480026195155,
1462
+ "grad_norm": 0.37899136543273926,
1463
+ "learning_rate": 6.484687083888148e-05,
1464
+ "loss": 0.1229,
1465
+ "step": 20800
1466
+ },
1467
+ {
1468
+ "epoch": 13.686967910936477,
1469
+ "grad_norm": 0.23778092861175537,
1470
+ "learning_rate": 6.418109187749667e-05,
1471
+ "loss": 0.1255,
1472
+ "step": 20900
1473
+ },
1474
+ {
1475
+ "epoch": 13.7524557956778,
1476
+ "grad_norm": 0.637147843837738,
1477
+ "learning_rate": 6.351531291611186e-05,
1478
+ "loss": 0.127,
1479
+ "step": 21000
1480
+ },
1481
+ {
1482
+ "epoch": 13.817943680419123,
1483
+ "grad_norm": 0.5459991693496704,
1484
+ "learning_rate": 6.284953395472704e-05,
1485
+ "loss": 0.1203,
1486
+ "step": 21100
1487
+ },
1488
+ {
1489
+ "epoch": 13.883431565160445,
1490
+ "grad_norm": 0.516917884349823,
1491
+ "learning_rate": 6.218375499334222e-05,
1492
+ "loss": 0.1248,
1493
+ "step": 21200
1494
+ },
1495
+ {
1496
+ "epoch": 13.948919449901767,
1497
+ "grad_norm": 0.34805047512054443,
1498
+ "learning_rate": 6.151797603195739e-05,
1499
+ "loss": 0.1316,
1500
+ "step": 21300
1501
  }
1502
  ],
1503
  "logging_steps": 100,
 
1517
  "attributes": {}
1518
  }
1519
  },
1520
+ "total_flos": 8.048737596945203e+16,
1521
  "train_batch_size": 1,
1522
  "trial_name": null,
1523
  "trial_params": null