ygaci commited on
Commit
c8b139e
·
verified ·
1 Parent(s): 64f0d7e

Training in progress, epoch 15, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6dc59b48fb66813ca78955fa135a5b9c5eeb223b35e78585ef359fa563ac2307
3
  size 218121344
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:693b8b9493a1341e036fe175f3c242fb0e70d29b2806414ec7f6d0e7af219541
3
  size 218121344
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dbfbdfc4b321b15c27cff0da422d2a53f2a28df1c3baf392c51ed726dc694b44
3
  size 109417018
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f8e052ac635caaca4a903cc21b648c6ca206f956c8750d678b1b4d52b5329dc6
3
  size 109417018
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:83dc8405814a784ea21d72ace0f4b91d221f004666d3673770e299e5a05e8fe9
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2e87811f0f1effcce9b8f93da83987850561a5f540a92a1eae3e98079fa8bf27
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:30442933123d7669c460d45164eac302a9648bbc2949627d23d3db5a60c61873
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:681b81b0832ded00f66f265bc2327f35108bc92cae631f3148849c5007cbbcc8
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 14.0,
5
  "eval_steps": 500,
6
- "global_step": 21378,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -1498,6 +1498,118 @@
1498
  "learning_rate": 6.151797603195739e-05,
1499
  "loss": 0.1316,
1500
  "step": 21300
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1501
  }
1502
  ],
1503
  "logging_steps": 100,
@@ -1517,7 +1629,7 @@
1517
  "attributes": {}
1518
  }
1519
  },
1520
- "total_flos": 8.048737596945203e+16,
1521
  "train_batch_size": 1,
1522
  "trial_name": null,
1523
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 15.0,
5
  "eval_steps": 500,
6
+ "global_step": 22905,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
1498
  "learning_rate": 6.151797603195739e-05,
1499
  "loss": 0.1316,
1500
  "step": 21300
1501
+ },
1502
+ {
1503
+ "epoch": 14.01440733464309,
1504
+ "grad_norm": 0.38168269395828247,
1505
+ "learning_rate": 6.085219707057257e-05,
1506
+ "loss": 0.126,
1507
+ "step": 21400
1508
+ },
1509
+ {
1510
+ "epoch": 14.079895219384413,
1511
+ "grad_norm": 0.385215699672699,
1512
+ "learning_rate": 6.018641810918775e-05,
1513
+ "loss": 0.1112,
1514
+ "step": 21500
1515
+ },
1516
+ {
1517
+ "epoch": 14.145383104125736,
1518
+ "grad_norm": 0.2507851719856262,
1519
+ "learning_rate": 5.9520639147802933e-05,
1520
+ "loss": 0.1104,
1521
+ "step": 21600
1522
+ },
1523
+ {
1524
+ "epoch": 14.21087098886706,
1525
+ "grad_norm": 0.3379518687725067,
1526
+ "learning_rate": 5.8854860186418116e-05,
1527
+ "loss": 0.117,
1528
+ "step": 21700
1529
+ },
1530
+ {
1531
+ "epoch": 14.276358873608382,
1532
+ "grad_norm": 0.3190099895000458,
1533
+ "learning_rate": 5.818908122503329e-05,
1534
+ "loss": 0.1188,
1535
+ "step": 21800
1536
+ },
1537
+ {
1538
+ "epoch": 14.341846758349705,
1539
+ "grad_norm": 0.4639456570148468,
1540
+ "learning_rate": 5.752330226364847e-05,
1541
+ "loss": 0.1179,
1542
+ "step": 21900
1543
+ },
1544
+ {
1545
+ "epoch": 14.407334643091028,
1546
+ "grad_norm": 0.3332715332508087,
1547
+ "learning_rate": 5.685752330226365e-05,
1548
+ "loss": 0.1167,
1549
+ "step": 22000
1550
+ },
1551
+ {
1552
+ "epoch": 14.472822527832351,
1553
+ "grad_norm": 1.894679307937622,
1554
+ "learning_rate": 5.619174434087883e-05,
1555
+ "loss": 0.1174,
1556
+ "step": 22100
1557
+ },
1558
+ {
1559
+ "epoch": 14.538310412573674,
1560
+ "grad_norm": 0.34946346282958984,
1561
+ "learning_rate": 5.552596537949402e-05,
1562
+ "loss": 0.1185,
1563
+ "step": 22200
1564
+ },
1565
+ {
1566
+ "epoch": 14.603798297314997,
1567
+ "grad_norm": 1.04374098777771,
1568
+ "learning_rate": 5.4860186418109194e-05,
1569
+ "loss": 0.1245,
1570
+ "step": 22300
1571
+ },
1572
+ {
1573
+ "epoch": 14.66928618205632,
1574
+ "grad_norm": 0.311128169298172,
1575
+ "learning_rate": 5.419440745672437e-05,
1576
+ "loss": 0.1291,
1577
+ "step": 22400
1578
+ },
1579
+ {
1580
+ "epoch": 14.734774066797643,
1581
+ "grad_norm": 4.8841962814331055,
1582
+ "learning_rate": 5.352862849533955e-05,
1583
+ "loss": 0.12,
1584
+ "step": 22500
1585
+ },
1586
+ {
1587
+ "epoch": 14.800261951538966,
1588
+ "grad_norm": 0.2752310335636139,
1589
+ "learning_rate": 5.286284953395473e-05,
1590
+ "loss": 0.1203,
1591
+ "step": 22600
1592
+ },
1593
+ {
1594
+ "epoch": 14.865749836280289,
1595
+ "grad_norm": 0.44144880771636963,
1596
+ "learning_rate": 5.2197070572569905e-05,
1597
+ "loss": 0.1212,
1598
+ "step": 22700
1599
+ },
1600
+ {
1601
+ "epoch": 14.931237721021612,
1602
+ "grad_norm": 0.19180706143379211,
1603
+ "learning_rate": 5.153129161118508e-05,
1604
+ "loss": 0.1211,
1605
+ "step": 22800
1606
+ },
1607
+ {
1608
+ "epoch": 14.996725605762935,
1609
+ "grad_norm": 0.24586378037929535,
1610
+ "learning_rate": 5.086551264980027e-05,
1611
+ "loss": 0.1237,
1612
+ "step": 22900
1613
  }
1614
  ],
1615
  "logging_steps": 100,
 
1629
  "attributes": {}
1630
  }
1631
  },
1632
+ "total_flos": 8.62364741318738e+16,
1633
  "train_batch_size": 1,
1634
  "trial_name": null,
1635
  "trial_params": null