tianzl66's picture
Release audited B300 LoRA training seed 44; actual zero warmup and complete in-domain HNS results
ffaca6a verified
Raw History Blame Contribute Delete
12.3 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0,
"eval_steps": 500,
"global_step": 1561,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.016020506247997435,
"grad_norm": 0.29081395268440247,
"learning_rate": 1.9988337222482776e-05,
"loss": 1.1990239715576172,
"step": 25
},
{
"epoch": 0.03204101249599487,
"grad_norm": 0.11618245393037796,
"learning_rate": 1.9951414786166656e-05,
"loss": 0.9829273986816406,
"step": 50
},
{
"epoch": 0.04806151874399231,
"grad_norm": 0.10595086961984634,
"learning_rate": 1.988930588781978e-05,
"loss": 0.92435546875,
"step": 75
},
{
"epoch": 0.06408202499198974,
"grad_norm": 0.09951659291982651,
"learning_rate": 1.9802167721513908e-05,
"loss": 0.9162257385253906,
"step": 100
},
{
"epoch": 0.08010253123998719,
"grad_norm": 0.08442762494087219,
"learning_rate": 1.9690220828967406e-05,
"loss": 0.8908739471435547,
"step": 125
},
{
"epoch": 0.09612303748798462,
"grad_norm": 0.10172899067401886,
"learning_rate": 1.9553748541366756e-05,
"loss": 0.9406523895263672,
"step": 150
},
{
"epoch": 0.11214354373598205,
"grad_norm": 0.11054546386003494,
"learning_rate": 1.9393096262271533e-05,
"loss": 0.9146512603759765,
"step": 175
},
{
"epoch": 0.12816404998397948,
"grad_norm": 0.10347191989421844,
"learning_rate": 1.9208670593417723e-05,
"loss": 0.9219544219970703,
"step": 200
},
{
"epoch": 0.14418455623197693,
"grad_norm": 0.10463545471429825,
"learning_rate": 1.9000938305631975e-05,
"loss": 0.9444712829589844,
"step": 225
},
{
"epoch": 0.16020506247997437,
"grad_norm": 0.10991481691598892,
"learning_rate": 1.877042515746132e-05,
"loss": 0.9139762115478516,
"step": 250
},
{
"epoch": 0.1762255687279718,
"grad_norm": 0.12124437093734741,
"learning_rate": 1.8517714564508342e-05,
"loss": 0.9185218048095704,
"step": 275
},
{
"epoch": 0.19224607497596924,
"grad_norm": 0.10786885023117065,
"learning_rate": 1.824344612283962e-05,
"loss": 0.9175537872314453,
"step": 300
},
{
"epoch": 0.20826658122396668,
"grad_norm": 0.11486810445785522,
"learning_rate": 1.7948313990204654e-05,
"loss": 0.8929530334472656,
"step": 325
},
{
"epoch": 0.2242870874719641,
"grad_norm": 0.11425229907035828,
"learning_rate": 1.7633065129162282e-05,
"loss": 0.8928473663330078,
"step": 350
},
{
"epoch": 0.24030759371996155,
"grad_norm": 0.11112677305936813,
"learning_rate": 1.729849741656112e-05,
"loss": 0.9173651123046875,
"step": 375
},
{
"epoch": 0.25632809996795897,
"grad_norm": 0.1455482393503189,
"learning_rate": 1.694545762415887e-05,
"loss": 0.9193742370605469,
"step": 400
},
{
"epoch": 0.2723486062159564,
"grad_norm": 0.14621497690677643,
"learning_rate": 1.657483927549141e-05,
"loss": 0.8958631134033204,
"step": 425
},
{
"epoch": 0.28836911246395386,
"grad_norm": 0.15076446533203125,
"learning_rate": 1.6187580384415785e-05,
"loss": 0.9202049255371094,
"step": 450
},
{
"epoch": 0.3043896187119513,
"grad_norm": 0.13701656460762024,
"learning_rate": 1.5784661081050744e-05,
"loss": 0.9259297943115234,
"step": 475
},
{
"epoch": 0.32041012495994875,
"grad_norm": 0.14176012575626373,
"learning_rate": 1.536710113112345e-05,
"loss": 0.9286505126953125,
"step": 500
},
{
"epoch": 0.3364306312079462,
"grad_norm": 0.1571187525987625,
"learning_rate": 1.4935957355000695e-05,
"loss": 0.9105192565917969,
"step": 525
},
{
"epoch": 0.3524511374559436,
"grad_norm": 0.13166415691375732,
"learning_rate": 1.4492320952936956e-05,
"loss": 0.9180181121826172,
"step": 550
},
{
"epoch": 0.36847164370394103,
"grad_norm": 0.1260545551776886,
"learning_rate": 1.403731474330893e-05,
"loss": 0.8934297943115235,
"step": 575
},
{
"epoch": 0.3844921499519385,
"grad_norm": 0.15511398017406464,
"learning_rate": 1.3572090320826396e-05,
"loss": 0.9190473175048828,
"step": 600
},
{
"epoch": 0.4005126561999359,
"grad_norm": 0.1460423320531845,
"learning_rate": 1.3097825141911824e-05,
"loss": 0.9066439056396485,
"step": 625
},
{
"epoch": 0.41653316244793337,
"grad_norm": 0.17855304479599,
"learning_rate": 1.261571954462547e-05,
"loss": 0.899425277709961,
"step": 650
},
{
"epoch": 0.4325536686959308,
"grad_norm": 0.16846759617328644,
"learning_rate": 1.2126993710678361e-05,
"loss": 0.9050975799560547,
"step": 675
},
{
"epoch": 0.4485741749439282,
"grad_norm": 0.16906635463237762,
"learning_rate": 1.1632884577222106e-05,
"loss": 0.9127249145507812,
"step": 700
},
{
"epoch": 0.46459468119192565,
"grad_norm": 0.15867382287979126,
"learning_rate": 1.1134642706231687e-05,
"loss": 0.9023185729980469,
"step": 725
},
{
"epoch": 0.4806151874399231,
"grad_norm": 0.16034585237503052,
"learning_rate": 1.0633529119404571e-05,
"loss": 0.9134428405761719,
"step": 750
},
{
"epoch": 0.49663569368792054,
"grad_norm": 0.16358235478401184,
"learning_rate": 1.013081210658687e-05,
"loss": 0.8995558166503906,
"step": 775
},
{
"epoch": 0.5126561999359179,
"grad_norm": 0.15758606791496277,
"learning_rate": 9.627764015804225e-06,
"loss": 0.8870352172851562,
"step": 800
},
{
"epoch": 0.5286767061839154,
"grad_norm": 0.16830310225486755,
"learning_rate": 9.125658033021683e-06,
"loss": 0.8971255493164062,
"step": 825
},
{
"epoch": 0.5446972124319128,
"grad_norm": 0.16267818212509155,
"learning_rate": 8.6257649597828e-06,
"loss": 0.9000907135009766,
"step": 850
},
{
"epoch": 0.5607177186799103,
"grad_norm": 0.15891553461551666,
"learning_rate": 8.129349996883551e-06,
"loss": 0.899490966796875,
"step": 875
},
{
"epoch": 0.5767382249279077,
"grad_norm": 0.14320583641529083,
"learning_rate": 7.637669542221446e-06,
"loss": 0.8845912933349609,
"step": 900
},
{
"epoch": 0.5927587311759052,
"grad_norm": 0.16313546895980835,
"learning_rate": 7.1519680109242486e-06,
"loss": 0.9202222442626953,
"step": 925
},
{
"epoch": 0.6087792374239026,
"grad_norm": 0.17441782355308533,
"learning_rate": 6.673474685806436e-06,
"loss": 0.9076329803466797,
"step": 950
},
{
"epoch": 0.6247997436719,
"grad_norm": 0.14890457689762115,
"learning_rate": 6.2034006061246295e-06,
"loss": 0.9191471862792969,
"step": 975
},
{
"epoch": 0.6408202499198975,
"grad_norm": 0.17238198220729828,
"learning_rate": 5.742935502506485e-06,
"loss": 0.9056331634521484,
"step": 1000
},
{
"epoch": 0.656840756167895,
"grad_norm": 0.1785092055797577,
"learning_rate": 5.293244785810452e-06,
"loss": 0.9191907501220703,
"step": 1025
},
{
"epoch": 0.6728612624158924,
"grad_norm": 0.15417782962322235,
"learning_rate": 4.855466597537515e-06,
"loss": 0.9085941314697266,
"step": 1050
},
{
"epoch": 0.6888817686638897,
"grad_norm": 0.1511709988117218,
"learning_rate": 4.430708929260105e-06,
"loss": 0.9095938110351562,
"step": 1075
},
{
"epoch": 0.7049022749118872,
"grad_norm": 0.1562037169933319,
"learning_rate": 4.0200468183587556e-06,
"loss": 0.8718026733398437,
"step": 1100
},
{
"epoch": 0.7209227811598846,
"grad_norm": 0.16901475191116333,
"learning_rate": 3.624519627163946e-06,
"loss": 0.8942552947998047,
"step": 1125
},
{
"epoch": 0.7369432874078821,
"grad_norm": 0.15361268818378448,
"learning_rate": 3.24512841238944e-06,
"loss": 0.923448715209961,
"step": 1150
},
{
"epoch": 0.7529637936558795,
"grad_norm": 0.166425421833992,
"learning_rate": 2.8828333915149675e-06,
"loss": 0.9021080017089844,
"step": 1175
},
{
"epoch": 0.768984299903877,
"grad_norm": 0.17839516699314117,
"learning_rate": 2.5385515125306683e-06,
"loss": 0.9276787567138672,
"step": 1200
},
{
"epoch": 0.7850048061518744,
"grad_norm": 0.17206133902072906,
"learning_rate": 2.213154133194122e-06,
"loss": 0.9214431762695312,
"step": 1225
},
{
"epoch": 0.8010253123998718,
"grad_norm": 0.1896800845861435,
"learning_rate": 1.907464815673702e-06,
"loss": 0.9126302337646485,
"step": 1250
},
{
"epoch": 0.8170458186478693,
"grad_norm": 0.1757972687482834,
"learning_rate": 1.622257242159756e-06,
"loss": 0.9005342102050782,
"step": 1275
},
{
"epoch": 0.8330663248958667,
"grad_norm": 0.1621798872947693,
"learning_rate": 1.3582532567192509e-06,
"loss": 0.8823390960693359,
"step": 1300
},
{
"epoch": 0.8490868311438642,
"grad_norm": 0.15191422402858734,
"learning_rate": 1.1161210383496479e-06,
"loss": 0.9226659393310547,
"step": 1325
},
{
"epoch": 0.8651073373918616,
"grad_norm": 0.18351206183433533,
"learning_rate": 8.964734098561001e-07,
"loss": 0.9057015228271484,
"step": 1350
},
{
"epoch": 0.881127843639859,
"grad_norm": 0.16578508913516998,
"learning_rate": 6.998662868319139e-07,
"loss": 0.8983650970458984,
"step": 1375
},
{
"epoch": 0.8971483498878564,
"grad_norm": 0.17020930349826813,
"learning_rate": 5.267972706679991e-07,
"loss": 0.9050147247314453,
"step": 1400
},
{
"epoch": 0.9131688561358539,
"grad_norm": 0.17416594922542572,
"learning_rate": 3.777043891521559e-07,
"loss": 0.8866185760498047,
"step": 1425
},
{
"epoch": 0.9291893623838513,
"grad_norm": 0.16963180899620056,
"learning_rate": 2.529649878457985e-07,
"loss": 0.9323723602294922,
"step": 1450
},
{
"epoch": 0.9452098686318487,
"grad_norm": 0.18463164567947388,
"learning_rate": 1.528947750439036e-07,
"loss": 0.9038640594482422,
"step": 1475
},
{
"epoch": 0.9612303748798462,
"grad_norm": 0.1755848526954651,
"learning_rate": 7.774702273535938e-08,
"loss": 0.904787826538086,
"step": 1500
},
{
"epoch": 0.9772508811278436,
"grad_norm": 0.1582230031490326,
"learning_rate": 2.77119255860403e-08,
"loss": 0.8763291168212891,
"step": 1525
},
{
"epoch": 0.9932713873758411,
"grad_norm": 0.20413701236248016,
"learning_rate": 2.916119566973574e-09,
"loss": 0.920826416015625,
"step": 1550
}
],
"logging_steps": 25,
"max_steps": 1561,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 3.711778068942029e+18,
"train_batch_size": 16,
"trial_name": null,
"trial_params": null
}