aleegis's picture
Training in progress, epoch 0, checkpoint
fcde4f1 verified
Raw
History Blame Contribute Delete
32.6 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.32935364347468093,
"eval_steps": 500,
"global_step": 800,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.0016467682173734047,
"grad_norm": 0.8692666888237,
"learning_rate": 0.0002,
"loss": 1.2712,
"step": 4
},
{
"epoch": 0.0032935364347468094,
"grad_norm": 1.4892252683639526,
"learning_rate": 0.0002,
"loss": 0.7591,
"step": 8
},
{
"epoch": 0.004940304652120214,
"grad_norm": 0.8453137874603271,
"learning_rate": 0.0002,
"loss": 0.4761,
"step": 12
},
{
"epoch": 0.006587072869493619,
"grad_norm": 1.1559343338012695,
"learning_rate": 0.0002,
"loss": 0.4557,
"step": 16
},
{
"epoch": 0.008233841086867023,
"grad_norm": 0.5908960103988647,
"learning_rate": 0.0002,
"loss": 0.4186,
"step": 20
},
{
"epoch": 0.009880609304240428,
"grad_norm": 0.7283934354782104,
"learning_rate": 0.0002,
"loss": 0.4435,
"step": 24
},
{
"epoch": 0.011527377521613832,
"grad_norm": 0.40564286708831787,
"learning_rate": 0.0002,
"loss": 0.3857,
"step": 28
},
{
"epoch": 0.013174145738987238,
"grad_norm": 0.9832593202590942,
"learning_rate": 0.0002,
"loss": 0.3869,
"step": 32
},
{
"epoch": 0.014820913956360642,
"grad_norm": 0.6356058120727539,
"learning_rate": 0.0002,
"loss": 0.3772,
"step": 36
},
{
"epoch": 0.016467682173734045,
"grad_norm": 0.4383929669857025,
"learning_rate": 0.0002,
"loss": 0.4346,
"step": 40
},
{
"epoch": 0.018114450391107453,
"grad_norm": 0.5270354747772217,
"learning_rate": 0.0002,
"loss": 0.4174,
"step": 44
},
{
"epoch": 0.019761218608480857,
"grad_norm": 0.43965744972229004,
"learning_rate": 0.0002,
"loss": 0.4119,
"step": 48
},
{
"epoch": 0.02140798682585426,
"grad_norm": 0.4961897134780884,
"learning_rate": 0.0002,
"loss": 0.3731,
"step": 52
},
{
"epoch": 0.023054755043227664,
"grad_norm": 0.33899012207984924,
"learning_rate": 0.0002,
"loss": 0.3535,
"step": 56
},
{
"epoch": 0.02470152326060107,
"grad_norm": 0.4382150173187256,
"learning_rate": 0.0002,
"loss": 0.3709,
"step": 60
},
{
"epoch": 0.026348291477974475,
"grad_norm": 0.3768168091773987,
"learning_rate": 0.0002,
"loss": 0.3706,
"step": 64
},
{
"epoch": 0.02799505969534788,
"grad_norm": 0.4140666127204895,
"learning_rate": 0.0002,
"loss": 0.3388,
"step": 68
},
{
"epoch": 0.029641827912721283,
"grad_norm": 0.2670955955982208,
"learning_rate": 0.0002,
"loss": 0.3232,
"step": 72
},
{
"epoch": 0.03128859613009469,
"grad_norm": 0.5184640884399414,
"learning_rate": 0.0002,
"loss": 0.3637,
"step": 76
},
{
"epoch": 0.03293536434746809,
"grad_norm": 0.36012595891952515,
"learning_rate": 0.0002,
"loss": 0.3722,
"step": 80
},
{
"epoch": 0.0345821325648415,
"grad_norm": 0.7544593811035156,
"learning_rate": 0.0002,
"loss": 0.3786,
"step": 84
},
{
"epoch": 0.036228900782214905,
"grad_norm": 0.39954182505607605,
"learning_rate": 0.0002,
"loss": 0.3838,
"step": 88
},
{
"epoch": 0.03787566899958831,
"grad_norm": 0.4210023581981659,
"learning_rate": 0.0002,
"loss": 0.3908,
"step": 92
},
{
"epoch": 0.03952243721696171,
"grad_norm": 0.5908896923065186,
"learning_rate": 0.0002,
"loss": 0.3777,
"step": 96
},
{
"epoch": 0.04116920543433512,
"grad_norm": 0.2767857611179352,
"learning_rate": 0.0002,
"loss": 0.3515,
"step": 100
},
{
"epoch": 0.04281597365170852,
"grad_norm": 0.43755510449409485,
"learning_rate": 0.0002,
"loss": 0.4092,
"step": 104
},
{
"epoch": 0.044462741869081925,
"grad_norm": 0.38882681727409363,
"learning_rate": 0.0002,
"loss": 0.3872,
"step": 108
},
{
"epoch": 0.04610951008645533,
"grad_norm": 0.3989730775356293,
"learning_rate": 0.0002,
"loss": 0.3964,
"step": 112
},
{
"epoch": 0.04775627830382874,
"grad_norm": 0.22192604839801788,
"learning_rate": 0.0002,
"loss": 0.3919,
"step": 116
},
{
"epoch": 0.04940304652120214,
"grad_norm": 0.4765632152557373,
"learning_rate": 0.0002,
"loss": 0.3777,
"step": 120
},
{
"epoch": 0.05104981473857555,
"grad_norm": 0.30803513526916504,
"learning_rate": 0.0002,
"loss": 0.3523,
"step": 124
},
{
"epoch": 0.05269658295594895,
"grad_norm": 0.2666870653629303,
"learning_rate": 0.0002,
"loss": 0.4015,
"step": 128
},
{
"epoch": 0.054343351173322355,
"grad_norm": 0.36492490768432617,
"learning_rate": 0.0002,
"loss": 0.3507,
"step": 132
},
{
"epoch": 0.05599011939069576,
"grad_norm": 0.4382271468639374,
"learning_rate": 0.0002,
"loss": 0.346,
"step": 136
},
{
"epoch": 0.05763688760806916,
"grad_norm": 0.3766026198863983,
"learning_rate": 0.0002,
"loss": 0.4013,
"step": 140
},
{
"epoch": 0.059283655825442566,
"grad_norm": 0.3905723989009857,
"learning_rate": 0.0002,
"loss": 0.3881,
"step": 144
},
{
"epoch": 0.06093042404281598,
"grad_norm": 0.3422534167766571,
"learning_rate": 0.0002,
"loss": 0.3546,
"step": 148
},
{
"epoch": 0.06257719226018937,
"grad_norm": 0.3236246705055237,
"learning_rate": 0.0002,
"loss": 0.3359,
"step": 152
},
{
"epoch": 0.06422396047756278,
"grad_norm": 0.3136650621891022,
"learning_rate": 0.0002,
"loss": 0.3321,
"step": 156
},
{
"epoch": 0.06587072869493618,
"grad_norm": 0.4051145017147064,
"learning_rate": 0.0002,
"loss": 0.368,
"step": 160
},
{
"epoch": 0.0675174969123096,
"grad_norm": 0.35720133781433105,
"learning_rate": 0.0002,
"loss": 0.3832,
"step": 164
},
{
"epoch": 0.069164265129683,
"grad_norm": 0.42161595821380615,
"learning_rate": 0.0002,
"loss": 0.3636,
"step": 168
},
{
"epoch": 0.0708110333470564,
"grad_norm": 0.3453030288219452,
"learning_rate": 0.0002,
"loss": 0.3402,
"step": 172
},
{
"epoch": 0.07245780156442981,
"grad_norm": 2.319596767425537,
"learning_rate": 0.0002,
"loss": 0.5047,
"step": 176
},
{
"epoch": 0.07410456978180321,
"grad_norm": 0.3278649151325226,
"learning_rate": 0.0002,
"loss": 0.3978,
"step": 180
},
{
"epoch": 0.07575133799917662,
"grad_norm": 0.5178958773612976,
"learning_rate": 0.0002,
"loss": 0.4057,
"step": 184
},
{
"epoch": 0.07739810621655002,
"grad_norm": 0.2471940517425537,
"learning_rate": 0.0002,
"loss": 0.3383,
"step": 188
},
{
"epoch": 0.07904487443392343,
"grad_norm": 0.28006836771965027,
"learning_rate": 0.0002,
"loss": 0.3455,
"step": 192
},
{
"epoch": 0.08069164265129683,
"grad_norm": 0.38219529390335083,
"learning_rate": 0.0002,
"loss": 0.3622,
"step": 196
},
{
"epoch": 0.08233841086867023,
"grad_norm": 0.4880613088607788,
"learning_rate": 0.0002,
"loss": 0.3748,
"step": 200
},
{
"epoch": 0.08398517908604364,
"grad_norm": 0.35214322805404663,
"learning_rate": 0.0002,
"loss": 0.3906,
"step": 204
},
{
"epoch": 0.08563194730341704,
"grad_norm": 0.4287274479866028,
"learning_rate": 0.0002,
"loss": 0.4119,
"step": 208
},
{
"epoch": 0.08727871552079045,
"grad_norm": 0.2343526929616928,
"learning_rate": 0.0002,
"loss": 0.3821,
"step": 212
},
{
"epoch": 0.08892548373816385,
"grad_norm": 0.3329123556613922,
"learning_rate": 0.0002,
"loss": 0.3515,
"step": 216
},
{
"epoch": 0.09057225195553725,
"grad_norm": 0.45170557498931885,
"learning_rate": 0.0002,
"loss": 0.4265,
"step": 220
},
{
"epoch": 0.09221902017291066,
"grad_norm": 0.330639123916626,
"learning_rate": 0.0002,
"loss": 0.384,
"step": 224
},
{
"epoch": 0.09386578839028406,
"grad_norm": 0.1902407705783844,
"learning_rate": 0.0002,
"loss": 0.3526,
"step": 228
},
{
"epoch": 0.09551255660765748,
"grad_norm": 0.3268851339817047,
"learning_rate": 0.0002,
"loss": 0.3612,
"step": 232
},
{
"epoch": 0.09715932482503088,
"grad_norm": 0.5299010872840881,
"learning_rate": 0.0002,
"loss": 0.3665,
"step": 236
},
{
"epoch": 0.09880609304240429,
"grad_norm": 0.3987724184989929,
"learning_rate": 0.0002,
"loss": 0.3799,
"step": 240
},
{
"epoch": 0.10045286125977769,
"grad_norm": 0.2613527178764343,
"learning_rate": 0.0002,
"loss": 0.3579,
"step": 244
},
{
"epoch": 0.1020996294771511,
"grad_norm": 0.37342676520347595,
"learning_rate": 0.0002,
"loss": 0.3912,
"step": 248
},
{
"epoch": 0.1037463976945245,
"grad_norm": 0.3383709490299225,
"learning_rate": 0.0002,
"loss": 0.406,
"step": 252
},
{
"epoch": 0.1053931659118979,
"grad_norm": 0.23550820350646973,
"learning_rate": 0.0002,
"loss": 0.3788,
"step": 256
},
{
"epoch": 0.1070399341292713,
"grad_norm": 0.3194361627101898,
"learning_rate": 0.0002,
"loss": 0.3729,
"step": 260
},
{
"epoch": 0.10868670234664471,
"grad_norm": 0.8462342023849487,
"learning_rate": 0.0002,
"loss": 0.3411,
"step": 264
},
{
"epoch": 0.11033347056401811,
"grad_norm": 0.558782160282135,
"learning_rate": 0.0002,
"loss": 0.3922,
"step": 268
},
{
"epoch": 0.11198023878139152,
"grad_norm": 0.24560636281967163,
"learning_rate": 0.0002,
"loss": 0.3578,
"step": 272
},
{
"epoch": 0.11362700699876492,
"grad_norm": 0.32584550976753235,
"learning_rate": 0.0002,
"loss": 0.3313,
"step": 276
},
{
"epoch": 0.11527377521613832,
"grad_norm": 0.29577043652534485,
"learning_rate": 0.0002,
"loss": 0.3522,
"step": 280
},
{
"epoch": 0.11692054343351173,
"grad_norm": 0.47984978556632996,
"learning_rate": 0.0002,
"loss": 0.3596,
"step": 284
},
{
"epoch": 0.11856731165088513,
"grad_norm": 0.29009076952934265,
"learning_rate": 0.0002,
"loss": 0.3895,
"step": 288
},
{
"epoch": 0.12021407986825854,
"grad_norm": 0.2614005506038666,
"learning_rate": 0.0002,
"loss": 0.3597,
"step": 292
},
{
"epoch": 0.12186084808563195,
"grad_norm": 0.379079133272171,
"learning_rate": 0.0002,
"loss": 0.3866,
"step": 296
},
{
"epoch": 0.12350761630300536,
"grad_norm": 0.17416591942310333,
"learning_rate": 0.0002,
"loss": 0.3773,
"step": 300
},
{
"epoch": 0.12515438452037875,
"grad_norm": 0.2912764549255371,
"learning_rate": 0.0002,
"loss": 0.365,
"step": 304
},
{
"epoch": 0.12680115273775217,
"grad_norm": 0.3341335654258728,
"learning_rate": 0.0002,
"loss": 0.3425,
"step": 308
},
{
"epoch": 0.12844792095512556,
"grad_norm": 0.2572537660598755,
"learning_rate": 0.0002,
"loss": 0.3515,
"step": 312
},
{
"epoch": 0.13009468917249897,
"grad_norm": 0.19273734092712402,
"learning_rate": 0.0002,
"loss": 0.3873,
"step": 316
},
{
"epoch": 0.13174145738987236,
"grad_norm": 0.34613487124443054,
"learning_rate": 0.0002,
"loss": 0.3683,
"step": 320
},
{
"epoch": 0.13338822560724578,
"grad_norm": 0.33817189931869507,
"learning_rate": 0.0002,
"loss": 0.3599,
"step": 324
},
{
"epoch": 0.1350349938246192,
"grad_norm": 0.41500359773635864,
"learning_rate": 0.0002,
"loss": 0.3436,
"step": 328
},
{
"epoch": 0.1366817620419926,
"grad_norm": 0.35156941413879395,
"learning_rate": 0.0002,
"loss": 0.352,
"step": 332
},
{
"epoch": 0.138328530259366,
"grad_norm": 0.3244033455848694,
"learning_rate": 0.0002,
"loss": 0.3489,
"step": 336
},
{
"epoch": 0.1399752984767394,
"grad_norm": 0.4915976822376251,
"learning_rate": 0.0002,
"loss": 0.3834,
"step": 340
},
{
"epoch": 0.1416220666941128,
"grad_norm": 0.595040500164032,
"learning_rate": 0.0002,
"loss": 0.375,
"step": 344
},
{
"epoch": 0.1432688349114862,
"grad_norm": 0.3264763057231903,
"learning_rate": 0.0002,
"loss": 0.3261,
"step": 348
},
{
"epoch": 0.14491560312885962,
"grad_norm": 0.5096532702445984,
"learning_rate": 0.0002,
"loss": 0.3651,
"step": 352
},
{
"epoch": 0.146562371346233,
"grad_norm": 0.6608021855354309,
"learning_rate": 0.0002,
"loss": 0.3675,
"step": 356
},
{
"epoch": 0.14820913956360643,
"grad_norm": 0.2926951050758362,
"learning_rate": 0.0002,
"loss": 0.4021,
"step": 360
},
{
"epoch": 0.14985590778097982,
"grad_norm": 0.42783549427986145,
"learning_rate": 0.0002,
"loss": 0.3448,
"step": 364
},
{
"epoch": 0.15150267599835324,
"grad_norm": 0.33004024624824524,
"learning_rate": 0.0002,
"loss": 0.3812,
"step": 368
},
{
"epoch": 0.15314944421572663,
"grad_norm": 0.3624507188796997,
"learning_rate": 0.0002,
"loss": 0.3618,
"step": 372
},
{
"epoch": 0.15479621243310004,
"grad_norm": 0.3344748318195343,
"learning_rate": 0.0002,
"loss": 0.3853,
"step": 376
},
{
"epoch": 0.15644298065047343,
"grad_norm": 0.343755841255188,
"learning_rate": 0.0002,
"loss": 0.332,
"step": 380
},
{
"epoch": 0.15808974886784685,
"grad_norm": 0.5197455883026123,
"learning_rate": 0.0002,
"loss": 0.4023,
"step": 384
},
{
"epoch": 0.15973651708522024,
"grad_norm": 0.29629090428352356,
"learning_rate": 0.0002,
"loss": 0.3549,
"step": 388
},
{
"epoch": 0.16138328530259366,
"grad_norm": 0.16533423960208893,
"learning_rate": 0.0002,
"loss": 0.3743,
"step": 392
},
{
"epoch": 0.16303005351996708,
"grad_norm": 0.25626063346862793,
"learning_rate": 0.0002,
"loss": 0.364,
"step": 396
},
{
"epoch": 0.16467682173734047,
"grad_norm": 0.31135621666908264,
"learning_rate": 0.0002,
"loss": 0.3454,
"step": 400
},
{
"epoch": 0.16632358995471389,
"grad_norm": 0.3969805836677551,
"learning_rate": 0.0002,
"loss": 0.3623,
"step": 404
},
{
"epoch": 0.16797035817208728,
"grad_norm": 0.35962575674057007,
"learning_rate": 0.0002,
"loss": 0.3706,
"step": 408
},
{
"epoch": 0.1696171263894607,
"grad_norm": 0.36572885513305664,
"learning_rate": 0.0002,
"loss": 0.3462,
"step": 412
},
{
"epoch": 0.17126389460683408,
"grad_norm": 0.4497930705547333,
"learning_rate": 0.0002,
"loss": 0.4298,
"step": 416
},
{
"epoch": 0.1729106628242075,
"grad_norm": 0.4056977331638336,
"learning_rate": 0.0002,
"loss": 0.4171,
"step": 420
},
{
"epoch": 0.1745574310415809,
"grad_norm": 0.3339053988456726,
"learning_rate": 0.0002,
"loss": 0.3756,
"step": 424
},
{
"epoch": 0.1762041992589543,
"grad_norm": 0.38297727704048157,
"learning_rate": 0.0002,
"loss": 0.4004,
"step": 428
},
{
"epoch": 0.1778509674763277,
"grad_norm": 0.2339114397764206,
"learning_rate": 0.0002,
"loss": 0.3696,
"step": 432
},
{
"epoch": 0.17949773569370112,
"grad_norm": 0.3464541733264923,
"learning_rate": 0.0002,
"loss": 0.3948,
"step": 436
},
{
"epoch": 0.1811445039110745,
"grad_norm": 0.2537291944026947,
"learning_rate": 0.0002,
"loss": 0.3155,
"step": 440
},
{
"epoch": 0.18279127212844792,
"grad_norm": 0.36388587951660156,
"learning_rate": 0.0002,
"loss": 0.3821,
"step": 444
},
{
"epoch": 0.1844380403458213,
"grad_norm": 0.3073045015335083,
"learning_rate": 0.0002,
"loss": 0.3446,
"step": 448
},
{
"epoch": 0.18608480856319473,
"grad_norm": 0.29027312994003296,
"learning_rate": 0.0002,
"loss": 0.3555,
"step": 452
},
{
"epoch": 0.18773157678056812,
"grad_norm": 0.24101793766021729,
"learning_rate": 0.0002,
"loss": 0.3449,
"step": 456
},
{
"epoch": 0.18937834499794154,
"grad_norm": 0.25319597125053406,
"learning_rate": 0.0002,
"loss": 0.3453,
"step": 460
},
{
"epoch": 0.19102511321531496,
"grad_norm": 0.21023103594779968,
"learning_rate": 0.0002,
"loss": 0.3476,
"step": 464
},
{
"epoch": 0.19267188143268835,
"grad_norm": 0.35161590576171875,
"learning_rate": 0.0002,
"loss": 0.3596,
"step": 468
},
{
"epoch": 0.19431864965006176,
"grad_norm": 0.3231665790081024,
"learning_rate": 0.0002,
"loss": 0.3943,
"step": 472
},
{
"epoch": 0.19596541786743515,
"grad_norm": 0.2584836483001709,
"learning_rate": 0.0002,
"loss": 0.3618,
"step": 476
},
{
"epoch": 0.19761218608480857,
"grad_norm": 0.5873728394508362,
"learning_rate": 0.0002,
"loss": 0.3601,
"step": 480
},
{
"epoch": 0.19925895430218196,
"grad_norm": 0.37877875566482544,
"learning_rate": 0.0002,
"loss": 0.2994,
"step": 484
},
{
"epoch": 0.20090572251955538,
"grad_norm": 0.40954428911209106,
"learning_rate": 0.0002,
"loss": 0.3881,
"step": 488
},
{
"epoch": 0.20255249073692877,
"grad_norm": 0.3592691719532013,
"learning_rate": 0.0002,
"loss": 0.3803,
"step": 492
},
{
"epoch": 0.2041992589543022,
"grad_norm": 0.6396195292472839,
"learning_rate": 0.0002,
"loss": 0.4616,
"step": 496
},
{
"epoch": 0.20584602717167558,
"grad_norm": 0.41789042949676514,
"learning_rate": 0.0002,
"loss": 0.3382,
"step": 500
},
{
"epoch": 0.207492795389049,
"grad_norm": 0.8054398894309998,
"learning_rate": 0.0002,
"loss": 0.3984,
"step": 504
},
{
"epoch": 0.20913956360642239,
"grad_norm": 0.3785901665687561,
"learning_rate": 0.0002,
"loss": 0.3625,
"step": 508
},
{
"epoch": 0.2107863318237958,
"grad_norm": 0.36928969621658325,
"learning_rate": 0.0002,
"loss": 0.3746,
"step": 512
},
{
"epoch": 0.2124331000411692,
"grad_norm": 0.37541359663009644,
"learning_rate": 0.0002,
"loss": 0.4138,
"step": 516
},
{
"epoch": 0.2140798682585426,
"grad_norm": 0.5162031650543213,
"learning_rate": 0.0002,
"loss": 0.3852,
"step": 520
},
{
"epoch": 0.21572663647591603,
"grad_norm": 0.3314428925514221,
"learning_rate": 0.0002,
"loss": 0.3621,
"step": 524
},
{
"epoch": 0.21737340469328942,
"grad_norm": 0.5314354300498962,
"learning_rate": 0.0002,
"loss": 0.4387,
"step": 528
},
{
"epoch": 0.21902017291066284,
"grad_norm": 0.3398245573043823,
"learning_rate": 0.0002,
"loss": 0.3846,
"step": 532
},
{
"epoch": 0.22066694112803623,
"grad_norm": 0.28560298681259155,
"learning_rate": 0.0002,
"loss": 0.3634,
"step": 536
},
{
"epoch": 0.22231370934540964,
"grad_norm": 0.27383607625961304,
"learning_rate": 0.0002,
"loss": 0.3485,
"step": 540
},
{
"epoch": 0.22396047756278303,
"grad_norm": 0.4186703562736511,
"learning_rate": 0.0002,
"loss": 0.3729,
"step": 544
},
{
"epoch": 0.22560724578015645,
"grad_norm": 0.4143611192703247,
"learning_rate": 0.0002,
"loss": 0.3882,
"step": 548
},
{
"epoch": 0.22725401399752984,
"grad_norm": 0.4646351635456085,
"learning_rate": 0.0002,
"loss": 0.4018,
"step": 552
},
{
"epoch": 0.22890078221490326,
"grad_norm": 0.3014436960220337,
"learning_rate": 0.0002,
"loss": 0.3779,
"step": 556
},
{
"epoch": 0.23054755043227665,
"grad_norm": 0.29496312141418457,
"learning_rate": 0.0002,
"loss": 0.3634,
"step": 560
},
{
"epoch": 0.23219431864965007,
"grad_norm": 0.17998939752578735,
"learning_rate": 0.0002,
"loss": 0.3265,
"step": 564
},
{
"epoch": 0.23384108686702346,
"grad_norm": 0.3189595937728882,
"learning_rate": 0.0002,
"loss": 0.3868,
"step": 568
},
{
"epoch": 0.23548785508439687,
"grad_norm": 0.32916879653930664,
"learning_rate": 0.0002,
"loss": 0.3122,
"step": 572
},
{
"epoch": 0.23713462330177026,
"grad_norm": 0.44192901253700256,
"learning_rate": 0.0002,
"loss": 0.4039,
"step": 576
},
{
"epoch": 0.23878139151914368,
"grad_norm": 0.3216501474380493,
"learning_rate": 0.0002,
"loss": 0.3455,
"step": 580
},
{
"epoch": 0.24042815973651707,
"grad_norm": 0.45848771929740906,
"learning_rate": 0.0002,
"loss": 0.3973,
"step": 584
},
{
"epoch": 0.2420749279538905,
"grad_norm": 0.431364506483078,
"learning_rate": 0.0002,
"loss": 0.4,
"step": 588
},
{
"epoch": 0.2437216961712639,
"grad_norm": 0.20512379705905914,
"learning_rate": 0.0002,
"loss": 0.3254,
"step": 592
},
{
"epoch": 0.2453684643886373,
"grad_norm": 0.45757392048835754,
"learning_rate": 0.0002,
"loss": 0.3932,
"step": 596
},
{
"epoch": 0.24701523260601072,
"grad_norm": 0.29159244894981384,
"learning_rate": 0.0002,
"loss": 0.3848,
"step": 600
},
{
"epoch": 0.2486620008233841,
"grad_norm": 0.23504222929477692,
"learning_rate": 0.0002,
"loss": 0.3555,
"step": 604
},
{
"epoch": 0.2503087690407575,
"grad_norm": 0.33391910791397095,
"learning_rate": 0.0002,
"loss": 0.3494,
"step": 608
},
{
"epoch": 0.2519555372581309,
"grad_norm": 0.3338538706302643,
"learning_rate": 0.0002,
"loss": 0.3185,
"step": 612
},
{
"epoch": 0.25360230547550433,
"grad_norm": 0.3252895176410675,
"learning_rate": 0.0002,
"loss": 0.3613,
"step": 616
},
{
"epoch": 0.25524907369287775,
"grad_norm": 0.3765605092048645,
"learning_rate": 0.0002,
"loss": 0.3782,
"step": 620
},
{
"epoch": 0.2568958419102511,
"grad_norm": 0.28585657477378845,
"learning_rate": 0.0002,
"loss": 0.3586,
"step": 624
},
{
"epoch": 0.25854261012762453,
"grad_norm": 0.3525017201900482,
"learning_rate": 0.0002,
"loss": 0.355,
"step": 628
},
{
"epoch": 0.26018937834499795,
"grad_norm": 0.5153087377548218,
"learning_rate": 0.0002,
"loss": 0.3512,
"step": 632
},
{
"epoch": 0.26183614656237136,
"grad_norm": 0.4493424892425537,
"learning_rate": 0.0002,
"loss": 0.3709,
"step": 636
},
{
"epoch": 0.2634829147797447,
"grad_norm": 0.2840404510498047,
"learning_rate": 0.0002,
"loss": 0.3733,
"step": 640
},
{
"epoch": 0.26512968299711814,
"grad_norm": 0.25880661606788635,
"learning_rate": 0.0002,
"loss": 0.4248,
"step": 644
},
{
"epoch": 0.26677645121449156,
"grad_norm": 0.34614700078964233,
"learning_rate": 0.0002,
"loss": 0.358,
"step": 648
},
{
"epoch": 0.268423219431865,
"grad_norm": 0.3996017873287201,
"learning_rate": 0.0002,
"loss": 0.38,
"step": 652
},
{
"epoch": 0.2700699876492384,
"grad_norm": 0.33780571818351746,
"learning_rate": 0.0002,
"loss": 0.4254,
"step": 656
},
{
"epoch": 0.27171675586661176,
"grad_norm": 0.2469349503517151,
"learning_rate": 0.0002,
"loss": 0.3837,
"step": 660
},
{
"epoch": 0.2733635240839852,
"grad_norm": 0.3020316958427429,
"learning_rate": 0.0002,
"loss": 0.3575,
"step": 664
},
{
"epoch": 0.2750102923013586,
"grad_norm": 0.39028212428092957,
"learning_rate": 0.0002,
"loss": 0.368,
"step": 668
},
{
"epoch": 0.276657060518732,
"grad_norm": 0.23725393414497375,
"learning_rate": 0.0002,
"loss": 0.456,
"step": 672
},
{
"epoch": 0.2783038287361054,
"grad_norm": 0.3587561845779419,
"learning_rate": 0.0002,
"loss": 0.3459,
"step": 676
},
{
"epoch": 0.2799505969534788,
"grad_norm": 0.45671024918556213,
"learning_rate": 0.0002,
"loss": 0.4069,
"step": 680
},
{
"epoch": 0.2815973651708522,
"grad_norm": 0.43547502160072327,
"learning_rate": 0.0002,
"loss": 0.3693,
"step": 684
},
{
"epoch": 0.2832441333882256,
"grad_norm": 0.31809914112091064,
"learning_rate": 0.0002,
"loss": 0.3519,
"step": 688
},
{
"epoch": 0.284890901605599,
"grad_norm": 0.4354214668273926,
"learning_rate": 0.0002,
"loss": 0.355,
"step": 692
},
{
"epoch": 0.2865376698229724,
"grad_norm": 0.3542259633541107,
"learning_rate": 0.0002,
"loss": 0.3854,
"step": 696
},
{
"epoch": 0.2881844380403458,
"grad_norm": 0.31998395919799805,
"learning_rate": 0.0002,
"loss": 0.3586,
"step": 700
},
{
"epoch": 0.28983120625771924,
"grad_norm": 0.3070733845233917,
"learning_rate": 0.0002,
"loss": 0.3362,
"step": 704
},
{
"epoch": 0.2914779744750926,
"grad_norm": 0.3756466209888458,
"learning_rate": 0.0002,
"loss": 0.3572,
"step": 708
},
{
"epoch": 0.293124742692466,
"grad_norm": 0.5045058131217957,
"learning_rate": 0.0002,
"loss": 0.3886,
"step": 712
},
{
"epoch": 0.29477151090983944,
"grad_norm": 0.49769386649131775,
"learning_rate": 0.0002,
"loss": 0.3932,
"step": 716
},
{
"epoch": 0.29641827912721286,
"grad_norm": 0.3411276638507843,
"learning_rate": 0.0002,
"loss": 0.3513,
"step": 720
},
{
"epoch": 0.2980650473445863,
"grad_norm": 0.17998747527599335,
"learning_rate": 0.0002,
"loss": 0.3293,
"step": 724
},
{
"epoch": 0.29971181556195964,
"grad_norm": 0.5225507616996765,
"learning_rate": 0.0002,
"loss": 0.3985,
"step": 728
},
{
"epoch": 0.30135858377933306,
"grad_norm": 0.3224283456802368,
"learning_rate": 0.0002,
"loss": 0.3791,
"step": 732
},
{
"epoch": 0.3030053519967065,
"grad_norm": 0.3151153028011322,
"learning_rate": 0.0002,
"loss": 0.335,
"step": 736
},
{
"epoch": 0.3046521202140799,
"grad_norm": 0.27512961626052856,
"learning_rate": 0.0002,
"loss": 0.3225,
"step": 740
},
{
"epoch": 0.30629888843145325,
"grad_norm": 0.48961198329925537,
"learning_rate": 0.0002,
"loss": 0.3305,
"step": 744
},
{
"epoch": 0.30794565664882667,
"grad_norm": 0.6176120638847351,
"learning_rate": 0.0002,
"loss": 0.4045,
"step": 748
},
{
"epoch": 0.3095924248662001,
"grad_norm": 0.42022645473480225,
"learning_rate": 0.0002,
"loss": 0.3803,
"step": 752
},
{
"epoch": 0.3112391930835735,
"grad_norm": 0.2890242338180542,
"learning_rate": 0.0002,
"loss": 0.3744,
"step": 756
},
{
"epoch": 0.31288596130094687,
"grad_norm": 0.3050364851951599,
"learning_rate": 0.0002,
"loss": 0.3655,
"step": 760
},
{
"epoch": 0.3145327295183203,
"grad_norm": 0.2680593430995941,
"learning_rate": 0.0002,
"loss": 0.3557,
"step": 764
},
{
"epoch": 0.3161794977356937,
"grad_norm": 0.7212200164794922,
"learning_rate": 0.0002,
"loss": 0.4029,
"step": 768
},
{
"epoch": 0.3178262659530671,
"grad_norm": 0.5011200308799744,
"learning_rate": 0.0002,
"loss": 0.3927,
"step": 772
},
{
"epoch": 0.3194730341704405,
"grad_norm": 0.44199439883232117,
"learning_rate": 0.0002,
"loss": 0.3531,
"step": 776
},
{
"epoch": 0.3211198023878139,
"grad_norm": 0.3872445821762085,
"learning_rate": 0.0002,
"loss": 0.3475,
"step": 780
},
{
"epoch": 0.3227665706051873,
"grad_norm": 0.2929086685180664,
"learning_rate": 0.0002,
"loss": 0.3831,
"step": 784
},
{
"epoch": 0.32441333882256074,
"grad_norm": 0.36971160769462585,
"learning_rate": 0.0002,
"loss": 0.3578,
"step": 788
},
{
"epoch": 0.32606010703993416,
"grad_norm": 0.509065568447113,
"learning_rate": 0.0002,
"loss": 0.3796,
"step": 792
},
{
"epoch": 0.3277068752573075,
"grad_norm": 0.3227056562900543,
"learning_rate": 0.0002,
"loss": 0.3966,
"step": 796
},
{
"epoch": 0.32935364347468093,
"grad_norm": 0.4549015164375305,
"learning_rate": 0.0002,
"loss": 0.3564,
"step": 800
}
],
"logging_steps": 4,
"max_steps": 800,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 2.88348602105856e+17,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}