Colin1337X's picture
Upload folder using huggingface_hub
b18428f verified
Raw History Blame Contribute Delete
23.7 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.04513469669049836,
"eval_steps": 500,
"global_step": 13000,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.00034718997454229514,
"grad_norm": 0.8856930732727051,
"learning_rate": 0.00019999994169931,
"loss": 2.4069,
"step": 100
},
{
"epoch": 0.0006943799490845903,
"grad_norm": 0.8890424966812134,
"learning_rate": 0.0001999997644357777,
"loss": 2.3882,
"step": 200
},
{
"epoch": 0.0010415699236268853,
"grad_norm": 0.8619076013565063,
"learning_rate": 0.00019999946820366553,
"loss": 2.4208,
"step": 300
},
{
"epoch": 0.0013887598981691806,
"grad_norm": 0.8981874585151672,
"learning_rate": 0.00019999905300332595,
"loss": 2.3859,
"step": 400
},
{
"epoch": 0.0017359498727114757,
"grad_norm": 1.1088451147079468,
"learning_rate": 0.0001999985188352529,
"loss": 2.4031,
"step": 500
},
{
"epoch": 0.0020831398472537705,
"grad_norm": 0.8845599889755249,
"learning_rate": 0.00019999786570008187,
"loss": 2.3765,
"step": 600
},
{
"epoch": 0.002430329821796066,
"grad_norm": 1.1892898082733154,
"learning_rate": 0.0001999970935985899,
"loss": 2.4083,
"step": 700
},
{
"epoch": 0.002777519796338361,
"grad_norm": 1.2907609939575195,
"learning_rate": 0.00019999620253169554,
"loss": 2.4262,
"step": 800
},
{
"epoch": 0.003124709770880656,
"grad_norm": 1.511342167854309,
"learning_rate": 0.00019999519250045886,
"loss": 2.4077,
"step": 900
},
{
"epoch": 0.0034718997454229513,
"grad_norm": 1.2708256244659424,
"learning_rate": 0.00019999406350608152,
"loss": 2.4238,
"step": 1000
},
{
"epoch": 0.003819089719965246,
"grad_norm": 1.380018949508667,
"learning_rate": 0.00019999281554990668,
"loss": 2.4429,
"step": 1100
},
{
"epoch": 0.004166279694507541,
"grad_norm": 1.5074928998947144,
"learning_rate": 0.000199991448633419,
"loss": 2.4212,
"step": 1200
},
{
"epoch": 0.004513469669049836,
"grad_norm": 1.20973801612854,
"learning_rate": 0.00019998996275824465,
"loss": 2.4092,
"step": 1300
},
{
"epoch": 0.004860659643592132,
"grad_norm": 1.4226404428482056,
"learning_rate": 0.0001999883579261514,
"loss": 2.3977,
"step": 1400
},
{
"epoch": 0.005207849618134427,
"grad_norm": 1.437606930732727,
"learning_rate": 0.0001999866341390485,
"loss": 2.4527,
"step": 1500
},
{
"epoch": 0.005555039592676722,
"grad_norm": 1.4428914785385132,
"learning_rate": 0.00019998479139898668,
"loss": 2.4495,
"step": 1600
},
{
"epoch": 0.005902229567219017,
"grad_norm": 2.573493719100952,
"learning_rate": 0.00019998282970815832,
"loss": 2.4216,
"step": 1700
},
{
"epoch": 0.006249419541761312,
"grad_norm": 1.491316795349121,
"learning_rate": 0.00019998074906889715,
"loss": 2.4209,
"step": 1800
},
{
"epoch": 0.006596609516303607,
"grad_norm": 1.9518392086029053,
"learning_rate": 0.00019997854948367846,
"loss": 2.4081,
"step": 1900
},
{
"epoch": 0.006943799490845903,
"grad_norm": 1.6811418533325195,
"learning_rate": 0.0001999762309551191,
"loss": 2.4255,
"step": 2000
},
{
"epoch": 0.007290989465388197,
"grad_norm": 1.316292643547058,
"learning_rate": 0.00019997379348597744,
"loss": 2.4039,
"step": 2100
},
{
"epoch": 0.007638179439930492,
"grad_norm": 1.297400712966919,
"learning_rate": 0.00019997123707915325,
"loss": 2.4193,
"step": 2200
},
{
"epoch": 0.007985369414472789,
"grad_norm": 1.5318177938461304,
"learning_rate": 0.00019996856173768785,
"loss": 2.4258,
"step": 2300
},
{
"epoch": 0.008332559389015082,
"grad_norm": 2.240832567214966,
"learning_rate": 0.00019996576746476413,
"loss": 2.4457,
"step": 2400
},
{
"epoch": 0.008679749363557377,
"grad_norm": 1.50482976436615,
"learning_rate": 0.00019996285426370636,
"loss": 2.4521,
"step": 2500
},
{
"epoch": 0.009026939338099673,
"grad_norm": 1.1736887693405151,
"learning_rate": 0.00019995982213798035,
"loss": 2.4226,
"step": 2600
},
{
"epoch": 0.009374129312641968,
"grad_norm": 1.489713430404663,
"learning_rate": 0.00019995667109119335,
"loss": 2.4314,
"step": 2700
},
{
"epoch": 0.009721319287184263,
"grad_norm": 2.00687313079834,
"learning_rate": 0.00019995340112709422,
"loss": 2.4389,
"step": 2800
},
{
"epoch": 0.010068509261726559,
"grad_norm": 1.5748944282531738,
"learning_rate": 0.00019995001224957308,
"loss": 2.4483,
"step": 2900
},
{
"epoch": 0.010415699236268854,
"grad_norm": 1.7045289278030396,
"learning_rate": 0.00019994650446266175,
"loss": 2.4407,
"step": 3000
},
{
"epoch": 0.01076288921081115,
"grad_norm": 1.3988068103790283,
"learning_rate": 0.0001999428777705333,
"loss": 2.462,
"step": 3100
},
{
"epoch": 0.011110079185353445,
"grad_norm": 1.3608533143997192,
"learning_rate": 0.00019993913217750245,
"loss": 2.4472,
"step": 3200
},
{
"epoch": 0.011457269159895738,
"grad_norm": 1.2952604293823242,
"learning_rate": 0.00019993526768802525,
"loss": 2.4427,
"step": 3300
},
{
"epoch": 0.011804459134438033,
"grad_norm": 2.3562355041503906,
"learning_rate": 0.0001999312843066992,
"loss": 2.4659,
"step": 3400
},
{
"epoch": 0.012151649108980329,
"grad_norm": 2.296459913253784,
"learning_rate": 0.00019992718203826337,
"loss": 2.4795,
"step": 3500
},
{
"epoch": 0.012498839083522624,
"grad_norm": 1.3606270551681519,
"learning_rate": 0.00019992296088759814,
"loss": 2.463,
"step": 3600
},
{
"epoch": 0.01284602905806492,
"grad_norm": 1.6714372634887695,
"learning_rate": 0.00019991862085972536,
"loss": 2.43,
"step": 3700
},
{
"epoch": 0.013193219032607215,
"grad_norm": 2.1929798126220703,
"learning_rate": 0.0001999141619598083,
"loss": 2.4513,
"step": 3800
},
{
"epoch": 0.01354040900714951,
"grad_norm": 1.7777063846588135,
"learning_rate": 0.00019990958419315166,
"loss": 2.472,
"step": 3900
},
{
"epoch": 0.013887598981691805,
"grad_norm": 1.5554057359695435,
"learning_rate": 0.00019990488756520164,
"loss": 2.4663,
"step": 4000
},
{
"epoch": 0.0142347889562341,
"grad_norm": 1.551933765411377,
"learning_rate": 0.00019990007208154565,
"loss": 2.5033,
"step": 4100
},
{
"epoch": 0.014581978930776394,
"grad_norm": 1.9665030241012573,
"learning_rate": 0.00019989513774791267,
"loss": 2.4518,
"step": 4200
},
{
"epoch": 0.01492916890531869,
"grad_norm": 1.5771666765213013,
"learning_rate": 0.000199890084570173,
"loss": 2.4703,
"step": 4300
},
{
"epoch": 0.015276358879860985,
"grad_norm": 1.750036597251892,
"learning_rate": 0.0001998849125543384,
"loss": 2.495,
"step": 4400
},
{
"epoch": 0.01562354885440328,
"grad_norm": 1.4305938482284546,
"learning_rate": 0.00019987962170656188,
"loss": 2.473,
"step": 4500
},
{
"epoch": 0.015970738828945577,
"grad_norm": 1.410703420639038,
"learning_rate": 0.00019987421203313799,
"loss": 2.4629,
"step": 4600
},
{
"epoch": 0.01631792880348787,
"grad_norm": 1.448388695716858,
"learning_rate": 0.0001998686835405025,
"loss": 2.4591,
"step": 4700
},
{
"epoch": 0.016665118778030164,
"grad_norm": 1.7654149532318115,
"learning_rate": 0.00019986303623523258,
"loss": 2.4754,
"step": 4800
},
{
"epoch": 0.01701230875257246,
"grad_norm": 1.8118611574172974,
"learning_rate": 0.0001998572701240468,
"loss": 2.4657,
"step": 4900
},
{
"epoch": 0.017359498727114755,
"grad_norm": 1.8808069229125977,
"learning_rate": 0.00019985138521380505,
"loss": 2.4736,
"step": 5000
},
{
"epoch": 0.01770668870165705,
"grad_norm": 1.5441018342971802,
"learning_rate": 0.00019984538151150846,
"loss": 2.4785,
"step": 5100
},
{
"epoch": 0.018053878676199345,
"grad_norm": 1.6582081317901611,
"learning_rate": 0.00019983925902429967,
"loss": 2.4841,
"step": 5200
},
{
"epoch": 0.01840106865074164,
"grad_norm": 1.5702968835830688,
"learning_rate": 0.00019983301775946245,
"loss": 2.4868,
"step": 5300
},
{
"epoch": 0.018748258625283936,
"grad_norm": 2.0732314586639404,
"learning_rate": 0.00019982665772442205,
"loss": 2.4601,
"step": 5400
},
{
"epoch": 0.01909544859982623,
"grad_norm": 1.4423972368240356,
"learning_rate": 0.00019982017892674483,
"loss": 2.4773,
"step": 5500
},
{
"epoch": 0.019442638574368527,
"grad_norm": 1.9959121942520142,
"learning_rate": 0.00019981358137413863,
"loss": 2.4989,
"step": 5600
},
{
"epoch": 0.019789828548910822,
"grad_norm": 1.7514522075653076,
"learning_rate": 0.0001998068650744524,
"loss": 2.4689,
"step": 5700
},
{
"epoch": 0.020137018523453117,
"grad_norm": 1.6253585815429688,
"learning_rate": 0.00019980003003567653,
"loss": 2.4683,
"step": 5800
},
{
"epoch": 0.020484208497995413,
"grad_norm": 1.9775470495224,
"learning_rate": 0.0001997930762659425,
"loss": 2.4787,
"step": 5900
},
{
"epoch": 0.020831398472537708,
"grad_norm": 1.4767321348190308,
"learning_rate": 0.00019978600377352324,
"loss": 2.4865,
"step": 6000
},
{
"epoch": 0.021178588447080003,
"grad_norm": 2.2066593170166016,
"learning_rate": 0.00019977881256683273,
"loss": 2.4826,
"step": 6100
},
{
"epoch": 0.0215257784216223,
"grad_norm": 1.9816187620162964,
"learning_rate": 0.00019977150265442626,
"loss": 2.486,
"step": 6200
},
{
"epoch": 0.021872968396164594,
"grad_norm": 1.573564052581787,
"learning_rate": 0.0001997640740450004,
"loss": 2.4893,
"step": 6300
},
{
"epoch": 0.02222015837070689,
"grad_norm": 2.1582605838775635,
"learning_rate": 0.00019975652674739285,
"loss": 2.4896,
"step": 6400
},
{
"epoch": 0.02256734834524918,
"grad_norm": 1.991675615310669,
"learning_rate": 0.00019974886077058255,
"loss": 2.5041,
"step": 6500
},
{
"epoch": 0.022914538319791476,
"grad_norm": 1.781675934791565,
"learning_rate": 0.00019974107612368962,
"loss": 2.475,
"step": 6600
},
{
"epoch": 0.02326172829433377,
"grad_norm": 1.6764287948608398,
"learning_rate": 0.00019973317281597538,
"loss": 2.5048,
"step": 6700
},
{
"epoch": 0.023608918268876067,
"grad_norm": 2.408081531524658,
"learning_rate": 0.00019972515085684233,
"loss": 2.49,
"step": 6800
},
{
"epoch": 0.023956108243418362,
"grad_norm": 2.09415864944458,
"learning_rate": 0.00019971701025583402,
"loss": 2.5058,
"step": 6900
},
{
"epoch": 0.024303298217960657,
"grad_norm": 1.9584619998931885,
"learning_rate": 0.00019970875102263532,
"loss": 2.4775,
"step": 7000
},
{
"epoch": 0.024650488192502953,
"grad_norm": 1.894208550453186,
"learning_rate": 0.00019970037316707212,
"loss": 2.4997,
"step": 7100
},
{
"epoch": 0.024997678167045248,
"grad_norm": 1.6168571710586548,
"learning_rate": 0.00019969187669911142,
"loss": 2.4641,
"step": 7200
},
{
"epoch": 0.025344868141587543,
"grad_norm": 2.0909547805786133,
"learning_rate": 0.0001996832616288614,
"loss": 2.48,
"step": 7300
},
{
"epoch": 0.02569205811612984,
"grad_norm": 2.2160308361053467,
"learning_rate": 0.0001996745279665713,
"loss": 2.512,
"step": 7400
},
{
"epoch": 0.026039248090672134,
"grad_norm": 1.6042606830596924,
"learning_rate": 0.0001996656757226315,
"loss": 2.4831,
"step": 7500
},
{
"epoch": 0.02638643806521443,
"grad_norm": 2.216085910797119,
"learning_rate": 0.00019965670490757335,
"loss": 2.4778,
"step": 7600
},
{
"epoch": 0.026733628039756725,
"grad_norm": 1.5685747861862183,
"learning_rate": 0.00019964761553206936,
"loss": 2.4861,
"step": 7700
},
{
"epoch": 0.02708081801429902,
"grad_norm": 1.4898216724395752,
"learning_rate": 0.00019963840760693307,
"loss": 2.5092,
"step": 7800
},
{
"epoch": 0.027428007988841315,
"grad_norm": 1.7341415882110596,
"learning_rate": 0.00019962908114311902,
"loss": 2.5128,
"step": 7900
},
{
"epoch": 0.02777519796338361,
"grad_norm": 1.6686646938323975,
"learning_rate": 0.0001996196361517228,
"loss": 2.4971,
"step": 8000
},
{
"epoch": 0.028122387937925906,
"grad_norm": 2.1513888835906982,
"learning_rate": 0.00019961007264398105,
"loss": 2.4927,
"step": 8100
},
{
"epoch": 0.0284695779124682,
"grad_norm": 1.3236947059631348,
"learning_rate": 0.0001996003906312713,
"loss": 2.4967,
"step": 8200
},
{
"epoch": 0.028816767887010496,
"grad_norm": 1.6354811191558838,
"learning_rate": 0.00019959059012511214,
"loss": 2.4954,
"step": 8300
},
{
"epoch": 0.029163957861552788,
"grad_norm": 1.8133015632629395,
"learning_rate": 0.00019958067113716316,
"loss": 2.4878,
"step": 8400
},
{
"epoch": 0.029511147836095084,
"grad_norm": 2.2775697708129883,
"learning_rate": 0.00019957063367922486,
"loss": 2.4828,
"step": 8500
},
{
"epoch": 0.02985833781063738,
"grad_norm": 2.0718913078308105,
"learning_rate": 0.0001995604777632387,
"loss": 2.5039,
"step": 8600
},
{
"epoch": 0.030205527785179674,
"grad_norm": 1.5673631429672241,
"learning_rate": 0.00019955020340128697,
"loss": 2.509,
"step": 8700
},
{
"epoch": 0.03055271775972197,
"grad_norm": 2.0072789192199707,
"learning_rate": 0.00019953981060559305,
"loss": 2.491,
"step": 8800
},
{
"epoch": 0.030899907734264265,
"grad_norm": 2.3539609909057617,
"learning_rate": 0.00019952929938852113,
"loss": 2.4983,
"step": 8900
},
{
"epoch": 0.03124709770880656,
"grad_norm": 1.8251752853393555,
"learning_rate": 0.00019951866976257622,
"loss": 2.5033,
"step": 9000
},
{
"epoch": 0.031594287683348855,
"grad_norm": 1.6241949796676636,
"learning_rate": 0.00019950792174040433,
"loss": 2.5094,
"step": 9100
},
{
"epoch": 0.031941477657891154,
"grad_norm": 1.5035438537597656,
"learning_rate": 0.00019949705533479225,
"loss": 2.4837,
"step": 9200
},
{
"epoch": 0.032288667632433446,
"grad_norm": 1.893409252166748,
"learning_rate": 0.0001994860705586676,
"loss": 2.5021,
"step": 9300
},
{
"epoch": 0.03263585760697574,
"grad_norm": 1.8297152519226074,
"learning_rate": 0.00019947496742509885,
"loss": 2.5236,
"step": 9400
},
{
"epoch": 0.03298304758151804,
"grad_norm": 2.074918508529663,
"learning_rate": 0.00019946374594729524,
"loss": 2.482,
"step": 9500
},
{
"epoch": 0.03333023755606033,
"grad_norm": 1.6647151708602905,
"learning_rate": 0.0001994524061386069,
"loss": 2.5089,
"step": 9600
},
{
"epoch": 0.03367742753060263,
"grad_norm": 1.5688285827636719,
"learning_rate": 0.0001994409480125246,
"loss": 2.5003,
"step": 9700
},
{
"epoch": 0.03402461750514492,
"grad_norm": 1.8745393753051758,
"learning_rate": 0.00019942937158268,
"loss": 2.4874,
"step": 9800
},
{
"epoch": 0.03437180747968722,
"grad_norm": 1.7022066116333008,
"learning_rate": 0.0001994176768628454,
"loss": 2.5015,
"step": 9900
},
{
"epoch": 0.03471899745422951,
"grad_norm": 1.9979243278503418,
"learning_rate": 0.00019940586386693393,
"loss": 2.5114,
"step": 10000
},
{
"epoch": 0.03506618742877181,
"grad_norm": 1.6992466449737549,
"learning_rate": 0.00019939393260899932,
"loss": 2.4883,
"step": 10100
},
{
"epoch": 0.0354133774033141,
"grad_norm": 2.0469751358032227,
"learning_rate": 0.0001993818831032361,
"loss": 2.5112,
"step": 10200
},
{
"epoch": 0.0357605673778564,
"grad_norm": 1.9578917026519775,
"learning_rate": 0.0001993697153639794,
"loss": 2.4968,
"step": 10300
},
{
"epoch": 0.03610775735239869,
"grad_norm": 2.140726089477539,
"learning_rate": 0.000199357429405705,
"loss": 2.5204,
"step": 10400
},
{
"epoch": 0.03645494732694099,
"grad_norm": 1.6723066568374634,
"learning_rate": 0.00019934502524302948,
"loss": 2.5112,
"step": 10500
},
{
"epoch": 0.03680213730148328,
"grad_norm": 1.6627919673919678,
"learning_rate": 0.00019933250289070982,
"loss": 2.5194,
"step": 10600
},
{
"epoch": 0.03714932727602558,
"grad_norm": 1.9078980684280396,
"learning_rate": 0.00019931986236364379,
"loss": 2.519,
"step": 10700
},
{
"epoch": 0.03749651725056787,
"grad_norm": 1.7164194583892822,
"learning_rate": 0.00019930710367686963,
"loss": 2.5215,
"step": 10800
},
{
"epoch": 0.03784370722511017,
"grad_norm": 3.068312883377075,
"learning_rate": 0.0001992942268455662,
"loss": 2.4945,
"step": 10900
},
{
"epoch": 0.03819089719965246,
"grad_norm": 2.505039691925049,
"learning_rate": 0.00019928123188505298,
"loss": 2.5483,
"step": 11000
},
{
"epoch": 0.038538087174194754,
"grad_norm": 2.145158529281616,
"learning_rate": 0.0001992681188107899,
"loss": 2.5419,
"step": 11100
},
{
"epoch": 0.03888527714873705,
"grad_norm": 1.8821165561676025,
"learning_rate": 0.00019925488763837743,
"loss": 2.5087,
"step": 11200
},
{
"epoch": 0.039232467123279345,
"grad_norm": 2.026528835296631,
"learning_rate": 0.00019924153838355653,
"loss": 2.5185,
"step": 11300
},
{
"epoch": 0.039579657097821644,
"grad_norm": 1.9918049573898315,
"learning_rate": 0.00019922807106220866,
"loss": 2.5275,
"step": 11400
},
{
"epoch": 0.039926847072363936,
"grad_norm": 2.444620132446289,
"learning_rate": 0.00019921448569035577,
"loss": 2.526,
"step": 11500
},
{
"epoch": 0.040274037046906234,
"grad_norm": 1.7009233236312866,
"learning_rate": 0.00019920078228416018,
"loss": 2.5264,
"step": 11600
},
{
"epoch": 0.040621227021448526,
"grad_norm": 1.8496342897415161,
"learning_rate": 0.0001991869608599247,
"loss": 2.535,
"step": 11700
},
{
"epoch": 0.040968416995990825,
"grad_norm": 2.300175905227661,
"learning_rate": 0.0001991730214340925,
"loss": 2.5206,
"step": 11800
},
{
"epoch": 0.04131560697053312,
"grad_norm": 1.6839985847473145,
"learning_rate": 0.00019915896402324722,
"loss": 2.4998,
"step": 11900
},
{
"epoch": 0.041662796945075416,
"grad_norm": 2.480787754058838,
"learning_rate": 0.00019914478864411272,
"loss": 2.535,
"step": 12000
},
{
"epoch": 0.04200998691961771,
"grad_norm": 2.3724234104156494,
"learning_rate": 0.00019913049531355333,
"loss": 2.5177,
"step": 12100
},
{
"epoch": 0.042357176894160006,
"grad_norm": 1.5725799798965454,
"learning_rate": 0.00019911608404857365,
"loss": 2.5088,
"step": 12200
},
{
"epoch": 0.0427043668687023,
"grad_norm": 2.066396951675415,
"learning_rate": 0.0001991015548663186,
"loss": 2.508,
"step": 12300
},
{
"epoch": 0.0430515568432446,
"grad_norm": 1.7157249450683594,
"learning_rate": 0.0001990869077840734,
"loss": 2.4944,
"step": 12400
},
{
"epoch": 0.04339874681778689,
"grad_norm": 1.8114676475524902,
"learning_rate": 0.00019907214281926348,
"loss": 2.5094,
"step": 12500
},
{
"epoch": 0.04374593679232919,
"grad_norm": 2.00364351272583,
"learning_rate": 0.00019905725998945457,
"loss": 2.4996,
"step": 12600
},
{
"epoch": 0.04409312676687148,
"grad_norm": 1.508483648300171,
"learning_rate": 0.00019904225931235262,
"loss": 2.5075,
"step": 12700
},
{
"epoch": 0.04444031674141378,
"grad_norm": 1.5554753541946411,
"learning_rate": 0.00019902714080580373,
"loss": 2.5246,
"step": 12800
},
{
"epoch": 0.04478750671595607,
"grad_norm": 2.1974375247955322,
"learning_rate": 0.00019901190448779423,
"loss": 2.5293,
"step": 12900
},
{
"epoch": 0.04513469669049836,
"grad_norm": 2.0555918216705322,
"learning_rate": 0.00019899655037645064,
"loss": 2.5065,
"step": 13000
}
],
"logging_steps": 100,
"max_steps": 288027,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 3.5663994175872614e+18,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}