sindhusatish97's picture
Upload QLoRA SFT checkpoint for Qwen/Qwen2.5-14B
51f8123 verified
Raw
History Blame Contribute Delete
8.28 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.2144987766866642,
"eval_steps": 500,
"global_step": 800,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.005362469417166605,
"grad_norm": 0.050072263926267624,
"learning_rate": 1.4961796246648793e-05,
"loss": 1.0673207283020019,
"step": 20
},
{
"epoch": 0.01072493883433321,
"grad_norm": 0.06825340539216995,
"learning_rate": 1.4921581769436997e-05,
"loss": 0.9185627937316895,
"step": 40
},
{
"epoch": 0.016087408251499815,
"grad_norm": 0.06827432662248611,
"learning_rate": 1.48813672922252e-05,
"loss": 0.7999343872070312,
"step": 60
},
{
"epoch": 0.02144987766866642,
"grad_norm": 0.05807405710220337,
"learning_rate": 1.4841152815013404e-05,
"loss": 0.7322770595550537,
"step": 80
},
{
"epoch": 0.026812347085833025,
"grad_norm": 0.06654328852891922,
"learning_rate": 1.4800938337801608e-05,
"loss": 0.7097890377044678,
"step": 100
},
{
"epoch": 0.03217481650299963,
"grad_norm": 0.09104783087968826,
"learning_rate": 1.4760723860589812e-05,
"loss": 0.6513629913330078,
"step": 120
},
{
"epoch": 0.03753728592016624,
"grad_norm": 0.10718850791454315,
"learning_rate": 1.4720509383378015e-05,
"loss": 0.678717851638794,
"step": 140
},
{
"epoch": 0.04289975533733284,
"grad_norm": 0.09187154471874237,
"learning_rate": 1.4680294906166219e-05,
"loss": 0.647278118133545,
"step": 160
},
{
"epoch": 0.04826222475449945,
"grad_norm": 0.07148946076631546,
"learning_rate": 1.4640080428954423e-05,
"loss": 0.6737877368927002,
"step": 180
},
{
"epoch": 0.05362469417166605,
"grad_norm": 0.08909227699041367,
"learning_rate": 1.4599865951742626e-05,
"loss": 0.6373191356658936,
"step": 200
},
{
"epoch": 0.05898716358883266,
"grad_norm": 0.07850278168916702,
"learning_rate": 1.455965147453083e-05,
"loss": 0.6020126819610596,
"step": 220
},
{
"epoch": 0.06434963300599926,
"grad_norm": 0.09538089483976364,
"learning_rate": 1.4519436997319034e-05,
"loss": 0.6096773147583008,
"step": 240
},
{
"epoch": 0.06971210242316586,
"grad_norm": 0.07478228211402893,
"learning_rate": 1.447922252010724e-05,
"loss": 0.6299086093902588,
"step": 260
},
{
"epoch": 0.07507457184033248,
"grad_norm": 0.1514953374862671,
"learning_rate": 1.4439008042895443e-05,
"loss": 0.5591042518615723,
"step": 280
},
{
"epoch": 0.08043704125749908,
"grad_norm": 0.08260886371135712,
"learning_rate": 1.4398793565683647e-05,
"loss": 0.6200376987457276,
"step": 300
},
{
"epoch": 0.08579951067466568,
"grad_norm": 0.17698714137077332,
"learning_rate": 1.435857908847185e-05,
"loss": 0.6023219585418701,
"step": 320
},
{
"epoch": 0.0911619800918323,
"grad_norm": 0.06104859337210655,
"learning_rate": 1.4318364611260054e-05,
"loss": 0.6181454658508301,
"step": 340
},
{
"epoch": 0.0965244495089989,
"grad_norm": 0.04990549385547638,
"learning_rate": 1.4278150134048258e-05,
"loss": 0.5593632698059082,
"step": 360
},
{
"epoch": 0.1018869189261655,
"grad_norm": 0.09426380693912506,
"learning_rate": 1.4237935656836461e-05,
"loss": 0.5790591716766358,
"step": 380
},
{
"epoch": 0.1072493883433321,
"grad_norm": 0.08783263713121414,
"learning_rate": 1.4197721179624665e-05,
"loss": 0.585063886642456,
"step": 400
},
{
"epoch": 0.11261185776049872,
"grad_norm": 0.06869607418775558,
"learning_rate": 1.4157506702412869e-05,
"loss": 0.5638764381408692,
"step": 420
},
{
"epoch": 0.11797432717766532,
"grad_norm": 0.10537438839673996,
"learning_rate": 1.4117292225201072e-05,
"loss": 0.6060166835784913,
"step": 440
},
{
"epoch": 0.12333679659483192,
"grad_norm": 0.09851580113172531,
"learning_rate": 1.4077077747989278e-05,
"loss": 0.5605969905853272,
"step": 460
},
{
"epoch": 0.12869926601199852,
"grad_norm": 0.11954096704721451,
"learning_rate": 1.4036863270777482e-05,
"loss": 0.5549856662750244,
"step": 480
},
{
"epoch": 0.13406173542916514,
"grad_norm": 0.13259431719779968,
"learning_rate": 1.3996648793565685e-05,
"loss": 0.5893547534942627,
"step": 500
},
{
"epoch": 0.13942420484633172,
"grad_norm": 0.11842650175094604,
"learning_rate": 1.3956434316353889e-05,
"loss": 0.6237683773040772,
"step": 520
},
{
"epoch": 0.14478667426349834,
"grad_norm": 0.1204022690653801,
"learning_rate": 1.3916219839142093e-05,
"loss": 0.572803258895874,
"step": 540
},
{
"epoch": 0.15014914368066495,
"grad_norm": 0.1345946341753006,
"learning_rate": 1.3876005361930296e-05,
"loss": 0.5632933139801025,
"step": 560
},
{
"epoch": 0.15551161309783154,
"grad_norm": 0.11733393371105194,
"learning_rate": 1.38357908847185e-05,
"loss": 0.6197309494018555,
"step": 580
},
{
"epoch": 0.16087408251499816,
"grad_norm": 0.0731734186410904,
"learning_rate": 1.3795576407506704e-05,
"loss": 0.5823808670043945,
"step": 600
},
{
"epoch": 0.16623655193216477,
"grad_norm": 0.09452618658542633,
"learning_rate": 1.3755361930294907e-05,
"loss": 0.5599356651306152,
"step": 620
},
{
"epoch": 0.17159902134933136,
"grad_norm": 0.09183815121650696,
"learning_rate": 1.3715147453083111e-05,
"loss": 0.5465828895568847,
"step": 640
},
{
"epoch": 0.17696149076649798,
"grad_norm": 0.0953364372253418,
"learning_rate": 1.3674932975871315e-05,
"loss": 0.5516108989715576,
"step": 660
},
{
"epoch": 0.1823239601836646,
"grad_norm": 0.11190114170312881,
"learning_rate": 1.3634718498659519e-05,
"loss": 0.5717048645019531,
"step": 680
},
{
"epoch": 0.18768642960083118,
"grad_norm": 0.11502158641815186,
"learning_rate": 1.3594504021447722e-05,
"loss": 0.528355598449707,
"step": 700
},
{
"epoch": 0.1930488990179978,
"grad_norm": 0.12480133026838303,
"learning_rate": 1.3554289544235926e-05,
"loss": 0.5860391616821289,
"step": 720
},
{
"epoch": 0.19841136843516438,
"grad_norm": 0.14408785104751587,
"learning_rate": 1.351407506702413e-05,
"loss": 0.5422697544097901,
"step": 740
},
{
"epoch": 0.203773837852331,
"grad_norm": 0.12405668199062347,
"learning_rate": 1.3473860589812333e-05,
"loss": 0.5876667499542236,
"step": 760
},
{
"epoch": 0.2091363072694976,
"grad_norm": 0.12171291559934616,
"learning_rate": 1.3433646112600537e-05,
"loss": 0.563751220703125,
"step": 780
},
{
"epoch": 0.2144987766866642,
"grad_norm": 0.10827518254518509,
"learning_rate": 1.339343163538874e-05,
"loss": 0.5700247764587403,
"step": 800
}
],
"logging_steps": 20,
"max_steps": 7460,
"num_input_tokens_seen": 0,
"num_train_epochs": 2,
"save_steps": 200,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 9.803862124388966e+16,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}