Download trainer_state.json from PIXELZX/XION0.2-27B: direct link, hf CLI and curl.
- Browser
- Download file 13.5 kB
-
https://huggingface.co/PIXELZX/XION0.2-27B/resolve/main/trainer_state.json
- Command line
-
hf download hf://PIXELZX/XION0.2-27B/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/PIXELZX/XION0.2-27B/resolve/main/trainer_state.json
13.5 kB
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 3.0, | |
| "eval_steps": 22, | |
| "global_step": 66, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.045454545454545456, | |
| "grad_norm": 13.541594505310059, | |
| "learning_rate": 1e-05, | |
| "loss": 1.687255859375, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.09090909090909091, | |
| "grad_norm": 270.9620056152344, | |
| "learning_rate": 9.994336695915041e-06, | |
| "loss": 4.320972442626953, | |
| "step": 2 | |
| }, | |
| { | |
| "epoch": 0.13636363636363635, | |
| "grad_norm": 8.045546531677246, | |
| "learning_rate": 9.977359612865424e-06, | |
| "loss": 0.990966796875, | |
| "step": 3 | |
| }, | |
| { | |
| "epoch": 0.18181818181818182, | |
| "grad_norm": 4.051703453063965, | |
| "learning_rate": 9.949107209404664e-06, | |
| "loss": 1.09613037109375, | |
| "step": 4 | |
| }, | |
| { | |
| "epoch": 0.22727272727272727, | |
| "grad_norm": 2.503697633743286, | |
| "learning_rate": 9.909643486313533e-06, | |
| "loss": 1.016815185546875, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.2727272727272727, | |
| "grad_norm": 3.1819581985473633, | |
| "learning_rate": 9.859057841617709e-06, | |
| "loss": 0.91766357421875, | |
| "step": 6 | |
| }, | |
| { | |
| "epoch": 0.3181818181818182, | |
| "grad_norm": 1.810927391052246, | |
| "learning_rate": 9.797464868072489e-06, | |
| "loss": 0.9587249755859375, | |
| "step": 7 | |
| }, | |
| { | |
| "epoch": 0.36363636363636365, | |
| "grad_norm": 1.6619356870651245, | |
| "learning_rate": 9.725004093573343e-06, | |
| "loss": 0.80902099609375, | |
| "step": 8 | |
| }, | |
| { | |
| "epoch": 0.4090909090909091, | |
| "grad_norm": 2.2837491035461426, | |
| "learning_rate": 9.641839665080363e-06, | |
| "loss": 0.8079757690429688, | |
| "step": 9 | |
| }, | |
| { | |
| "epoch": 0.45454545454545453, | |
| "grad_norm": 3.019440174102783, | |
| "learning_rate": 9.548159976772593e-06, | |
| "loss": 0.8288116455078125, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.5, | |
| "grad_norm": 1.1921716928482056, | |
| "learning_rate": 9.444177243274619e-06, | |
| "loss": 0.6759376525878906, | |
| "step": 11 | |
| }, | |
| { | |
| "epoch": 0.5454545454545454, | |
| "grad_norm": 2.224130153656006, | |
| "learning_rate": 9.330127018922195e-06, | |
| "loss": 0.54034423828125, | |
| "step": 12 | |
| }, | |
| { | |
| "epoch": 0.5909090909090909, | |
| "grad_norm": 1.386547327041626, | |
| "learning_rate": 9.206267664155906e-06, | |
| "loss": 0.49675750732421875, | |
| "step": 13 | |
| }, | |
| { | |
| "epoch": 0.6363636363636364, | |
| "grad_norm": 1.4452524185180664, | |
| "learning_rate": 9.07287976025168e-06, | |
| "loss": 0.4326019287109375, | |
| "step": 14 | |
| }, | |
| { | |
| "epoch": 0.6818181818181818, | |
| "grad_norm": 2.8639578819274902, | |
| "learning_rate": 8.930265473713939e-06, | |
| "loss": 0.587005615234375, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.7272727272727273, | |
| "grad_norm": 1.1509618759155273, | |
| "learning_rate": 8.778747871771293e-06, | |
| "loss": 0.49512481689453125, | |
| "step": 16 | |
| }, | |
| { | |
| "epoch": 0.7727272727272727, | |
| "grad_norm": 1.06486177444458, | |
| "learning_rate": 8.61867019052535e-06, | |
| "loss": 0.6640739440917969, | |
| "step": 17 | |
| }, | |
| { | |
| "epoch": 0.8181818181818182, | |
| "grad_norm": 0.9991989731788635, | |
| "learning_rate": 8.450395057410561e-06, | |
| "loss": 0.437957763671875, | |
| "step": 18 | |
| }, | |
| { | |
| "epoch": 0.8636363636363636, | |
| "grad_norm": 0.7819029092788696, | |
| "learning_rate": 8.274303669726427e-06, | |
| "loss": 0.5671710968017578, | |
| "step": 19 | |
| }, | |
| { | |
| "epoch": 0.9090909090909091, | |
| "grad_norm": 0.8972281813621521, | |
| "learning_rate": 8.090794931103026e-06, | |
| "loss": 0.3094291687011719, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.9545454545454546, | |
| "grad_norm": 0.8230551481246948, | |
| "learning_rate": 7.900284547855992e-06, | |
| "loss": 0.42014312744140625, | |
| "step": 21 | |
| }, | |
| { | |
| "epoch": 1.0, | |
| "grad_norm": 0.9112458229064941, | |
| "learning_rate": 7.703204087277989e-06, | |
| "loss": 0.3252525329589844, | |
| "step": 22 | |
| }, | |
| { | |
| "epoch": 1.0, | |
| "eval_loss": 0.59442138671875, | |
| "eval_runtime": 13.685, | |
| "eval_samples_per_second": 0.365, | |
| "eval_steps_per_second": 0.073, | |
| "step": 22 | |
| }, | |
| { | |
| "checkpoint_runtime": 130.4115 | |
| }, | |
| { | |
| "epoch": 1.0454545454545454, | |
| "grad_norm": 1.6123707294464111, | |
| "learning_rate": 7.500000000000001e-06, | |
| "loss": 0.5166397094726562, | |
| "step": 23 | |
| }, | |
| { | |
| "epoch": 1.0909090909090908, | |
| "grad_norm": 1.3553637266159058, | |
| "learning_rate": 7.291132608637053e-06, | |
| "loss": 0.3297233581542969, | |
| "step": 24 | |
| }, | |
| { | |
| "epoch": 1.1363636363636362, | |
| "grad_norm": 1.3093994855880737, | |
| "learning_rate": 7.0770750650094335e-06, | |
| "loss": 0.49069786071777344, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 1.1818181818181819, | |
| "grad_norm": 0.996208667755127, | |
| "learning_rate": 6.858312278301638e-06, | |
| "loss": 0.2602996826171875, | |
| "step": 26 | |
| }, | |
| { | |
| "epoch": 1.2272727272727273, | |
| "grad_norm": 0.8691992163658142, | |
| "learning_rate": 6.635339816587109e-06, | |
| "loss": 0.09621810913085938, | |
| "step": 27 | |
| }, | |
| { | |
| "epoch": 1.2727272727272727, | |
| "grad_norm": 0.73832106590271, | |
| "learning_rate": 6.408662784207149e-06, | |
| "loss": 0.2693767547607422, | |
| "step": 28 | |
| }, | |
| { | |
| "epoch": 1.3181818181818181, | |
| "grad_norm": 0.9352995753288269, | |
| "learning_rate": 6.178794677547138e-06, | |
| "loss": 0.2981605529785156, | |
| "step": 29 | |
| }, | |
| { | |
| "epoch": 1.3636363636363638, | |
| "grad_norm": 0.6079849004745483, | |
| "learning_rate": 5.946256221802052e-06, | |
| "loss": 0.26114463806152344, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 1.4090909090909092, | |
| "grad_norm": 1.1613916158676147, | |
| "learning_rate": 5.711574191366427e-06, | |
| "loss": 0.3663311004638672, | |
| "step": 31 | |
| }, | |
| { | |
| "epoch": 1.4545454545454546, | |
| "grad_norm": 1.0265836715698242, | |
| "learning_rate": 5.475280216520913e-06, | |
| "loss": 0.13217544555664062, | |
| "step": 32 | |
| }, | |
| { | |
| "epoch": 1.5, | |
| "grad_norm": 0.9267441630363464, | |
| "learning_rate": 5.237909579118713e-06, | |
| "loss": 0.3060760498046875, | |
| "step": 33 | |
| }, | |
| { | |
| "epoch": 1.5454545454545454, | |
| "grad_norm": 0.6758596301078796, | |
| "learning_rate": 5e-06, | |
| "loss": 0.053920745849609375, | |
| "step": 34 | |
| }, | |
| { | |
| "epoch": 1.5909090909090908, | |
| "grad_norm": 0.6371091604232788, | |
| "learning_rate": 4.762090420881289e-06, | |
| "loss": 0.11317062377929688, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 1.6363636363636362, | |
| "grad_norm": 0.6892620325088501, | |
| "learning_rate": 4.524719783479088e-06, | |
| "loss": 0.1634387969970703, | |
| "step": 36 | |
| }, | |
| { | |
| "epoch": 1.6818181818181817, | |
| "grad_norm": 1.267598032951355, | |
| "learning_rate": 4.2884258086335755e-06, | |
| "loss": 0.1157989501953125, | |
| "step": 37 | |
| }, | |
| { | |
| "epoch": 1.7272727272727273, | |
| "grad_norm": 0.618965208530426, | |
| "learning_rate": 4.053743778197951e-06, | |
| "loss": 0.2590446472167969, | |
| "step": 38 | |
| }, | |
| { | |
| "epoch": 1.7727272727272727, | |
| "grad_norm": 0.8626583814620972, | |
| "learning_rate": 3.821205322452863e-06, | |
| "loss": 0.4122934341430664, | |
| "step": 39 | |
| }, | |
| { | |
| "epoch": 1.8181818181818183, | |
| "grad_norm": 0.5675498843193054, | |
| "learning_rate": 3.5913372157928515e-06, | |
| "loss": 0.18222999572753906, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 1.8636363636363638, | |
| "grad_norm": 1.346577525138855, | |
| "learning_rate": 3.3646601834128924e-06, | |
| "loss": 0.36645030975341797, | |
| "step": 41 | |
| }, | |
| { | |
| "epoch": 1.9090909090909092, | |
| "grad_norm": 0.5783951282501221, | |
| "learning_rate": 3.141687721698363e-06, | |
| "loss": 0.19093704223632812, | |
| "step": 42 | |
| }, | |
| { | |
| "epoch": 1.9545454545454546, | |
| "grad_norm": 0.7677918672561646, | |
| "learning_rate": 2.9229249349905686e-06, | |
| "loss": 0.24652671813964844, | |
| "step": 43 | |
| }, | |
| { | |
| "epoch": 2.0, | |
| "grad_norm": 0.6522639393806458, | |
| "learning_rate": 2.708867391362948e-06, | |
| "loss": 0.13533401489257812, | |
| "step": 44 | |
| }, | |
| { | |
| "epoch": 2.0, | |
| "eval_loss": 0.59344482421875, | |
| "eval_runtime": 4.3126, | |
| "eval_samples_per_second": 1.159, | |
| "eval_steps_per_second": 0.232, | |
| "step": 44 | |
| }, | |
| { | |
| "checkpoint_runtime": 134.2782 | |
| }, | |
| { | |
| "epoch": 2.0454545454545454, | |
| "grad_norm": 1.315116286277771, | |
| "learning_rate": 2.5000000000000015e-06, | |
| "loss": 0.2724342346191406, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 2.090909090909091, | |
| "grad_norm": 0.5187699198722839, | |
| "learning_rate": 2.296795912722014e-06, | |
| "loss": 0.18209326267242432, | |
| "step": 46 | |
| }, | |
| { | |
| "epoch": 2.1363636363636362, | |
| "grad_norm": 0.9916004538536072, | |
| "learning_rate": 2.09971545214401e-06, | |
| "loss": 0.2614755630493164, | |
| "step": 47 | |
| }, | |
| { | |
| "epoch": 2.1818181818181817, | |
| "grad_norm": 0.5112076997756958, | |
| "learning_rate": 1.9092050688969736e-06, | |
| "loss": 0.1507863998413086, | |
| "step": 48 | |
| }, | |
| { | |
| "epoch": 2.227272727272727, | |
| "grad_norm": 0.3336804211139679, | |
| "learning_rate": 1.7256963302735752e-06, | |
| "loss": 0.024288177490234375, | |
| "step": 49 | |
| }, | |
| { | |
| "epoch": 2.2727272727272725, | |
| "grad_norm": 0.40631377696990967, | |
| "learning_rate": 1.549604942589441e-06, | |
| "loss": 0.19109153747558594, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 2.3181818181818183, | |
| "grad_norm": 0.8899633288383484, | |
| "learning_rate": 1.3813298094746491e-06, | |
| "loss": 0.17161941528320312, | |
| "step": 51 | |
| }, | |
| { | |
| "epoch": 2.3636363636363638, | |
| "grad_norm": 0.5578643679618835, | |
| "learning_rate": 1.2212521282287093e-06, | |
| "loss": 0.17924213409423828, | |
| "step": 52 | |
| }, | |
| { | |
| "epoch": 2.409090909090909, | |
| "grad_norm": 1.057443618774414, | |
| "learning_rate": 1.0697345262860638e-06, | |
| "loss": 0.21059799194335938, | |
| "step": 53 | |
| }, | |
| { | |
| "epoch": 2.4545454545454546, | |
| "grad_norm": 0.6124624609947205, | |
| "learning_rate": 9.271202397483214e-07, | |
| "loss": 0.050838470458984375, | |
| "step": 54 | |
| }, | |
| { | |
| "epoch": 2.5, | |
| "grad_norm": 0.5138548016548157, | |
| "learning_rate": 7.937323358440935e-07, | |
| "loss": 0.22317028045654297, | |
| "step": 55 | |
| }, | |
| { | |
| "epoch": 2.5454545454545454, | |
| "grad_norm": 0.5870931148529053, | |
| "learning_rate": 6.698729810778065e-07, | |
| "loss": 0.02097320556640625, | |
| "step": 56 | |
| }, | |
| { | |
| "epoch": 2.590909090909091, | |
| "grad_norm": 0.47857600450515747, | |
| "learning_rate": 5.558227567253832e-07, | |
| "loss": 0.056339263916015625, | |
| "step": 57 | |
| }, | |
| { | |
| "epoch": 2.6363636363636362, | |
| "grad_norm": 0.4128285050392151, | |
| "learning_rate": 4.5184002322740784e-07, | |
| "loss": 0.10932445526123047, | |
| "step": 58 | |
| }, | |
| { | |
| "epoch": 2.6818181818181817, | |
| "grad_norm": 0.9088109135627747, | |
| "learning_rate": 3.581603349196372e-07, | |
| "loss": 0.04517364501953125, | |
| "step": 59 | |
| }, | |
| { | |
| "epoch": 2.7272727272727275, | |
| "grad_norm": 0.5057564973831177, | |
| "learning_rate": 2.7499590642665773e-07, | |
| "loss": 0.19766902923583984, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 2.7727272727272725, | |
| "grad_norm": 0.792408287525177, | |
| "learning_rate": 2.0253513192751374e-07, | |
| "loss": 0.33954524993896484, | |
| "step": 61 | |
| }, | |
| { | |
| "epoch": 2.8181818181818183, | |
| "grad_norm": 0.47474050521850586, | |
| "learning_rate": 1.4094215838229176e-07, | |
| "loss": 0.12943077087402344, | |
| "step": 62 | |
| }, | |
| { | |
| "epoch": 2.8636363636363638, | |
| "grad_norm": 0.9088992476463318, | |
| "learning_rate": 9.035651368646647e-08, | |
| "loss": 0.3276538848876953, | |
| "step": 63 | |
| }, | |
| { | |
| "epoch": 2.909090909090909, | |
| "grad_norm": 0.4451668858528137, | |
| "learning_rate": 5.089279059533658e-08, | |
| "loss": 0.1589193344116211, | |
| "step": 64 | |
| }, | |
| { | |
| "epoch": 2.9545454545454546, | |
| "grad_norm": 0.6257097721099854, | |
| "learning_rate": 2.264038713457706e-08, | |
| "loss": 0.2027444839477539, | |
| "step": 65 | |
| }, | |
| { | |
| "epoch": 3.0, | |
| "grad_norm": 0.4987045228481293, | |
| "learning_rate": 5.6633040849601865e-09, | |
| "loss": 0.09051704406738281, | |
| "step": 66 | |
| }, | |
| { | |
| "epoch": 3.0, | |
| "eval_loss": 0.61431884765625, | |
| "eval_runtime": 4.3337, | |
| "eval_samples_per_second": 1.154, | |
| "eval_steps_per_second": 0.231, | |
| "step": 66 | |
| } | |
| ], | |
| "logging_steps": 1.0, | |
| "max_steps": 66, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 3, | |
| "save_steps": 22, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 8.409622391826678e+17, | |
| "train_batch_size": 1, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |