Instructions to use 0x1202/ecb8379e-81b0-4bd5-a690-c05a5108b46a with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use 0x1202/ecb8379e-81b0-4bd5-a690-c05a5108b46a with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("unsloth/Hermes-3-Llama-3.1-8B") model = PeftModel.from_pretrained(base_model, "0x1202/ecb8379e-81b0-4bd5-a690-c05a5108b46a") - Notebooks
- Google Colab
- Kaggle
| { | |
| "best_metric": 1.4643950462341309, | |
| "best_model_checkpoint": "miner_id_24/checkpoint-200", | |
| "epoch": 0.6872852233676976, | |
| "eval_steps": 50, | |
| "global_step": 200, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.003436426116838488, | |
| "grad_norm": 1.5098187923431396, | |
| "learning_rate": 1e-05, | |
| "loss": 2.124, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.003436426116838488, | |
| "eval_loss": 2.3156180381774902, | |
| "eval_runtime": 41.6515, | |
| "eval_samples_per_second": 11.764, | |
| "eval_steps_per_second": 2.953, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.006872852233676976, | |
| "grad_norm": 1.4637274742126465, | |
| "learning_rate": 2e-05, | |
| "loss": 2.1284, | |
| "step": 2 | |
| }, | |
| { | |
| "epoch": 0.010309278350515464, | |
| "grad_norm": 1.4501724243164062, | |
| "learning_rate": 3e-05, | |
| "loss": 2.1827, | |
| "step": 3 | |
| }, | |
| { | |
| "epoch": 0.013745704467353952, | |
| "grad_norm": 1.4608981609344482, | |
| "learning_rate": 4e-05, | |
| "loss": 2.1878, | |
| "step": 4 | |
| }, | |
| { | |
| "epoch": 0.01718213058419244, | |
| "grad_norm": 1.3086926937103271, | |
| "learning_rate": 5e-05, | |
| "loss": 2.158, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.020618556701030927, | |
| "grad_norm": 0.9834317564964294, | |
| "learning_rate": 6e-05, | |
| "loss": 1.937, | |
| "step": 6 | |
| }, | |
| { | |
| "epoch": 0.024054982817869417, | |
| "grad_norm": 0.6933495402336121, | |
| "learning_rate": 7e-05, | |
| "loss": 2.0259, | |
| "step": 7 | |
| }, | |
| { | |
| "epoch": 0.027491408934707903, | |
| "grad_norm": 1.041727900505066, | |
| "learning_rate": 8e-05, | |
| "loss": 1.9273, | |
| "step": 8 | |
| }, | |
| { | |
| "epoch": 0.030927835051546393, | |
| "grad_norm": 1.1928552389144897, | |
| "learning_rate": 9e-05, | |
| "loss": 1.7531, | |
| "step": 9 | |
| }, | |
| { | |
| "epoch": 0.03436426116838488, | |
| "grad_norm": 1.0820860862731934, | |
| "learning_rate": 0.0001, | |
| "loss": 2.0411, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.037800687285223365, | |
| "grad_norm": 0.6688265800476074, | |
| "learning_rate": 9.999316524962345e-05, | |
| "loss": 1.822, | |
| "step": 11 | |
| }, | |
| { | |
| "epoch": 0.041237113402061855, | |
| "grad_norm": 0.5884153842926025, | |
| "learning_rate": 9.997266286704631e-05, | |
| "loss": 2.0072, | |
| "step": 12 | |
| }, | |
| { | |
| "epoch": 0.044673539518900345, | |
| "grad_norm": 0.6440837979316711, | |
| "learning_rate": 9.993849845741524e-05, | |
| "loss": 1.8632, | |
| "step": 13 | |
| }, | |
| { | |
| "epoch": 0.048109965635738834, | |
| "grad_norm": 0.6202774047851562, | |
| "learning_rate": 9.989068136093873e-05, | |
| "loss": 1.7895, | |
| "step": 14 | |
| }, | |
| { | |
| "epoch": 0.05154639175257732, | |
| "grad_norm": 0.6742629408836365, | |
| "learning_rate": 9.98292246503335e-05, | |
| "loss": 1.9201, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.054982817869415807, | |
| "grad_norm": 0.6068807244300842, | |
| "learning_rate": 9.975414512725057e-05, | |
| "loss": 1.7121, | |
| "step": 16 | |
| }, | |
| { | |
| "epoch": 0.058419243986254296, | |
| "grad_norm": 0.6303403377532959, | |
| "learning_rate": 9.966546331768191e-05, | |
| "loss": 1.7242, | |
| "step": 17 | |
| }, | |
| { | |
| "epoch": 0.061855670103092786, | |
| "grad_norm": 0.6808639168739319, | |
| "learning_rate": 9.956320346634876e-05, | |
| "loss": 1.709, | |
| "step": 18 | |
| }, | |
| { | |
| "epoch": 0.06529209621993128, | |
| "grad_norm": 0.6570630669593811, | |
| "learning_rate": 9.944739353007344e-05, | |
| "loss": 1.9049, | |
| "step": 19 | |
| }, | |
| { | |
| "epoch": 0.06872852233676977, | |
| "grad_norm": 0.7587320804595947, | |
| "learning_rate": 9.931806517013612e-05, | |
| "loss": 1.7553, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.07216494845360824, | |
| "grad_norm": 0.8001805543899536, | |
| "learning_rate": 9.917525374361912e-05, | |
| "loss": 1.6671, | |
| "step": 21 | |
| }, | |
| { | |
| "epoch": 0.07560137457044673, | |
| "grad_norm": 0.878776490688324, | |
| "learning_rate": 9.901899829374047e-05, | |
| "loss": 1.6972, | |
| "step": 22 | |
| }, | |
| { | |
| "epoch": 0.07903780068728522, | |
| "grad_norm": 0.7770026922225952, | |
| "learning_rate": 9.884934153917997e-05, | |
| "loss": 1.7453, | |
| "step": 23 | |
| }, | |
| { | |
| "epoch": 0.08247422680412371, | |
| "grad_norm": 0.856813371181488, | |
| "learning_rate": 9.86663298624003e-05, | |
| "loss": 1.4652, | |
| "step": 24 | |
| }, | |
| { | |
| "epoch": 0.0859106529209622, | |
| "grad_norm": 0.7691881060600281, | |
| "learning_rate": 9.847001329696653e-05, | |
| "loss": 1.6475, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 0.08934707903780069, | |
| "grad_norm": 0.7521170377731323, | |
| "learning_rate": 9.826044551386744e-05, | |
| "loss": 1.5121, | |
| "step": 26 | |
| }, | |
| { | |
| "epoch": 0.09278350515463918, | |
| "grad_norm": 0.8257611393928528, | |
| "learning_rate": 9.803768380684242e-05, | |
| "loss": 1.6485, | |
| "step": 27 | |
| }, | |
| { | |
| "epoch": 0.09621993127147767, | |
| "grad_norm": 0.8972536325454712, | |
| "learning_rate": 9.780178907671789e-05, | |
| "loss": 1.717, | |
| "step": 28 | |
| }, | |
| { | |
| "epoch": 0.09965635738831616, | |
| "grad_norm": 0.9242669939994812, | |
| "learning_rate": 9.755282581475769e-05, | |
| "loss": 1.4351, | |
| "step": 29 | |
| }, | |
| { | |
| "epoch": 0.10309278350515463, | |
| "grad_norm": 0.9386427998542786, | |
| "learning_rate": 9.729086208503174e-05, | |
| "loss": 1.5317, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.10652920962199312, | |
| "grad_norm": 0.8852460980415344, | |
| "learning_rate": 9.701596950580806e-05, | |
| "loss": 1.5161, | |
| "step": 31 | |
| }, | |
| { | |
| "epoch": 0.10996563573883161, | |
| "grad_norm": 0.953780472278595, | |
| "learning_rate": 9.672822322997305e-05, | |
| "loss": 1.5146, | |
| "step": 32 | |
| }, | |
| { | |
| "epoch": 0.1134020618556701, | |
| "grad_norm": 0.9156774878501892, | |
| "learning_rate": 9.642770192448536e-05, | |
| "loss": 1.4676, | |
| "step": 33 | |
| }, | |
| { | |
| "epoch": 0.11683848797250859, | |
| "grad_norm": 0.9750638604164124, | |
| "learning_rate": 9.611448774886924e-05, | |
| "loss": 1.3955, | |
| "step": 34 | |
| }, | |
| { | |
| "epoch": 0.12027491408934708, | |
| "grad_norm": 1.0895854234695435, | |
| "learning_rate": 9.578866633275288e-05, | |
| "loss": 1.5778, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 0.12371134020618557, | |
| "grad_norm": 1.0576300621032715, | |
| "learning_rate": 9.545032675245813e-05, | |
| "loss": 1.3418, | |
| "step": 36 | |
| }, | |
| { | |
| "epoch": 0.12714776632302405, | |
| "grad_norm": 1.1727350950241089, | |
| "learning_rate": 9.509956150664796e-05, | |
| "loss": 1.4531, | |
| "step": 37 | |
| }, | |
| { | |
| "epoch": 0.13058419243986255, | |
| "grad_norm": 1.2390329837799072, | |
| "learning_rate": 9.473646649103818e-05, | |
| "loss": 1.2666, | |
| "step": 38 | |
| }, | |
| { | |
| "epoch": 0.13402061855670103, | |
| "grad_norm": 1.385895013809204, | |
| "learning_rate": 9.43611409721806e-05, | |
| "loss": 1.4164, | |
| "step": 39 | |
| }, | |
| { | |
| "epoch": 0.13745704467353953, | |
| "grad_norm": 1.1739875078201294, | |
| "learning_rate": 9.397368756032445e-05, | |
| "loss": 1.2798, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.140893470790378, | |
| "grad_norm": 1.230619192123413, | |
| "learning_rate": 9.357421218136386e-05, | |
| "loss": 1.3988, | |
| "step": 41 | |
| }, | |
| { | |
| "epoch": 0.14432989690721648, | |
| "grad_norm": 1.536525011062622, | |
| "learning_rate": 9.316282404787871e-05, | |
| "loss": 1.0722, | |
| "step": 42 | |
| }, | |
| { | |
| "epoch": 0.14776632302405499, | |
| "grad_norm": 1.4745525121688843, | |
| "learning_rate": 9.273963562927695e-05, | |
| "loss": 1.36, | |
| "step": 43 | |
| }, | |
| { | |
| "epoch": 0.15120274914089346, | |
| "grad_norm": 1.437075138092041, | |
| "learning_rate": 9.230476262104677e-05, | |
| "loss": 1.1868, | |
| "step": 44 | |
| }, | |
| { | |
| "epoch": 0.15463917525773196, | |
| "grad_norm": 1.5111098289489746, | |
| "learning_rate": 9.185832391312644e-05, | |
| "loss": 1.2112, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 0.15807560137457044, | |
| "grad_norm": 1.5478641986846924, | |
| "learning_rate": 9.140044155740101e-05, | |
| "loss": 1.1819, | |
| "step": 46 | |
| }, | |
| { | |
| "epoch": 0.16151202749140894, | |
| "grad_norm": 1.713692545890808, | |
| "learning_rate": 9.093124073433463e-05, | |
| "loss": 1.0426, | |
| "step": 47 | |
| }, | |
| { | |
| "epoch": 0.16494845360824742, | |
| "grad_norm": 1.713294267654419, | |
| "learning_rate": 9.045084971874738e-05, | |
| "loss": 0.968, | |
| "step": 48 | |
| }, | |
| { | |
| "epoch": 0.16838487972508592, | |
| "grad_norm": 1.9837310314178467, | |
| "learning_rate": 8.995939984474624e-05, | |
| "loss": 0.9734, | |
| "step": 49 | |
| }, | |
| { | |
| "epoch": 0.1718213058419244, | |
| "grad_norm": 2.544609785079956, | |
| "learning_rate": 8.945702546981969e-05, | |
| "loss": 1.4537, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.1718213058419244, | |
| "eval_loss": 1.6402361392974854, | |
| "eval_runtime": 42.6957, | |
| "eval_samples_per_second": 11.477, | |
| "eval_steps_per_second": 2.881, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.17525773195876287, | |
| "grad_norm": 1.2197675704956055, | |
| "learning_rate": 8.894386393810563e-05, | |
| "loss": 1.9356, | |
| "step": 51 | |
| }, | |
| { | |
| "epoch": 0.17869415807560138, | |
| "grad_norm": 0.8116856217384338, | |
| "learning_rate": 8.842005554284296e-05, | |
| "loss": 2.0348, | |
| "step": 52 | |
| }, | |
| { | |
| "epoch": 0.18213058419243985, | |
| "grad_norm": 0.5514941215515137, | |
| "learning_rate": 8.788574348801675e-05, | |
| "loss": 1.9794, | |
| "step": 53 | |
| }, | |
| { | |
| "epoch": 0.18556701030927836, | |
| "grad_norm": 0.47552359104156494, | |
| "learning_rate": 8.73410738492077e-05, | |
| "loss": 2.0041, | |
| "step": 54 | |
| }, | |
| { | |
| "epoch": 0.18900343642611683, | |
| "grad_norm": 0.3883759081363678, | |
| "learning_rate": 8.678619553365659e-05, | |
| "loss": 1.9981, | |
| "step": 55 | |
| }, | |
| { | |
| "epoch": 0.19243986254295534, | |
| "grad_norm": 0.3964765667915344, | |
| "learning_rate": 8.622126023955446e-05, | |
| "loss": 1.8896, | |
| "step": 56 | |
| }, | |
| { | |
| "epoch": 0.1958762886597938, | |
| "grad_norm": 0.40116173028945923, | |
| "learning_rate": 8.564642241456986e-05, | |
| "loss": 1.8873, | |
| "step": 57 | |
| }, | |
| { | |
| "epoch": 0.19931271477663232, | |
| "grad_norm": 0.4208248555660248, | |
| "learning_rate": 8.506183921362443e-05, | |
| "loss": 1.8127, | |
| "step": 58 | |
| }, | |
| { | |
| "epoch": 0.2027491408934708, | |
| "grad_norm": 0.4747980833053589, | |
| "learning_rate": 8.44676704559283e-05, | |
| "loss": 1.6512, | |
| "step": 59 | |
| }, | |
| { | |
| "epoch": 0.20618556701030927, | |
| "grad_norm": 0.4318777322769165, | |
| "learning_rate": 8.386407858128706e-05, | |
| "loss": 1.8278, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.20962199312714777, | |
| "grad_norm": 0.5040547847747803, | |
| "learning_rate": 8.32512286056924e-05, | |
| "loss": 1.7725, | |
| "step": 61 | |
| }, | |
| { | |
| "epoch": 0.21305841924398625, | |
| "grad_norm": 0.48844555020332336, | |
| "learning_rate": 8.262928807620843e-05, | |
| "loss": 1.6757, | |
| "step": 62 | |
| }, | |
| { | |
| "epoch": 0.21649484536082475, | |
| "grad_norm": 0.5083892941474915, | |
| "learning_rate": 8.199842702516583e-05, | |
| "loss": 1.8451, | |
| "step": 63 | |
| }, | |
| { | |
| "epoch": 0.21993127147766323, | |
| "grad_norm": 0.5018050074577332, | |
| "learning_rate": 8.135881792367686e-05, | |
| "loss": 1.7201, | |
| "step": 64 | |
| }, | |
| { | |
| "epoch": 0.22336769759450173, | |
| "grad_norm": 0.5488980412483215, | |
| "learning_rate": 8.07106356344834e-05, | |
| "loss": 1.6684, | |
| "step": 65 | |
| }, | |
| { | |
| "epoch": 0.2268041237113402, | |
| "grad_norm": 0.5470010638237, | |
| "learning_rate": 8.005405736415126e-05, | |
| "loss": 1.7186, | |
| "step": 66 | |
| }, | |
| { | |
| "epoch": 0.23024054982817868, | |
| "grad_norm": 0.5802435278892517, | |
| "learning_rate": 7.938926261462366e-05, | |
| "loss": 1.5814, | |
| "step": 67 | |
| }, | |
| { | |
| "epoch": 0.23367697594501718, | |
| "grad_norm": 0.5508120059967041, | |
| "learning_rate": 7.871643313414718e-05, | |
| "loss": 1.7299, | |
| "step": 68 | |
| }, | |
| { | |
| "epoch": 0.23711340206185566, | |
| "grad_norm": 0.5759636759757996, | |
| "learning_rate": 7.803575286758364e-05, | |
| "loss": 1.503, | |
| "step": 69 | |
| }, | |
| { | |
| "epoch": 0.24054982817869416, | |
| "grad_norm": 0.5945281982421875, | |
| "learning_rate": 7.734740790612136e-05, | |
| "loss": 1.7036, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.24398625429553264, | |
| "grad_norm": 0.5784856081008911, | |
| "learning_rate": 7.66515864363997e-05, | |
| "loss": 1.534, | |
| "step": 71 | |
| }, | |
| { | |
| "epoch": 0.24742268041237114, | |
| "grad_norm": 0.6172903776168823, | |
| "learning_rate": 7.594847868906076e-05, | |
| "loss": 1.5518, | |
| "step": 72 | |
| }, | |
| { | |
| "epoch": 0.2508591065292096, | |
| "grad_norm": 0.6575155854225159, | |
| "learning_rate": 7.52382768867422e-05, | |
| "loss": 1.5429, | |
| "step": 73 | |
| }, | |
| { | |
| "epoch": 0.2542955326460481, | |
| "grad_norm": 0.6710128784179688, | |
| "learning_rate": 7.452117519152542e-05, | |
| "loss": 1.6403, | |
| "step": 74 | |
| }, | |
| { | |
| "epoch": 0.25773195876288657, | |
| "grad_norm": 0.7106481194496155, | |
| "learning_rate": 7.379736965185368e-05, | |
| "loss": 1.6003, | |
| "step": 75 | |
| }, | |
| { | |
| "epoch": 0.2611683848797251, | |
| "grad_norm": 0.7457655072212219, | |
| "learning_rate": 7.30670581489344e-05, | |
| "loss": 1.5166, | |
| "step": 76 | |
| }, | |
| { | |
| "epoch": 0.2646048109965636, | |
| "grad_norm": 0.8298302888870239, | |
| "learning_rate": 7.233044034264034e-05, | |
| "loss": 1.4188, | |
| "step": 77 | |
| }, | |
| { | |
| "epoch": 0.26804123711340205, | |
| "grad_norm": 0.7822423577308655, | |
| "learning_rate": 7.158771761692464e-05, | |
| "loss": 1.4283, | |
| "step": 78 | |
| }, | |
| { | |
| "epoch": 0.27147766323024053, | |
| "grad_norm": 0.7826134562492371, | |
| "learning_rate": 7.083909302476453e-05, | |
| "loss": 1.4861, | |
| "step": 79 | |
| }, | |
| { | |
| "epoch": 0.27491408934707906, | |
| "grad_norm": 0.7911308407783508, | |
| "learning_rate": 7.008477123264848e-05, | |
| "loss": 1.5385, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.27835051546391754, | |
| "grad_norm": 0.8717237114906311, | |
| "learning_rate": 6.932495846462261e-05, | |
| "loss": 1.5897, | |
| "step": 81 | |
| }, | |
| { | |
| "epoch": 0.281786941580756, | |
| "grad_norm": 0.810140073299408, | |
| "learning_rate": 6.855986244591104e-05, | |
| "loss": 1.3252, | |
| "step": 82 | |
| }, | |
| { | |
| "epoch": 0.2852233676975945, | |
| "grad_norm": 0.9688878059387207, | |
| "learning_rate": 6.778969234612584e-05, | |
| "loss": 1.4507, | |
| "step": 83 | |
| }, | |
| { | |
| "epoch": 0.28865979381443296, | |
| "grad_norm": 0.9685982465744019, | |
| "learning_rate": 6.701465872208216e-05, | |
| "loss": 1.521, | |
| "step": 84 | |
| }, | |
| { | |
| "epoch": 0.2920962199312715, | |
| "grad_norm": 0.8572627305984497, | |
| "learning_rate": 6.623497346023418e-05, | |
| "loss": 1.3448, | |
| "step": 85 | |
| }, | |
| { | |
| "epoch": 0.29553264604810997, | |
| "grad_norm": 0.9771127104759216, | |
| "learning_rate": 6.545084971874738e-05, | |
| "loss": 1.3461, | |
| "step": 86 | |
| }, | |
| { | |
| "epoch": 0.29896907216494845, | |
| "grad_norm": 1.0161019563674927, | |
| "learning_rate": 6.466250186922325e-05, | |
| "loss": 1.3099, | |
| "step": 87 | |
| }, | |
| { | |
| "epoch": 0.3024054982817869, | |
| "grad_norm": 1.0854941606521606, | |
| "learning_rate": 6.387014543809223e-05, | |
| "loss": 1.2461, | |
| "step": 88 | |
| }, | |
| { | |
| "epoch": 0.30584192439862545, | |
| "grad_norm": 1.0072003602981567, | |
| "learning_rate": 6.307399704769099e-05, | |
| "loss": 1.308, | |
| "step": 89 | |
| }, | |
| { | |
| "epoch": 0.30927835051546393, | |
| "grad_norm": 1.1473398208618164, | |
| "learning_rate": 6.227427435703997e-05, | |
| "loss": 1.1992, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.3127147766323024, | |
| "grad_norm": 1.1940810680389404, | |
| "learning_rate": 6.147119600233758e-05, | |
| "loss": 1.2966, | |
| "step": 91 | |
| }, | |
| { | |
| "epoch": 0.3161512027491409, | |
| "grad_norm": 1.736954689025879, | |
| "learning_rate": 6.066498153718735e-05, | |
| "loss": 0.9234, | |
| "step": 92 | |
| }, | |
| { | |
| "epoch": 0.31958762886597936, | |
| "grad_norm": 1.2535929679870605, | |
| "learning_rate": 5.985585137257401e-05, | |
| "loss": 1.2638, | |
| "step": 93 | |
| }, | |
| { | |
| "epoch": 0.3230240549828179, | |
| "grad_norm": 1.2844491004943848, | |
| "learning_rate": 5.90440267166055e-05, | |
| "loss": 1.224, | |
| "step": 94 | |
| }, | |
| { | |
| "epoch": 0.32646048109965636, | |
| "grad_norm": 1.21943199634552, | |
| "learning_rate": 5.8229729514036705e-05, | |
| "loss": 1.0463, | |
| "step": 95 | |
| }, | |
| { | |
| "epoch": 0.32989690721649484, | |
| "grad_norm": 1.3590772151947021, | |
| "learning_rate": 5.74131823855921e-05, | |
| "loss": 1.0311, | |
| "step": 96 | |
| }, | |
| { | |
| "epoch": 0.3333333333333333, | |
| "grad_norm": 1.2511202096939087, | |
| "learning_rate": 5.6594608567103456e-05, | |
| "loss": 0.8306, | |
| "step": 97 | |
| }, | |
| { | |
| "epoch": 0.33676975945017185, | |
| "grad_norm": 1.4306681156158447, | |
| "learning_rate": 5.577423184847932e-05, | |
| "loss": 0.9824, | |
| "step": 98 | |
| }, | |
| { | |
| "epoch": 0.3402061855670103, | |
| "grad_norm": 1.5601223707199097, | |
| "learning_rate": 5.495227651252315e-05, | |
| "loss": 1.106, | |
| "step": 99 | |
| }, | |
| { | |
| "epoch": 0.3436426116838488, | |
| "grad_norm": 1.7903016805648804, | |
| "learning_rate": 5.4128967273616625e-05, | |
| "loss": 1.1582, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.3436426116838488, | |
| "eval_loss": 1.5493346452713013, | |
| "eval_runtime": 42.4479, | |
| "eval_samples_per_second": 11.544, | |
| "eval_steps_per_second": 2.898, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.3470790378006873, | |
| "grad_norm": 0.6229208111763, | |
| "learning_rate": 5.330452921628497e-05, | |
| "loss": 1.8814, | |
| "step": 101 | |
| }, | |
| { | |
| "epoch": 0.35051546391752575, | |
| "grad_norm": 0.6040644645690918, | |
| "learning_rate": 5.247918773366112e-05, | |
| "loss": 1.9904, | |
| "step": 102 | |
| }, | |
| { | |
| "epoch": 0.3539518900343643, | |
| "grad_norm": 0.541728138923645, | |
| "learning_rate": 5.165316846586541e-05, | |
| "loss": 1.9275, | |
| "step": 103 | |
| }, | |
| { | |
| "epoch": 0.35738831615120276, | |
| "grad_norm": 0.4601477086544037, | |
| "learning_rate": 5.0826697238317935e-05, | |
| "loss": 1.906, | |
| "step": 104 | |
| }, | |
| { | |
| "epoch": 0.36082474226804123, | |
| "grad_norm": 0.37705910205841064, | |
| "learning_rate": 5e-05, | |
| "loss": 1.9962, | |
| "step": 105 | |
| }, | |
| { | |
| "epoch": 0.3642611683848797, | |
| "grad_norm": 0.4009793698787689, | |
| "learning_rate": 4.917330276168208e-05, | |
| "loss": 1.9482, | |
| "step": 106 | |
| }, | |
| { | |
| "epoch": 0.36769759450171824, | |
| "grad_norm": 0.39194822311401367, | |
| "learning_rate": 4.834683153413459e-05, | |
| "loss": 1.7998, | |
| "step": 107 | |
| }, | |
| { | |
| "epoch": 0.3711340206185567, | |
| "grad_norm": 0.43941518664360046, | |
| "learning_rate": 4.7520812266338885e-05, | |
| "loss": 1.7441, | |
| "step": 108 | |
| }, | |
| { | |
| "epoch": 0.3745704467353952, | |
| "grad_norm": 0.42722004652023315, | |
| "learning_rate": 4.669547078371504e-05, | |
| "loss": 1.795, | |
| "step": 109 | |
| }, | |
| { | |
| "epoch": 0.37800687285223367, | |
| "grad_norm": 0.4189625382423401, | |
| "learning_rate": 4.5871032726383386e-05, | |
| "loss": 1.7932, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.38144329896907214, | |
| "grad_norm": 0.45230814814567566, | |
| "learning_rate": 4.504772348747687e-05, | |
| "loss": 1.7466, | |
| "step": 111 | |
| }, | |
| { | |
| "epoch": 0.3848797250859107, | |
| "grad_norm": 0.44370195269584656, | |
| "learning_rate": 4.4225768151520694e-05, | |
| "loss": 1.8481, | |
| "step": 112 | |
| }, | |
| { | |
| "epoch": 0.38831615120274915, | |
| "grad_norm": 0.47599515318870544, | |
| "learning_rate": 4.3405391432896555e-05, | |
| "loss": 1.6495, | |
| "step": 113 | |
| }, | |
| { | |
| "epoch": 0.3917525773195876, | |
| "grad_norm": 0.5135727524757385, | |
| "learning_rate": 4.2586817614407895e-05, | |
| "loss": 1.7011, | |
| "step": 114 | |
| }, | |
| { | |
| "epoch": 0.3951890034364261, | |
| "grad_norm": 0.4951974153518677, | |
| "learning_rate": 4.17702704859633e-05, | |
| "loss": 1.6805, | |
| "step": 115 | |
| }, | |
| { | |
| "epoch": 0.39862542955326463, | |
| "grad_norm": 0.540644109249115, | |
| "learning_rate": 4.095597328339452e-05, | |
| "loss": 1.6576, | |
| "step": 116 | |
| }, | |
| { | |
| "epoch": 0.4020618556701031, | |
| "grad_norm": 0.5553061962127686, | |
| "learning_rate": 4.0144148627425993e-05, | |
| "loss": 1.7229, | |
| "step": 117 | |
| }, | |
| { | |
| "epoch": 0.4054982817869416, | |
| "grad_norm": 0.5354465842247009, | |
| "learning_rate": 3.933501846281267e-05, | |
| "loss": 1.5919, | |
| "step": 118 | |
| }, | |
| { | |
| "epoch": 0.40893470790378006, | |
| "grad_norm": 0.5654662251472473, | |
| "learning_rate": 3.852880399766243e-05, | |
| "loss": 1.6166, | |
| "step": 119 | |
| }, | |
| { | |
| "epoch": 0.41237113402061853, | |
| "grad_norm": 0.5637564063072205, | |
| "learning_rate": 3.772572564296005e-05, | |
| "loss": 1.4867, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.41580756013745707, | |
| "grad_norm": 0.5877327919006348, | |
| "learning_rate": 3.6926002952309016e-05, | |
| "loss": 1.5709, | |
| "step": 121 | |
| }, | |
| { | |
| "epoch": 0.41924398625429554, | |
| "grad_norm": 0.563671886920929, | |
| "learning_rate": 3.612985456190778e-05, | |
| "loss": 1.442, | |
| "step": 122 | |
| }, | |
| { | |
| "epoch": 0.422680412371134, | |
| "grad_norm": 0.6311579346656799, | |
| "learning_rate": 3.533749813077677e-05, | |
| "loss": 1.5916, | |
| "step": 123 | |
| }, | |
| { | |
| "epoch": 0.4261168384879725, | |
| "grad_norm": 0.5859690308570862, | |
| "learning_rate": 3.4549150281252636e-05, | |
| "loss": 1.5786, | |
| "step": 124 | |
| }, | |
| { | |
| "epoch": 0.42955326460481097, | |
| "grad_norm": 0.6477120518684387, | |
| "learning_rate": 3.3765026539765834e-05, | |
| "loss": 1.5158, | |
| "step": 125 | |
| }, | |
| { | |
| "epoch": 0.4329896907216495, | |
| "grad_norm": 0.6799119114875793, | |
| "learning_rate": 3.298534127791785e-05, | |
| "loss": 1.5917, | |
| "step": 126 | |
| }, | |
| { | |
| "epoch": 0.436426116838488, | |
| "grad_norm": 0.6930612325668335, | |
| "learning_rate": 3.221030765387417e-05, | |
| "loss": 1.5057, | |
| "step": 127 | |
| }, | |
| { | |
| "epoch": 0.43986254295532645, | |
| "grad_norm": 0.7378568649291992, | |
| "learning_rate": 3.144013755408895e-05, | |
| "loss": 1.5408, | |
| "step": 128 | |
| }, | |
| { | |
| "epoch": 0.44329896907216493, | |
| "grad_norm": 0.7351049184799194, | |
| "learning_rate": 3.0675041535377405e-05, | |
| "loss": 1.3803, | |
| "step": 129 | |
| }, | |
| { | |
| "epoch": 0.44673539518900346, | |
| "grad_norm": 0.6970375776290894, | |
| "learning_rate": 2.991522876735154e-05, | |
| "loss": 1.5108, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.45017182130584193, | |
| "grad_norm": 0.7849711179733276, | |
| "learning_rate": 2.916090697523549e-05, | |
| "loss": 1.5658, | |
| "step": 131 | |
| }, | |
| { | |
| "epoch": 0.4536082474226804, | |
| "grad_norm": 0.8675287365913391, | |
| "learning_rate": 2.8412282383075363e-05, | |
| "loss": 1.4551, | |
| "step": 132 | |
| }, | |
| { | |
| "epoch": 0.4570446735395189, | |
| "grad_norm": 0.8072260022163391, | |
| "learning_rate": 2.766955965735968e-05, | |
| "loss": 1.3688, | |
| "step": 133 | |
| }, | |
| { | |
| "epoch": 0.46048109965635736, | |
| "grad_norm": 0.8894752860069275, | |
| "learning_rate": 2.693294185106562e-05, | |
| "loss": 1.3995, | |
| "step": 134 | |
| }, | |
| { | |
| "epoch": 0.4639175257731959, | |
| "grad_norm": 0.8725594282150269, | |
| "learning_rate": 2.6202630348146324e-05, | |
| "loss": 1.0523, | |
| "step": 135 | |
| }, | |
| { | |
| "epoch": 0.46735395189003437, | |
| "grad_norm": 0.9702212810516357, | |
| "learning_rate": 2.547882480847461e-05, | |
| "loss": 1.2428, | |
| "step": 136 | |
| }, | |
| { | |
| "epoch": 0.47079037800687284, | |
| "grad_norm": 0.9078148007392883, | |
| "learning_rate": 2.476172311325783e-05, | |
| "loss": 1.3728, | |
| "step": 137 | |
| }, | |
| { | |
| "epoch": 0.4742268041237113, | |
| "grad_norm": 0.9433721303939819, | |
| "learning_rate": 2.405152131093926e-05, | |
| "loss": 1.3099, | |
| "step": 138 | |
| }, | |
| { | |
| "epoch": 0.47766323024054985, | |
| "grad_norm": 1.0229471921920776, | |
| "learning_rate": 2.3348413563600325e-05, | |
| "loss": 1.3666, | |
| "step": 139 | |
| }, | |
| { | |
| "epoch": 0.48109965635738833, | |
| "grad_norm": 0.9528070688247681, | |
| "learning_rate": 2.2652592093878666e-05, | |
| "loss": 1.1095, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.4845360824742268, | |
| "grad_norm": 1.3753701448440552, | |
| "learning_rate": 2.196424713241637e-05, | |
| "loss": 0.9742, | |
| "step": 141 | |
| }, | |
| { | |
| "epoch": 0.4879725085910653, | |
| "grad_norm": 1.0363105535507202, | |
| "learning_rate": 2.128356686585282e-05, | |
| "loss": 0.9841, | |
| "step": 142 | |
| }, | |
| { | |
| "epoch": 0.49140893470790376, | |
| "grad_norm": 1.1423755884170532, | |
| "learning_rate": 2.061073738537635e-05, | |
| "loss": 1.1428, | |
| "step": 143 | |
| }, | |
| { | |
| "epoch": 0.4948453608247423, | |
| "grad_norm": 1.2904118299484253, | |
| "learning_rate": 1.9945942635848748e-05, | |
| "loss": 1.2221, | |
| "step": 144 | |
| }, | |
| { | |
| "epoch": 0.49828178694158076, | |
| "grad_norm": 1.2646178007125854, | |
| "learning_rate": 1.928936436551661e-05, | |
| "loss": 1.0977, | |
| "step": 145 | |
| }, | |
| { | |
| "epoch": 0.5017182130584192, | |
| "grad_norm": 1.158040165901184, | |
| "learning_rate": 1.8641182076323148e-05, | |
| "loss": 0.9034, | |
| "step": 146 | |
| }, | |
| { | |
| "epoch": 0.5051546391752577, | |
| "grad_norm": 1.3848685026168823, | |
| "learning_rate": 1.800157297483417e-05, | |
| "loss": 0.9816, | |
| "step": 147 | |
| }, | |
| { | |
| "epoch": 0.5085910652920962, | |
| "grad_norm": 1.431872844696045, | |
| "learning_rate": 1.7370711923791567e-05, | |
| "loss": 1.0277, | |
| "step": 148 | |
| }, | |
| { | |
| "epoch": 0.5120274914089347, | |
| "grad_norm": 1.4328848123550415, | |
| "learning_rate": 1.6748771394307585e-05, | |
| "loss": 0.9525, | |
| "step": 149 | |
| }, | |
| { | |
| "epoch": 0.5154639175257731, | |
| "grad_norm": 1.7907229661941528, | |
| "learning_rate": 1.6135921418712956e-05, | |
| "loss": 1.2614, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.5154639175257731, | |
| "eval_loss": 1.4730490446090698, | |
| "eval_runtime": 42.6646, | |
| "eval_samples_per_second": 11.485, | |
| "eval_steps_per_second": 2.883, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.5189003436426117, | |
| "grad_norm": 0.367071270942688, | |
| "learning_rate": 1.553232954407171e-05, | |
| "loss": 1.828, | |
| "step": 151 | |
| }, | |
| { | |
| "epoch": 0.5223367697594502, | |
| "grad_norm": 0.39169952273368835, | |
| "learning_rate": 1.4938160786375572e-05, | |
| "loss": 1.8921, | |
| "step": 152 | |
| }, | |
| { | |
| "epoch": 0.5257731958762887, | |
| "grad_norm": 0.3786236345767975, | |
| "learning_rate": 1.435357758543015e-05, | |
| "loss": 1.8307, | |
| "step": 153 | |
| }, | |
| { | |
| "epoch": 0.5292096219931272, | |
| "grad_norm": 0.38766780495643616, | |
| "learning_rate": 1.3778739760445552e-05, | |
| "loss": 1.9383, | |
| "step": 154 | |
| }, | |
| { | |
| "epoch": 0.5326460481099656, | |
| "grad_norm": 0.36101722717285156, | |
| "learning_rate": 1.3213804466343421e-05, | |
| "loss": 1.8938, | |
| "step": 155 | |
| }, | |
| { | |
| "epoch": 0.5360824742268041, | |
| "grad_norm": 0.38169556856155396, | |
| "learning_rate": 1.2658926150792322e-05, | |
| "loss": 1.8638, | |
| "step": 156 | |
| }, | |
| { | |
| "epoch": 0.5395189003436426, | |
| "grad_norm": 0.38779017329216003, | |
| "learning_rate": 1.2114256511983274e-05, | |
| "loss": 1.8177, | |
| "step": 157 | |
| }, | |
| { | |
| "epoch": 0.5429553264604811, | |
| "grad_norm": 0.3910422623157501, | |
| "learning_rate": 1.157994445715706e-05, | |
| "loss": 1.6613, | |
| "step": 158 | |
| }, | |
| { | |
| "epoch": 0.5463917525773195, | |
| "grad_norm": 0.4210437834262848, | |
| "learning_rate": 1.1056136061894384e-05, | |
| "loss": 1.8767, | |
| "step": 159 | |
| }, | |
| { | |
| "epoch": 0.5498281786941581, | |
| "grad_norm": 0.43400484323501587, | |
| "learning_rate": 1.0542974530180327e-05, | |
| "loss": 1.7092, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.5532646048109966, | |
| "grad_norm": 0.4364904463291168, | |
| "learning_rate": 1.0040600155253765e-05, | |
| "loss": 1.5604, | |
| "step": 161 | |
| }, | |
| { | |
| "epoch": 0.5567010309278351, | |
| "grad_norm": 0.4587176442146301, | |
| "learning_rate": 9.549150281252633e-06, | |
| "loss": 1.531, | |
| "step": 162 | |
| }, | |
| { | |
| "epoch": 0.5601374570446735, | |
| "grad_norm": 0.4823138415813446, | |
| "learning_rate": 9.068759265665384e-06, | |
| "loss": 1.695, | |
| "step": 163 | |
| }, | |
| { | |
| "epoch": 0.563573883161512, | |
| "grad_norm": 0.48356637358665466, | |
| "learning_rate": 8.599558442598998e-06, | |
| "loss": 1.6705, | |
| "step": 164 | |
| }, | |
| { | |
| "epoch": 0.5670103092783505, | |
| "grad_norm": 0.5010591745376587, | |
| "learning_rate": 8.141676086873572e-06, | |
| "loss": 1.6112, | |
| "step": 165 | |
| }, | |
| { | |
| "epoch": 0.570446735395189, | |
| "grad_norm": 0.5374965071678162, | |
| "learning_rate": 7.695237378953223e-06, | |
| "loss": 1.645, | |
| "step": 166 | |
| }, | |
| { | |
| "epoch": 0.5738831615120275, | |
| "grad_norm": 0.5316302180290222, | |
| "learning_rate": 7.260364370723044e-06, | |
| "loss": 1.5194, | |
| "step": 167 | |
| }, | |
| { | |
| "epoch": 0.5773195876288659, | |
| "grad_norm": 0.552695095539093, | |
| "learning_rate": 6.837175952121306e-06, | |
| "loss": 1.5257, | |
| "step": 168 | |
| }, | |
| { | |
| "epoch": 0.5807560137457045, | |
| "grad_norm": 0.5447157025337219, | |
| "learning_rate": 6.425787818636131e-06, | |
| "loss": 1.503, | |
| "step": 169 | |
| }, | |
| { | |
| "epoch": 0.584192439862543, | |
| "grad_norm": 0.5910674333572388, | |
| "learning_rate": 6.026312439675552e-06, | |
| "loss": 1.4441, | |
| "step": 170 | |
| }, | |
| { | |
| "epoch": 0.5876288659793815, | |
| "grad_norm": 0.5824146270751953, | |
| "learning_rate": 5.6388590278194096e-06, | |
| "loss": 1.525, | |
| "step": 171 | |
| }, | |
| { | |
| "epoch": 0.5910652920962199, | |
| "grad_norm": 0.5922824740409851, | |
| "learning_rate": 5.263533508961827e-06, | |
| "loss": 1.464, | |
| "step": 172 | |
| }, | |
| { | |
| "epoch": 0.5945017182130584, | |
| "grad_norm": 0.6297634243965149, | |
| "learning_rate": 4.900438493352055e-06, | |
| "loss": 1.5201, | |
| "step": 173 | |
| }, | |
| { | |
| "epoch": 0.5979381443298969, | |
| "grad_norm": 0.6708535552024841, | |
| "learning_rate": 4.549673247541875e-06, | |
| "loss": 1.4681, | |
| "step": 174 | |
| }, | |
| { | |
| "epoch": 0.6013745704467354, | |
| "grad_norm": 0.7168099880218506, | |
| "learning_rate": 4.2113336672471245e-06, | |
| "loss": 1.6194, | |
| "step": 175 | |
| }, | |
| { | |
| "epoch": 0.6048109965635738, | |
| "grad_norm": 0.7247710227966309, | |
| "learning_rate": 3.885512251130763e-06, | |
| "loss": 1.5599, | |
| "step": 176 | |
| }, | |
| { | |
| "epoch": 0.6082474226804123, | |
| "grad_norm": 0.7502146363258362, | |
| "learning_rate": 3.5722980755146517e-06, | |
| "loss": 1.4339, | |
| "step": 177 | |
| }, | |
| { | |
| "epoch": 0.6116838487972509, | |
| "grad_norm": 0.6751747131347656, | |
| "learning_rate": 3.271776770026963e-06, | |
| "loss": 1.2429, | |
| "step": 178 | |
| }, | |
| { | |
| "epoch": 0.6151202749140894, | |
| "grad_norm": 0.7085171341896057, | |
| "learning_rate": 2.9840304941919415e-06, | |
| "loss": 1.3779, | |
| "step": 179 | |
| }, | |
| { | |
| "epoch": 0.6185567010309279, | |
| "grad_norm": 0.7121344208717346, | |
| "learning_rate": 2.7091379149682685e-06, | |
| "loss": 1.2872, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.6219931271477663, | |
| "grad_norm": 0.7500658631324768, | |
| "learning_rate": 2.4471741852423237e-06, | |
| "loss": 1.4628, | |
| "step": 181 | |
| }, | |
| { | |
| "epoch": 0.6254295532646048, | |
| "grad_norm": 0.7546201348304749, | |
| "learning_rate": 2.1982109232821178e-06, | |
| "loss": 1.4433, | |
| "step": 182 | |
| }, | |
| { | |
| "epoch": 0.6288659793814433, | |
| "grad_norm": 0.8175408244132996, | |
| "learning_rate": 1.962316193157593e-06, | |
| "loss": 1.4879, | |
| "step": 183 | |
| }, | |
| { | |
| "epoch": 0.6323024054982818, | |
| "grad_norm": 0.7544344067573547, | |
| "learning_rate": 1.7395544861325718e-06, | |
| "loss": 1.2931, | |
| "step": 184 | |
| }, | |
| { | |
| "epoch": 0.6357388316151202, | |
| "grad_norm": 0.796675443649292, | |
| "learning_rate": 1.5299867030334814e-06, | |
| "loss": 1.1412, | |
| "step": 185 | |
| }, | |
| { | |
| "epoch": 0.6391752577319587, | |
| "grad_norm": 0.8795393109321594, | |
| "learning_rate": 1.333670137599713e-06, | |
| "loss": 1.1944, | |
| "step": 186 | |
| }, | |
| { | |
| "epoch": 0.6426116838487973, | |
| "grad_norm": 0.9023170471191406, | |
| "learning_rate": 1.1506584608200367e-06, | |
| "loss": 1.2908, | |
| "step": 187 | |
| }, | |
| { | |
| "epoch": 0.6460481099656358, | |
| "grad_norm": 0.9260828495025635, | |
| "learning_rate": 9.810017062595322e-07, | |
| "loss": 1.1943, | |
| "step": 188 | |
| }, | |
| { | |
| "epoch": 0.6494845360824743, | |
| "grad_norm": 0.9484823942184448, | |
| "learning_rate": 8.247462563808817e-07, | |
| "loss": 1.0709, | |
| "step": 189 | |
| }, | |
| { | |
| "epoch": 0.6529209621993127, | |
| "grad_norm": 0.9990953803062439, | |
| "learning_rate": 6.819348298638839e-07, | |
| "loss": 1.2035, | |
| "step": 190 | |
| }, | |
| { | |
| "epoch": 0.6563573883161512, | |
| "grad_norm": 1.1566455364227295, | |
| "learning_rate": 5.526064699265753e-07, | |
| "loss": 1.2767, | |
| "step": 191 | |
| }, | |
| { | |
| "epoch": 0.6597938144329897, | |
| "grad_norm": 1.4236087799072266, | |
| "learning_rate": 4.367965336512403e-07, | |
| "loss": 1.1267, | |
| "step": 192 | |
| }, | |
| { | |
| "epoch": 0.6632302405498282, | |
| "grad_norm": 1.0721385478973389, | |
| "learning_rate": 3.3453668231809286e-07, | |
| "loss": 0.963, | |
| "step": 193 | |
| }, | |
| { | |
| "epoch": 0.6666666666666666, | |
| "grad_norm": 1.0774070024490356, | |
| "learning_rate": 2.458548727494292e-07, | |
| "loss": 0.9541, | |
| "step": 194 | |
| }, | |
| { | |
| "epoch": 0.6701030927835051, | |
| "grad_norm": 1.155906081199646, | |
| "learning_rate": 1.7077534966650766e-07, | |
| "loss": 1.0398, | |
| "step": 195 | |
| }, | |
| { | |
| "epoch": 0.6735395189003437, | |
| "grad_norm": 1.3076586723327637, | |
| "learning_rate": 1.0931863906127327e-07, | |
| "loss": 0.9897, | |
| "step": 196 | |
| }, | |
| { | |
| "epoch": 0.6769759450171822, | |
| "grad_norm": 1.3334431648254395, | |
| "learning_rate": 6.150154258476315e-08, | |
| "loss": 0.8965, | |
| "step": 197 | |
| }, | |
| { | |
| "epoch": 0.6804123711340206, | |
| "grad_norm": 1.542555332183838, | |
| "learning_rate": 2.7337132953697554e-08, | |
| "loss": 0.9844, | |
| "step": 198 | |
| }, | |
| { | |
| "epoch": 0.6838487972508591, | |
| "grad_norm": 1.4503092765808105, | |
| "learning_rate": 6.834750376549792e-09, | |
| "loss": 1.0682, | |
| "step": 199 | |
| }, | |
| { | |
| "epoch": 0.6872852233676976, | |
| "grad_norm": 1.9322080612182617, | |
| "learning_rate": 0.0, | |
| "loss": 1.3643, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.6872852233676976, | |
| "eval_loss": 1.4643950462341309, | |
| "eval_runtime": 42.6799, | |
| "eval_samples_per_second": 11.481, | |
| "eval_steps_per_second": 2.882, | |
| "step": 200 | |
| } | |
| ], | |
| "logging_steps": 1, | |
| "max_steps": 200, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 1, | |
| "save_steps": 50, | |
| "stateful_callbacks": { | |
| "EarlyStoppingCallback": { | |
| "args": { | |
| "early_stopping_patience": 5, | |
| "early_stopping_threshold": 0.0 | |
| }, | |
| "attributes": { | |
| "early_stopping_patience_counter": 0 | |
| } | |
| }, | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 3.40546940401877e+17, | |
| "train_batch_size": 8, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |