Instructions to use aleegis10/8123eb65-93c2-4058-b1db-99432f2d21d8 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use aleegis10/8123eb65-93c2-4058-b1db-99432f2d21d8 with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("unsloth/Hermes-3-Llama-3.1-8B") model = PeftModel.from_pretrained(base_model, "aleegis10/8123eb65-93c2-4058-b1db-99432f2d21d8") - Notebooks
- Google Colab
- Kaggle
| { | |
| "best_metric": 0.9496371746063232, | |
| "best_model_checkpoint": "miner_id_24/checkpoint-200", | |
| "epoch": 0.9205983889528193, | |
| "eval_steps": 50, | |
| "global_step": 200, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.004602991944764097, | |
| "grad_norm": 1.3569234609603882, | |
| "learning_rate": 1e-05, | |
| "loss": 1.3583, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.004602991944764097, | |
| "eval_loss": 1.4002622365951538, | |
| "eval_runtime": 27.5381, | |
| "eval_samples_per_second": 13.291, | |
| "eval_steps_per_second": 3.341, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.009205983889528193, | |
| "grad_norm": 1.361998438835144, | |
| "learning_rate": 2e-05, | |
| "loss": 1.2473, | |
| "step": 2 | |
| }, | |
| { | |
| "epoch": 0.01380897583429229, | |
| "grad_norm": 1.2841529846191406, | |
| "learning_rate": 3e-05, | |
| "loss": 1.2645, | |
| "step": 3 | |
| }, | |
| { | |
| "epoch": 0.018411967779056387, | |
| "grad_norm": 1.3013325929641724, | |
| "learning_rate": 4e-05, | |
| "loss": 1.2937, | |
| "step": 4 | |
| }, | |
| { | |
| "epoch": 0.023014959723820484, | |
| "grad_norm": 0.9889752864837646, | |
| "learning_rate": 5e-05, | |
| "loss": 1.1661, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.02761795166858458, | |
| "grad_norm": 0.776945173740387, | |
| "learning_rate": 6e-05, | |
| "loss": 1.1802, | |
| "step": 6 | |
| }, | |
| { | |
| "epoch": 0.03222094361334868, | |
| "grad_norm": 0.6851246356964111, | |
| "learning_rate": 7e-05, | |
| "loss": 1.164, | |
| "step": 7 | |
| }, | |
| { | |
| "epoch": 0.03682393555811277, | |
| "grad_norm": 0.7995606064796448, | |
| "learning_rate": 8e-05, | |
| "loss": 1.2199, | |
| "step": 8 | |
| }, | |
| { | |
| "epoch": 0.04142692750287687, | |
| "grad_norm": 0.9208403825759888, | |
| "learning_rate": 9e-05, | |
| "loss": 1.1835, | |
| "step": 9 | |
| }, | |
| { | |
| "epoch": 0.04602991944764097, | |
| "grad_norm": 0.6956402659416199, | |
| "learning_rate": 0.0001, | |
| "loss": 1.2395, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.05063291139240506, | |
| "grad_norm": 0.5254325866699219, | |
| "learning_rate": 9.999316524962345e-05, | |
| "loss": 1.1411, | |
| "step": 11 | |
| }, | |
| { | |
| "epoch": 0.05523590333716916, | |
| "grad_norm": 0.5878555774688721, | |
| "learning_rate": 9.997266286704631e-05, | |
| "loss": 1.1407, | |
| "step": 12 | |
| }, | |
| { | |
| "epoch": 0.05983889528193326, | |
| "grad_norm": 0.6225937008857727, | |
| "learning_rate": 9.993849845741524e-05, | |
| "loss": 1.1928, | |
| "step": 13 | |
| }, | |
| { | |
| "epoch": 0.06444188722669736, | |
| "grad_norm": 0.5445913672447205, | |
| "learning_rate": 9.989068136093873e-05, | |
| "loss": 1.0413, | |
| "step": 14 | |
| }, | |
| { | |
| "epoch": 0.06904487917146145, | |
| "grad_norm": 0.5479422807693481, | |
| "learning_rate": 9.98292246503335e-05, | |
| "loss": 1.0938, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.07364787111622555, | |
| "grad_norm": 0.5646893978118896, | |
| "learning_rate": 9.975414512725057e-05, | |
| "loss": 1.0837, | |
| "step": 16 | |
| }, | |
| { | |
| "epoch": 0.07825086306098965, | |
| "grad_norm": 0.5424992442131042, | |
| "learning_rate": 9.966546331768191e-05, | |
| "loss": 1.1106, | |
| "step": 17 | |
| }, | |
| { | |
| "epoch": 0.08285385500575373, | |
| "grad_norm": 0.5262547135353088, | |
| "learning_rate": 9.956320346634876e-05, | |
| "loss": 1.0843, | |
| "step": 18 | |
| }, | |
| { | |
| "epoch": 0.08745684695051784, | |
| "grad_norm": 0.554241955280304, | |
| "learning_rate": 9.944739353007344e-05, | |
| "loss": 1.174, | |
| "step": 19 | |
| }, | |
| { | |
| "epoch": 0.09205983889528194, | |
| "grad_norm": 0.5389442443847656, | |
| "learning_rate": 9.931806517013612e-05, | |
| "loss": 1.0609, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.09666283084004602, | |
| "grad_norm": 0.5209745764732361, | |
| "learning_rate": 9.917525374361912e-05, | |
| "loss": 1.0349, | |
| "step": 21 | |
| }, | |
| { | |
| "epoch": 0.10126582278481013, | |
| "grad_norm": 0.5692411065101624, | |
| "learning_rate": 9.901899829374047e-05, | |
| "loss": 1.0565, | |
| "step": 22 | |
| }, | |
| { | |
| "epoch": 0.10586881472957423, | |
| "grad_norm": 0.5614836812019348, | |
| "learning_rate": 9.884934153917997e-05, | |
| "loss": 1.0753, | |
| "step": 23 | |
| }, | |
| { | |
| "epoch": 0.11047180667433831, | |
| "grad_norm": 0.5814270973205566, | |
| "learning_rate": 9.86663298624003e-05, | |
| "loss": 1.1013, | |
| "step": 24 | |
| }, | |
| { | |
| "epoch": 0.11507479861910241, | |
| "grad_norm": 0.5987984538078308, | |
| "learning_rate": 9.847001329696653e-05, | |
| "loss": 1.0675, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 0.11967779056386652, | |
| "grad_norm": 0.5689542293548584, | |
| "learning_rate": 9.826044551386744e-05, | |
| "loss": 1.1024, | |
| "step": 26 | |
| }, | |
| { | |
| "epoch": 0.12428078250863062, | |
| "grad_norm": 0.6012589931488037, | |
| "learning_rate": 9.803768380684242e-05, | |
| "loss": 1.076, | |
| "step": 27 | |
| }, | |
| { | |
| "epoch": 0.12888377445339472, | |
| "grad_norm": 0.5973519682884216, | |
| "learning_rate": 9.780178907671789e-05, | |
| "loss": 1.1807, | |
| "step": 28 | |
| }, | |
| { | |
| "epoch": 0.1334867663981588, | |
| "grad_norm": 0.6472519040107727, | |
| "learning_rate": 9.755282581475769e-05, | |
| "loss": 1.0086, | |
| "step": 29 | |
| }, | |
| { | |
| "epoch": 0.1380897583429229, | |
| "grad_norm": 0.6917528510093689, | |
| "learning_rate": 9.729086208503174e-05, | |
| "loss": 1.0473, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.142692750287687, | |
| "grad_norm": 0.6226741671562195, | |
| "learning_rate": 9.701596950580806e-05, | |
| "loss": 1.0797, | |
| "step": 31 | |
| }, | |
| { | |
| "epoch": 0.1472957422324511, | |
| "grad_norm": 0.6242543458938599, | |
| "learning_rate": 9.672822322997305e-05, | |
| "loss": 1.0174, | |
| "step": 32 | |
| }, | |
| { | |
| "epoch": 0.1518987341772152, | |
| "grad_norm": 0.678283154964447, | |
| "learning_rate": 9.642770192448536e-05, | |
| "loss": 0.9007, | |
| "step": 33 | |
| }, | |
| { | |
| "epoch": 0.1565017261219793, | |
| "grad_norm": 0.6283822655677795, | |
| "learning_rate": 9.611448774886924e-05, | |
| "loss": 1.0111, | |
| "step": 34 | |
| }, | |
| { | |
| "epoch": 0.1611047180667434, | |
| "grad_norm": 0.6623278856277466, | |
| "learning_rate": 9.578866633275288e-05, | |
| "loss": 1.0385, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 0.16570771001150747, | |
| "grad_norm": 0.6641826033592224, | |
| "learning_rate": 9.545032675245813e-05, | |
| "loss": 0.9875, | |
| "step": 36 | |
| }, | |
| { | |
| "epoch": 0.17031070195627157, | |
| "grad_norm": 0.6380876302719116, | |
| "learning_rate": 9.509956150664796e-05, | |
| "loss": 0.9478, | |
| "step": 37 | |
| }, | |
| { | |
| "epoch": 0.17491369390103567, | |
| "grad_norm": 0.741593062877655, | |
| "learning_rate": 9.473646649103818e-05, | |
| "loss": 0.9362, | |
| "step": 38 | |
| }, | |
| { | |
| "epoch": 0.17951668584579977, | |
| "grad_norm": 0.7561161518096924, | |
| "learning_rate": 9.43611409721806e-05, | |
| "loss": 1.0234, | |
| "step": 39 | |
| }, | |
| { | |
| "epoch": 0.18411967779056387, | |
| "grad_norm": 0.8054270148277283, | |
| "learning_rate": 9.397368756032445e-05, | |
| "loss": 1.0038, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.18872266973532797, | |
| "grad_norm": 0.8184277415275574, | |
| "learning_rate": 9.357421218136386e-05, | |
| "loss": 0.9829, | |
| "step": 41 | |
| }, | |
| { | |
| "epoch": 0.19332566168009205, | |
| "grad_norm": 0.7387017607688904, | |
| "learning_rate": 9.316282404787871e-05, | |
| "loss": 0.9606, | |
| "step": 42 | |
| }, | |
| { | |
| "epoch": 0.19792865362485615, | |
| "grad_norm": 0.8113865852355957, | |
| "learning_rate": 9.273963562927695e-05, | |
| "loss": 0.9288, | |
| "step": 43 | |
| }, | |
| { | |
| "epoch": 0.20253164556962025, | |
| "grad_norm": 0.8790879845619202, | |
| "learning_rate": 9.230476262104677e-05, | |
| "loss": 1.0036, | |
| "step": 44 | |
| }, | |
| { | |
| "epoch": 0.20713463751438435, | |
| "grad_norm": 0.9341160655021667, | |
| "learning_rate": 9.185832391312644e-05, | |
| "loss": 0.8873, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 0.21173762945914845, | |
| "grad_norm": 0.8669713735580444, | |
| "learning_rate": 9.140044155740101e-05, | |
| "loss": 0.8748, | |
| "step": 46 | |
| }, | |
| { | |
| "epoch": 0.21634062140391255, | |
| "grad_norm": 1.1552430391311646, | |
| "learning_rate": 9.093124073433463e-05, | |
| "loss": 0.868, | |
| "step": 47 | |
| }, | |
| { | |
| "epoch": 0.22094361334867663, | |
| "grad_norm": 1.1203601360321045, | |
| "learning_rate": 9.045084971874738e-05, | |
| "loss": 0.9555, | |
| "step": 48 | |
| }, | |
| { | |
| "epoch": 0.22554660529344073, | |
| "grad_norm": 1.2116848230361938, | |
| "learning_rate": 8.995939984474624e-05, | |
| "loss": 0.9604, | |
| "step": 49 | |
| }, | |
| { | |
| "epoch": 0.23014959723820483, | |
| "grad_norm": 1.5171940326690674, | |
| "learning_rate": 8.945702546981969e-05, | |
| "loss": 0.9042, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.23014959723820483, | |
| "eval_loss": 1.170979619026184, | |
| "eval_runtime": 28.1277, | |
| "eval_samples_per_second": 13.012, | |
| "eval_steps_per_second": 3.271, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.23475258918296893, | |
| "grad_norm": 1.181593418121338, | |
| "learning_rate": 8.894386393810563e-05, | |
| "loss": 1.0942, | |
| "step": 51 | |
| }, | |
| { | |
| "epoch": 0.23935558112773303, | |
| "grad_norm": 0.8320015668869019, | |
| "learning_rate": 8.842005554284296e-05, | |
| "loss": 1.126, | |
| "step": 52 | |
| }, | |
| { | |
| "epoch": 0.24395857307249713, | |
| "grad_norm": 0.6058194637298584, | |
| "learning_rate": 8.788574348801675e-05, | |
| "loss": 1.1179, | |
| "step": 53 | |
| }, | |
| { | |
| "epoch": 0.24856156501726123, | |
| "grad_norm": 0.518334686756134, | |
| "learning_rate": 8.73410738492077e-05, | |
| "loss": 1.0694, | |
| "step": 54 | |
| }, | |
| { | |
| "epoch": 0.25316455696202533, | |
| "grad_norm": 0.4713587462902069, | |
| "learning_rate": 8.678619553365659e-05, | |
| "loss": 1.0558, | |
| "step": 55 | |
| }, | |
| { | |
| "epoch": 0.25776754890678943, | |
| "grad_norm": 0.44763702154159546, | |
| "learning_rate": 8.622126023955446e-05, | |
| "loss": 0.9986, | |
| "step": 56 | |
| }, | |
| { | |
| "epoch": 0.26237054085155354, | |
| "grad_norm": 0.45434433221817017, | |
| "learning_rate": 8.564642241456986e-05, | |
| "loss": 1.1099, | |
| "step": 57 | |
| }, | |
| { | |
| "epoch": 0.2669735327963176, | |
| "grad_norm": 0.41789236664772034, | |
| "learning_rate": 8.506183921362443e-05, | |
| "loss": 1.0249, | |
| "step": 58 | |
| }, | |
| { | |
| "epoch": 0.2715765247410817, | |
| "grad_norm": 0.44575098156929016, | |
| "learning_rate": 8.44676704559283e-05, | |
| "loss": 1.0808, | |
| "step": 59 | |
| }, | |
| { | |
| "epoch": 0.2761795166858458, | |
| "grad_norm": 0.4922280013561249, | |
| "learning_rate": 8.386407858128706e-05, | |
| "loss": 1.1243, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.2807825086306099, | |
| "grad_norm": 0.42270946502685547, | |
| "learning_rate": 8.32512286056924e-05, | |
| "loss": 1.1156, | |
| "step": 61 | |
| }, | |
| { | |
| "epoch": 0.285385500575374, | |
| "grad_norm": 0.4785097539424896, | |
| "learning_rate": 8.262928807620843e-05, | |
| "loss": 1.0415, | |
| "step": 62 | |
| }, | |
| { | |
| "epoch": 0.2899884925201381, | |
| "grad_norm": 0.4559645652770996, | |
| "learning_rate": 8.199842702516583e-05, | |
| "loss": 1.1056, | |
| "step": 63 | |
| }, | |
| { | |
| "epoch": 0.2945914844649022, | |
| "grad_norm": 0.4374312162399292, | |
| "learning_rate": 8.135881792367686e-05, | |
| "loss": 0.9779, | |
| "step": 64 | |
| }, | |
| { | |
| "epoch": 0.2991944764096663, | |
| "grad_norm": 0.436872273683548, | |
| "learning_rate": 8.07106356344834e-05, | |
| "loss": 1.0822, | |
| "step": 65 | |
| }, | |
| { | |
| "epoch": 0.3037974683544304, | |
| "grad_norm": 0.4464951455593109, | |
| "learning_rate": 8.005405736415126e-05, | |
| "loss": 1.0588, | |
| "step": 66 | |
| }, | |
| { | |
| "epoch": 0.3084004602991945, | |
| "grad_norm": 0.4651201069355011, | |
| "learning_rate": 7.938926261462366e-05, | |
| "loss": 1.0817, | |
| "step": 67 | |
| }, | |
| { | |
| "epoch": 0.3130034522439586, | |
| "grad_norm": 0.4480772316455841, | |
| "learning_rate": 7.871643313414718e-05, | |
| "loss": 1.0574, | |
| "step": 68 | |
| }, | |
| { | |
| "epoch": 0.3176064441887227, | |
| "grad_norm": 0.4776754081249237, | |
| "learning_rate": 7.803575286758364e-05, | |
| "loss": 0.9678, | |
| "step": 69 | |
| }, | |
| { | |
| "epoch": 0.3222094361334868, | |
| "grad_norm": 0.4526923596858978, | |
| "learning_rate": 7.734740790612136e-05, | |
| "loss": 1.0424, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.32681242807825084, | |
| "grad_norm": 0.5076996088027954, | |
| "learning_rate": 7.66515864363997e-05, | |
| "loss": 1.1003, | |
| "step": 71 | |
| }, | |
| { | |
| "epoch": 0.33141542002301494, | |
| "grad_norm": 0.5037042498588562, | |
| "learning_rate": 7.594847868906076e-05, | |
| "loss": 0.9783, | |
| "step": 72 | |
| }, | |
| { | |
| "epoch": 0.33601841196777904, | |
| "grad_norm": 0.5034875273704529, | |
| "learning_rate": 7.52382768867422e-05, | |
| "loss": 0.9902, | |
| "step": 73 | |
| }, | |
| { | |
| "epoch": 0.34062140391254314, | |
| "grad_norm": 0.5297345519065857, | |
| "learning_rate": 7.452117519152542e-05, | |
| "loss": 0.9925, | |
| "step": 74 | |
| }, | |
| { | |
| "epoch": 0.34522439585730724, | |
| "grad_norm": 0.6121540665626526, | |
| "learning_rate": 7.379736965185368e-05, | |
| "loss": 1.1559, | |
| "step": 75 | |
| }, | |
| { | |
| "epoch": 0.34982738780207134, | |
| "grad_norm": 0.5733057260513306, | |
| "learning_rate": 7.30670581489344e-05, | |
| "loss": 1.0383, | |
| "step": 76 | |
| }, | |
| { | |
| "epoch": 0.35443037974683544, | |
| "grad_norm": 0.5548041462898254, | |
| "learning_rate": 7.233044034264034e-05, | |
| "loss": 1.0647, | |
| "step": 77 | |
| }, | |
| { | |
| "epoch": 0.35903337169159955, | |
| "grad_norm": 0.671428382396698, | |
| "learning_rate": 7.158771761692464e-05, | |
| "loss": 1.0375, | |
| "step": 78 | |
| }, | |
| { | |
| "epoch": 0.36363636363636365, | |
| "grad_norm": 0.5743448138237, | |
| "learning_rate": 7.083909302476453e-05, | |
| "loss": 1.0086, | |
| "step": 79 | |
| }, | |
| { | |
| "epoch": 0.36823935558112775, | |
| "grad_norm": 0.580971896648407, | |
| "learning_rate": 7.008477123264848e-05, | |
| "loss": 1.0486, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.37284234752589185, | |
| "grad_norm": 0.5747451186180115, | |
| "learning_rate": 6.932495846462261e-05, | |
| "loss": 1.0183, | |
| "step": 81 | |
| }, | |
| { | |
| "epoch": 0.37744533947065595, | |
| "grad_norm": 0.5990350842475891, | |
| "learning_rate": 6.855986244591104e-05, | |
| "loss": 1.0555, | |
| "step": 82 | |
| }, | |
| { | |
| "epoch": 0.38204833141542005, | |
| "grad_norm": 0.5989863872528076, | |
| "learning_rate": 6.778969234612584e-05, | |
| "loss": 1.0201, | |
| "step": 83 | |
| }, | |
| { | |
| "epoch": 0.3866513233601841, | |
| "grad_norm": 0.613135814666748, | |
| "learning_rate": 6.701465872208216e-05, | |
| "loss": 0.9682, | |
| "step": 84 | |
| }, | |
| { | |
| "epoch": 0.3912543153049482, | |
| "grad_norm": 0.637292206287384, | |
| "learning_rate": 6.623497346023418e-05, | |
| "loss": 0.9811, | |
| "step": 85 | |
| }, | |
| { | |
| "epoch": 0.3958573072497123, | |
| "grad_norm": 0.660577654838562, | |
| "learning_rate": 6.545084971874738e-05, | |
| "loss": 0.938, | |
| "step": 86 | |
| }, | |
| { | |
| "epoch": 0.4004602991944764, | |
| "grad_norm": 0.6664721369743347, | |
| "learning_rate": 6.466250186922325e-05, | |
| "loss": 1.0055, | |
| "step": 87 | |
| }, | |
| { | |
| "epoch": 0.4050632911392405, | |
| "grad_norm": 0.6611858010292053, | |
| "learning_rate": 6.387014543809223e-05, | |
| "loss": 0.9033, | |
| "step": 88 | |
| }, | |
| { | |
| "epoch": 0.4096662830840046, | |
| "grad_norm": 0.7252963781356812, | |
| "learning_rate": 6.307399704769099e-05, | |
| "loss": 0.9384, | |
| "step": 89 | |
| }, | |
| { | |
| "epoch": 0.4142692750287687, | |
| "grad_norm": 0.7126460075378418, | |
| "learning_rate": 6.227427435703997e-05, | |
| "loss": 0.8955, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.4188722669735328, | |
| "grad_norm": 0.6885860562324524, | |
| "learning_rate": 6.147119600233758e-05, | |
| "loss": 0.8945, | |
| "step": 91 | |
| }, | |
| { | |
| "epoch": 0.4234752589182969, | |
| "grad_norm": 0.7343589067459106, | |
| "learning_rate": 6.066498153718735e-05, | |
| "loss": 0.9281, | |
| "step": 92 | |
| }, | |
| { | |
| "epoch": 0.428078250863061, | |
| "grad_norm": 0.7609535455703735, | |
| "learning_rate": 5.985585137257401e-05, | |
| "loss": 0.8477, | |
| "step": 93 | |
| }, | |
| { | |
| "epoch": 0.4326812428078251, | |
| "grad_norm": 0.8755032420158386, | |
| "learning_rate": 5.90440267166055e-05, | |
| "loss": 0.9429, | |
| "step": 94 | |
| }, | |
| { | |
| "epoch": 0.4372842347525892, | |
| "grad_norm": 0.8269835710525513, | |
| "learning_rate": 5.8229729514036705e-05, | |
| "loss": 0.9683, | |
| "step": 95 | |
| }, | |
| { | |
| "epoch": 0.44188722669735325, | |
| "grad_norm": 0.9273999333381653, | |
| "learning_rate": 5.74131823855921e-05, | |
| "loss": 0.8656, | |
| "step": 96 | |
| }, | |
| { | |
| "epoch": 0.44649021864211735, | |
| "grad_norm": 0.9501352310180664, | |
| "learning_rate": 5.6594608567103456e-05, | |
| "loss": 0.9235, | |
| "step": 97 | |
| }, | |
| { | |
| "epoch": 0.45109321058688145, | |
| "grad_norm": 0.9327521324157715, | |
| "learning_rate": 5.577423184847932e-05, | |
| "loss": 0.8879, | |
| "step": 98 | |
| }, | |
| { | |
| "epoch": 0.45569620253164556, | |
| "grad_norm": 0.9389417171478271, | |
| "learning_rate": 5.495227651252315e-05, | |
| "loss": 0.8453, | |
| "step": 99 | |
| }, | |
| { | |
| "epoch": 0.46029919447640966, | |
| "grad_norm": 1.43354332447052, | |
| "learning_rate": 5.4128967273616625e-05, | |
| "loss": 1.0444, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.46029919447640966, | |
| "eval_loss": 1.0572433471679688, | |
| "eval_runtime": 28.1467, | |
| "eval_samples_per_second": 13.003, | |
| "eval_steps_per_second": 3.269, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.46490218642117376, | |
| "grad_norm": 0.7535362839698792, | |
| "learning_rate": 5.330452921628497e-05, | |
| "loss": 0.9929, | |
| "step": 101 | |
| }, | |
| { | |
| "epoch": 0.46950517836593786, | |
| "grad_norm": 0.8638166189193726, | |
| "learning_rate": 5.247918773366112e-05, | |
| "loss": 1.1147, | |
| "step": 102 | |
| }, | |
| { | |
| "epoch": 0.47410817031070196, | |
| "grad_norm": 0.65118807554245, | |
| "learning_rate": 5.165316846586541e-05, | |
| "loss": 1.1173, | |
| "step": 103 | |
| }, | |
| { | |
| "epoch": 0.47871116225546606, | |
| "grad_norm": 0.4439648389816284, | |
| "learning_rate": 5.0826697238317935e-05, | |
| "loss": 0.97, | |
| "step": 104 | |
| }, | |
| { | |
| "epoch": 0.48331415420023016, | |
| "grad_norm": 0.36523792147636414, | |
| "learning_rate": 5e-05, | |
| "loss": 0.917, | |
| "step": 105 | |
| }, | |
| { | |
| "epoch": 0.48791714614499426, | |
| "grad_norm": 0.3680187463760376, | |
| "learning_rate": 4.917330276168208e-05, | |
| "loss": 1.0251, | |
| "step": 106 | |
| }, | |
| { | |
| "epoch": 0.49252013808975836, | |
| "grad_norm": 0.3534122705459595, | |
| "learning_rate": 4.834683153413459e-05, | |
| "loss": 0.9913, | |
| "step": 107 | |
| }, | |
| { | |
| "epoch": 0.49712313003452246, | |
| "grad_norm": 0.3764943778514862, | |
| "learning_rate": 4.7520812266338885e-05, | |
| "loss": 1.1063, | |
| "step": 108 | |
| }, | |
| { | |
| "epoch": 0.5017261219792866, | |
| "grad_norm": 0.37086910009384155, | |
| "learning_rate": 4.669547078371504e-05, | |
| "loss": 0.9456, | |
| "step": 109 | |
| }, | |
| { | |
| "epoch": 0.5063291139240507, | |
| "grad_norm": 0.3929706811904907, | |
| "learning_rate": 4.5871032726383386e-05, | |
| "loss": 0.9939, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.5109321058688148, | |
| "grad_norm": 0.3963087499141693, | |
| "learning_rate": 4.504772348747687e-05, | |
| "loss": 0.9335, | |
| "step": 111 | |
| }, | |
| { | |
| "epoch": 0.5155350978135789, | |
| "grad_norm": 0.3965536952018738, | |
| "learning_rate": 4.4225768151520694e-05, | |
| "loss": 1.0609, | |
| "step": 112 | |
| }, | |
| { | |
| "epoch": 0.520138089758343, | |
| "grad_norm": 0.4234572649002075, | |
| "learning_rate": 4.3405391432896555e-05, | |
| "loss": 1.0034, | |
| "step": 113 | |
| }, | |
| { | |
| "epoch": 0.5247410817031071, | |
| "grad_norm": 0.4447743594646454, | |
| "learning_rate": 4.2586817614407895e-05, | |
| "loss": 0.99, | |
| "step": 114 | |
| }, | |
| { | |
| "epoch": 0.5293440736478712, | |
| "grad_norm": 0.4581643044948578, | |
| "learning_rate": 4.17702704859633e-05, | |
| "loss": 1.0898, | |
| "step": 115 | |
| }, | |
| { | |
| "epoch": 0.5339470655926352, | |
| "grad_norm": 0.4533340036869049, | |
| "learning_rate": 4.095597328339452e-05, | |
| "loss": 1.07, | |
| "step": 116 | |
| }, | |
| { | |
| "epoch": 0.5385500575373993, | |
| "grad_norm": 0.4555422067642212, | |
| "learning_rate": 4.0144148627425993e-05, | |
| "loss": 1.1011, | |
| "step": 117 | |
| }, | |
| { | |
| "epoch": 0.5431530494821634, | |
| "grad_norm": 0.4426473081111908, | |
| "learning_rate": 3.933501846281267e-05, | |
| "loss": 1.025, | |
| "step": 118 | |
| }, | |
| { | |
| "epoch": 0.5477560414269275, | |
| "grad_norm": 0.434041827917099, | |
| "learning_rate": 3.852880399766243e-05, | |
| "loss": 0.9371, | |
| "step": 119 | |
| }, | |
| { | |
| "epoch": 0.5523590333716916, | |
| "grad_norm": 0.4938693940639496, | |
| "learning_rate": 3.772572564296005e-05, | |
| "loss": 1.0817, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.5569620253164557, | |
| "grad_norm": 0.476229190826416, | |
| "learning_rate": 3.6926002952309016e-05, | |
| "loss": 1.0545, | |
| "step": 121 | |
| }, | |
| { | |
| "epoch": 0.5615650172612198, | |
| "grad_norm": 0.48693299293518066, | |
| "learning_rate": 3.612985456190778e-05, | |
| "loss": 1.1153, | |
| "step": 122 | |
| }, | |
| { | |
| "epoch": 0.5661680092059839, | |
| "grad_norm": 0.7360255718231201, | |
| "learning_rate": 3.533749813077677e-05, | |
| "loss": 0.981, | |
| "step": 123 | |
| }, | |
| { | |
| "epoch": 0.570771001150748, | |
| "grad_norm": 0.47931551933288574, | |
| "learning_rate": 3.4549150281252636e-05, | |
| "loss": 1.0229, | |
| "step": 124 | |
| }, | |
| { | |
| "epoch": 0.5753739930955121, | |
| "grad_norm": 0.4814293682575226, | |
| "learning_rate": 3.3765026539765834e-05, | |
| "loss": 1.0339, | |
| "step": 125 | |
| }, | |
| { | |
| "epoch": 0.5799769850402762, | |
| "grad_norm": 0.49065276980400085, | |
| "learning_rate": 3.298534127791785e-05, | |
| "loss": 1.0304, | |
| "step": 126 | |
| }, | |
| { | |
| "epoch": 0.5845799769850403, | |
| "grad_norm": 0.5081751942634583, | |
| "learning_rate": 3.221030765387417e-05, | |
| "loss": 1.011, | |
| "step": 127 | |
| }, | |
| { | |
| "epoch": 0.5891829689298044, | |
| "grad_norm": 0.5222693085670471, | |
| "learning_rate": 3.144013755408895e-05, | |
| "loss": 1.0084, | |
| "step": 128 | |
| }, | |
| { | |
| "epoch": 0.5937859608745685, | |
| "grad_norm": 0.5362012386322021, | |
| "learning_rate": 3.0675041535377405e-05, | |
| "loss": 1.1292, | |
| "step": 129 | |
| }, | |
| { | |
| "epoch": 0.5983889528193326, | |
| "grad_norm": 0.5396473407745361, | |
| "learning_rate": 2.991522876735154e-05, | |
| "loss": 0.9754, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.6029919447640967, | |
| "grad_norm": 0.5463119149208069, | |
| "learning_rate": 2.916090697523549e-05, | |
| "loss": 1.0385, | |
| "step": 131 | |
| }, | |
| { | |
| "epoch": 0.6075949367088608, | |
| "grad_norm": 0.5413902401924133, | |
| "learning_rate": 2.8412282383075363e-05, | |
| "loss": 0.9387, | |
| "step": 132 | |
| }, | |
| { | |
| "epoch": 0.6121979286536249, | |
| "grad_norm": 0.5790338516235352, | |
| "learning_rate": 2.766955965735968e-05, | |
| "loss": 0.9761, | |
| "step": 133 | |
| }, | |
| { | |
| "epoch": 0.616800920598389, | |
| "grad_norm": 0.5730407238006592, | |
| "learning_rate": 2.693294185106562e-05, | |
| "loss": 0.9741, | |
| "step": 134 | |
| }, | |
| { | |
| "epoch": 0.6214039125431531, | |
| "grad_norm": 0.5818395018577576, | |
| "learning_rate": 2.6202630348146324e-05, | |
| "loss": 0.9907, | |
| "step": 135 | |
| }, | |
| { | |
| "epoch": 0.6260069044879172, | |
| "grad_norm": 0.6313637495040894, | |
| "learning_rate": 2.547882480847461e-05, | |
| "loss": 1.0225, | |
| "step": 136 | |
| }, | |
| { | |
| "epoch": 0.6306098964326813, | |
| "grad_norm": 0.6129395961761475, | |
| "learning_rate": 2.476172311325783e-05, | |
| "loss": 0.9362, | |
| "step": 137 | |
| }, | |
| { | |
| "epoch": 0.6352128883774454, | |
| "grad_norm": 0.6460527777671814, | |
| "learning_rate": 2.405152131093926e-05, | |
| "loss": 0.9351, | |
| "step": 138 | |
| }, | |
| { | |
| "epoch": 0.6398158803222095, | |
| "grad_norm": 0.7201539874076843, | |
| "learning_rate": 2.3348413563600325e-05, | |
| "loss": 0.9889, | |
| "step": 139 | |
| }, | |
| { | |
| "epoch": 0.6444188722669736, | |
| "grad_norm": 0.6339139938354492, | |
| "learning_rate": 2.2652592093878666e-05, | |
| "loss": 0.9575, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.6490218642117376, | |
| "grad_norm": 0.6779326796531677, | |
| "learning_rate": 2.196424713241637e-05, | |
| "loss": 0.8807, | |
| "step": 141 | |
| }, | |
| { | |
| "epoch": 0.6536248561565017, | |
| "grad_norm": 0.686235785484314, | |
| "learning_rate": 2.128356686585282e-05, | |
| "loss": 0.8411, | |
| "step": 142 | |
| }, | |
| { | |
| "epoch": 0.6582278481012658, | |
| "grad_norm": 0.7197779417037964, | |
| "learning_rate": 2.061073738537635e-05, | |
| "loss": 0.8494, | |
| "step": 143 | |
| }, | |
| { | |
| "epoch": 0.6628308400460299, | |
| "grad_norm": 0.7556268572807312, | |
| "learning_rate": 1.9945942635848748e-05, | |
| "loss": 0.8874, | |
| "step": 144 | |
| }, | |
| { | |
| "epoch": 0.667433831990794, | |
| "grad_norm": 0.7817888855934143, | |
| "learning_rate": 1.928936436551661e-05, | |
| "loss": 0.9061, | |
| "step": 145 | |
| }, | |
| { | |
| "epoch": 0.6720368239355581, | |
| "grad_norm": 0.7922728657722473, | |
| "learning_rate": 1.8641182076323148e-05, | |
| "loss": 0.9328, | |
| "step": 146 | |
| }, | |
| { | |
| "epoch": 0.6766398158803222, | |
| "grad_norm": 0.7909112572669983, | |
| "learning_rate": 1.800157297483417e-05, | |
| "loss": 0.8048, | |
| "step": 147 | |
| }, | |
| { | |
| "epoch": 0.6812428078250863, | |
| "grad_norm": 0.9154995083808899, | |
| "learning_rate": 1.7370711923791567e-05, | |
| "loss": 0.894, | |
| "step": 148 | |
| }, | |
| { | |
| "epoch": 0.6858457997698504, | |
| "grad_norm": 1.005481481552124, | |
| "learning_rate": 1.6748771394307585e-05, | |
| "loss": 0.9392, | |
| "step": 149 | |
| }, | |
| { | |
| "epoch": 0.6904487917146145, | |
| "grad_norm": 1.3467577695846558, | |
| "learning_rate": 1.6135921418712956e-05, | |
| "loss": 1.0209, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.6904487917146145, | |
| "eval_loss": 0.9763730764389038, | |
| "eval_runtime": 28.1458, | |
| "eval_samples_per_second": 13.004, | |
| "eval_steps_per_second": 3.269, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.6950517836593786, | |
| "grad_norm": 0.39665156602859497, | |
| "learning_rate": 1.553232954407171e-05, | |
| "loss": 0.8523, | |
| "step": 151 | |
| }, | |
| { | |
| "epoch": 0.6996547756041427, | |
| "grad_norm": 0.5035924315452576, | |
| "learning_rate": 1.4938160786375572e-05, | |
| "loss": 1.031, | |
| "step": 152 | |
| }, | |
| { | |
| "epoch": 0.7042577675489068, | |
| "grad_norm": 0.5049951076507568, | |
| "learning_rate": 1.435357758543015e-05, | |
| "loss": 0.9999, | |
| "step": 153 | |
| }, | |
| { | |
| "epoch": 0.7088607594936709, | |
| "grad_norm": 0.48856598138809204, | |
| "learning_rate": 1.3778739760445552e-05, | |
| "loss": 1.068, | |
| "step": 154 | |
| }, | |
| { | |
| "epoch": 0.713463751438435, | |
| "grad_norm": 0.4881579577922821, | |
| "learning_rate": 1.3213804466343421e-05, | |
| "loss": 1.0129, | |
| "step": 155 | |
| }, | |
| { | |
| "epoch": 0.7180667433831991, | |
| "grad_norm": 0.47163426876068115, | |
| "learning_rate": 1.2658926150792322e-05, | |
| "loss": 1.0582, | |
| "step": 156 | |
| }, | |
| { | |
| "epoch": 0.7226697353279632, | |
| "grad_norm": 0.44049975275993347, | |
| "learning_rate": 1.2114256511983274e-05, | |
| "loss": 1.0073, | |
| "step": 157 | |
| }, | |
| { | |
| "epoch": 0.7272727272727273, | |
| "grad_norm": 0.40652769804000854, | |
| "learning_rate": 1.157994445715706e-05, | |
| "loss": 0.9726, | |
| "step": 158 | |
| }, | |
| { | |
| "epoch": 0.7318757192174914, | |
| "grad_norm": 0.4326465129852295, | |
| "learning_rate": 1.1056136061894384e-05, | |
| "loss": 1.0032, | |
| "step": 159 | |
| }, | |
| { | |
| "epoch": 0.7364787111622555, | |
| "grad_norm": 0.4349287748336792, | |
| "learning_rate": 1.0542974530180327e-05, | |
| "loss": 1.0564, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.7410817031070196, | |
| "grad_norm": 0.3993092477321625, | |
| "learning_rate": 1.0040600155253765e-05, | |
| "loss": 1.0186, | |
| "step": 161 | |
| }, | |
| { | |
| "epoch": 0.7456846950517837, | |
| "grad_norm": 0.43041473627090454, | |
| "learning_rate": 9.549150281252633e-06, | |
| "loss": 1.0773, | |
| "step": 162 | |
| }, | |
| { | |
| "epoch": 0.7502876869965478, | |
| "grad_norm": 0.421023428440094, | |
| "learning_rate": 9.068759265665384e-06, | |
| "loss": 0.9614, | |
| "step": 163 | |
| }, | |
| { | |
| "epoch": 0.7548906789413119, | |
| "grad_norm": 0.4321349263191223, | |
| "learning_rate": 8.599558442598998e-06, | |
| "loss": 1.0472, | |
| "step": 164 | |
| }, | |
| { | |
| "epoch": 0.759493670886076, | |
| "grad_norm": 0.45273786783218384, | |
| "learning_rate": 8.141676086873572e-06, | |
| "loss": 1.1186, | |
| "step": 165 | |
| }, | |
| { | |
| "epoch": 0.7640966628308401, | |
| "grad_norm": 0.4438074231147766, | |
| "learning_rate": 7.695237378953223e-06, | |
| "loss": 1.0857, | |
| "step": 166 | |
| }, | |
| { | |
| "epoch": 0.7686996547756041, | |
| "grad_norm": 0.4354915916919708, | |
| "learning_rate": 7.260364370723044e-06, | |
| "loss": 1.1232, | |
| "step": 167 | |
| }, | |
| { | |
| "epoch": 0.7733026467203682, | |
| "grad_norm": 0.44042718410491943, | |
| "learning_rate": 6.837175952121306e-06, | |
| "loss": 1.0425, | |
| "step": 168 | |
| }, | |
| { | |
| "epoch": 0.7779056386651323, | |
| "grad_norm": 0.4523642063140869, | |
| "learning_rate": 6.425787818636131e-06, | |
| "loss": 1.1064, | |
| "step": 169 | |
| }, | |
| { | |
| "epoch": 0.7825086306098964, | |
| "grad_norm": 0.44828060269355774, | |
| "learning_rate": 6.026312439675552e-06, | |
| "loss": 1.0005, | |
| "step": 170 | |
| }, | |
| { | |
| "epoch": 0.7871116225546605, | |
| "grad_norm": 0.46433570981025696, | |
| "learning_rate": 5.6388590278194096e-06, | |
| "loss": 1.1146, | |
| "step": 171 | |
| }, | |
| { | |
| "epoch": 0.7917146144994246, | |
| "grad_norm": 0.4922254979610443, | |
| "learning_rate": 5.263533508961827e-06, | |
| "loss": 1.0954, | |
| "step": 172 | |
| }, | |
| { | |
| "epoch": 0.7963176064441887, | |
| "grad_norm": 0.4795372188091278, | |
| "learning_rate": 4.900438493352055e-06, | |
| "loss": 1.0052, | |
| "step": 173 | |
| }, | |
| { | |
| "epoch": 0.8009205983889528, | |
| "grad_norm": 0.4732430577278137, | |
| "learning_rate": 4.549673247541875e-06, | |
| "loss": 1.079, | |
| "step": 174 | |
| }, | |
| { | |
| "epoch": 0.8055235903337169, | |
| "grad_norm": 0.4793175160884857, | |
| "learning_rate": 4.2113336672471245e-06, | |
| "loss": 1.0312, | |
| "step": 175 | |
| }, | |
| { | |
| "epoch": 0.810126582278481, | |
| "grad_norm": 0.5124372839927673, | |
| "learning_rate": 3.885512251130763e-06, | |
| "loss": 1.0154, | |
| "step": 176 | |
| }, | |
| { | |
| "epoch": 0.8147295742232451, | |
| "grad_norm": 0.529222309589386, | |
| "learning_rate": 3.5722980755146517e-06, | |
| "loss": 1.0745, | |
| "step": 177 | |
| }, | |
| { | |
| "epoch": 0.8193325661680092, | |
| "grad_norm": 0.511615514755249, | |
| "learning_rate": 3.271776770026963e-06, | |
| "loss": 1.0294, | |
| "step": 178 | |
| }, | |
| { | |
| "epoch": 0.8239355581127733, | |
| "grad_norm": 0.5564742684364319, | |
| "learning_rate": 2.9840304941919415e-06, | |
| "loss": 1.0023, | |
| "step": 179 | |
| }, | |
| { | |
| "epoch": 0.8285385500575374, | |
| "grad_norm": 0.5130892395973206, | |
| "learning_rate": 2.7091379149682685e-06, | |
| "loss": 0.9765, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.8331415420023015, | |
| "grad_norm": 0.5212732553482056, | |
| "learning_rate": 2.4471741852423237e-06, | |
| "loss": 1.0086, | |
| "step": 181 | |
| }, | |
| { | |
| "epoch": 0.8377445339470656, | |
| "grad_norm": 0.5394339561462402, | |
| "learning_rate": 2.1982109232821178e-06, | |
| "loss": 0.9639, | |
| "step": 182 | |
| }, | |
| { | |
| "epoch": 0.8423475258918297, | |
| "grad_norm": 0.5359840393066406, | |
| "learning_rate": 1.962316193157593e-06, | |
| "loss": 1.0448, | |
| "step": 183 | |
| }, | |
| { | |
| "epoch": 0.8469505178365938, | |
| "grad_norm": 0.5474115014076233, | |
| "learning_rate": 1.7395544861325718e-06, | |
| "loss": 0.9088, | |
| "step": 184 | |
| }, | |
| { | |
| "epoch": 0.8515535097813579, | |
| "grad_norm": 0.5707780718803406, | |
| "learning_rate": 1.5299867030334814e-06, | |
| "loss": 0.9835, | |
| "step": 185 | |
| }, | |
| { | |
| "epoch": 0.856156501726122, | |
| "grad_norm": 0.6048805117607117, | |
| "learning_rate": 1.333670137599713e-06, | |
| "loss": 0.9068, | |
| "step": 186 | |
| }, | |
| { | |
| "epoch": 0.8607594936708861, | |
| "grad_norm": 0.5939739346504211, | |
| "learning_rate": 1.1506584608200367e-06, | |
| "loss": 0.9709, | |
| "step": 187 | |
| }, | |
| { | |
| "epoch": 0.8653624856156502, | |
| "grad_norm": 0.6288136839866638, | |
| "learning_rate": 9.810017062595322e-07, | |
| "loss": 0.9486, | |
| "step": 188 | |
| }, | |
| { | |
| "epoch": 0.8699654775604143, | |
| "grad_norm": 0.6477128863334656, | |
| "learning_rate": 8.247462563808817e-07, | |
| "loss": 0.9473, | |
| "step": 189 | |
| }, | |
| { | |
| "epoch": 0.8745684695051784, | |
| "grad_norm": 0.6742326617240906, | |
| "learning_rate": 6.819348298638839e-07, | |
| "loss": 0.8992, | |
| "step": 190 | |
| }, | |
| { | |
| "epoch": 0.8791714614499425, | |
| "grad_norm": 0.6942002177238464, | |
| "learning_rate": 5.526064699265753e-07, | |
| "loss": 0.9257, | |
| "step": 191 | |
| }, | |
| { | |
| "epoch": 0.8837744533947065, | |
| "grad_norm": 0.7023844718933105, | |
| "learning_rate": 4.367965336512403e-07, | |
| "loss": 0.9569, | |
| "step": 192 | |
| }, | |
| { | |
| "epoch": 0.8883774453394706, | |
| "grad_norm": 0.7281048893928528, | |
| "learning_rate": 3.3453668231809286e-07, | |
| "loss": 0.8543, | |
| "step": 193 | |
| }, | |
| { | |
| "epoch": 0.8929804372842347, | |
| "grad_norm": 0.761889636516571, | |
| "learning_rate": 2.458548727494292e-07, | |
| "loss": 0.8321, | |
| "step": 194 | |
| }, | |
| { | |
| "epoch": 0.8975834292289988, | |
| "grad_norm": 0.7388656735420227, | |
| "learning_rate": 1.7077534966650766e-07, | |
| "loss": 0.8171, | |
| "step": 195 | |
| }, | |
| { | |
| "epoch": 0.9021864211737629, | |
| "grad_norm": 0.8363285660743713, | |
| "learning_rate": 1.0931863906127327e-07, | |
| "loss": 0.9059, | |
| "step": 196 | |
| }, | |
| { | |
| "epoch": 0.906789413118527, | |
| "grad_norm": 1.0079214572906494, | |
| "learning_rate": 6.150154258476315e-08, | |
| "loss": 0.8452, | |
| "step": 197 | |
| }, | |
| { | |
| "epoch": 0.9113924050632911, | |
| "grad_norm": 0.9205271005630493, | |
| "learning_rate": 2.7337132953697554e-08, | |
| "loss": 0.8705, | |
| "step": 198 | |
| }, | |
| { | |
| "epoch": 0.9159953970080552, | |
| "grad_norm": 1.1067923307418823, | |
| "learning_rate": 6.834750376549792e-09, | |
| "loss": 0.8839, | |
| "step": 199 | |
| }, | |
| { | |
| "epoch": 0.9205983889528193, | |
| "grad_norm": 1.5278611183166504, | |
| "learning_rate": 0.0, | |
| "loss": 0.9898, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.9205983889528193, | |
| "eval_loss": 0.9496371746063232, | |
| "eval_runtime": 28.1444, | |
| "eval_samples_per_second": 13.004, | |
| "eval_steps_per_second": 3.269, | |
| "step": 200 | |
| } | |
| ], | |
| "logging_steps": 1, | |
| "max_steps": 200, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 1, | |
| "save_steps": 50, | |
| "stateful_callbacks": { | |
| "EarlyStoppingCallback": { | |
| "args": { | |
| "early_stopping_patience": 5, | |
| "early_stopping_threshold": 0.0 | |
| }, | |
| "attributes": { | |
| "early_stopping_patience_counter": 0 | |
| } | |
| }, | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 3.0321122932791706e+17, | |
| "train_batch_size": 8, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |