Instructions to use alchemist69/ad4b2a37-8700-4e5c-9aef-98e2b0aab97c with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use alchemist69/ad4b2a37-8700-4e5c-9aef-98e2b0aab97c with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("unsloth/mistral-7b-instruct-v0.3") model = PeftModel.from_pretrained(base_model, "alchemist69/ad4b2a37-8700-4e5c-9aef-98e2b0aab97c") - Notebooks
- Google Colab
- Kaggle
Download last-checkpoint/trainer_state.json from alchemist69/ad4b2a37-8700-4e5c-9aef-98e2b0aab97c: direct link, hf CLI and curl.
- Browser
- Download file 27.5 kB
-
https://huggingface.co/alchemist69/ad4b2a37-8700-4e5c-9aef-98e2b0aab97c/resolve/cabce40ad127199c273cd25ac548ad100ceaa1f6/last-checkpoint/trainer_state.json
- Command line
-
hf download hf://alchemist69/ad4b2a37-8700-4e5c-9aef-98e2b0aab97c@cabce40ad127199c273cd25ac548ad100ceaa1f6/last-checkpoint/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/alchemist69/ad4b2a37-8700-4e5c-9aef-98e2b0aab97c/resolve/cabce40ad127199c273cd25ac548ad100ceaa1f6/last-checkpoint/trainer_state.json
27.5 kB
| { | |
| "best_metric": 0.24679341912269592, | |
| "best_model_checkpoint": "miner_id_24/checkpoint-150", | |
| "epoch": 0.5876591576885406, | |
| "eval_steps": 50, | |
| "global_step": 150, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.0039177277179236044, | |
| "grad_norm": 35.0140266418457, | |
| "learning_rate": 1e-05, | |
| "loss": 3.3377, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.0039177277179236044, | |
| "eval_loss": 1.376975655555725, | |
| "eval_runtime": 31.0307, | |
| "eval_samples_per_second": 13.857, | |
| "eval_steps_per_second": 3.48, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.007835455435847209, | |
| "grad_norm": 37.351539611816406, | |
| "learning_rate": 2e-05, | |
| "loss": 3.5823, | |
| "step": 2 | |
| }, | |
| { | |
| "epoch": 0.011753183153770812, | |
| "grad_norm": 27.027957916259766, | |
| "learning_rate": 3e-05, | |
| "loss": 3.0971, | |
| "step": 3 | |
| }, | |
| { | |
| "epoch": 0.015670910871694418, | |
| "grad_norm": 17.421297073364258, | |
| "learning_rate": 4e-05, | |
| "loss": 2.4171, | |
| "step": 4 | |
| }, | |
| { | |
| "epoch": 0.019588638589618023, | |
| "grad_norm": 13.550549507141113, | |
| "learning_rate": 5e-05, | |
| "loss": 1.7853, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.023506366307541625, | |
| "grad_norm": 6.565335273742676, | |
| "learning_rate": 6e-05, | |
| "loss": 1.5603, | |
| "step": 6 | |
| }, | |
| { | |
| "epoch": 0.02742409402546523, | |
| "grad_norm": 4.426646709442139, | |
| "learning_rate": 7e-05, | |
| "loss": 1.2931, | |
| "step": 7 | |
| }, | |
| { | |
| "epoch": 0.031341821743388835, | |
| "grad_norm": 3.8525550365448, | |
| "learning_rate": 8e-05, | |
| "loss": 1.3958, | |
| "step": 8 | |
| }, | |
| { | |
| "epoch": 0.03525954946131244, | |
| "grad_norm": 3.435671329498291, | |
| "learning_rate": 9e-05, | |
| "loss": 1.299, | |
| "step": 9 | |
| }, | |
| { | |
| "epoch": 0.039177277179236046, | |
| "grad_norm": 3.1870250701904297, | |
| "learning_rate": 0.0001, | |
| "loss": 1.0837, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.043095004897159644, | |
| "grad_norm": 17.622121810913086, | |
| "learning_rate": 9.999316524962345e-05, | |
| "loss": 1.1129, | |
| "step": 11 | |
| }, | |
| { | |
| "epoch": 0.04701273261508325, | |
| "grad_norm": 3.806305408477783, | |
| "learning_rate": 9.997266286704631e-05, | |
| "loss": 1.1438, | |
| "step": 12 | |
| }, | |
| { | |
| "epoch": 0.050930460333006855, | |
| "grad_norm": 3.5566141605377197, | |
| "learning_rate": 9.993849845741524e-05, | |
| "loss": 1.1891, | |
| "step": 13 | |
| }, | |
| { | |
| "epoch": 0.05484818805093046, | |
| "grad_norm": 3.109950065612793, | |
| "learning_rate": 9.989068136093873e-05, | |
| "loss": 1.2167, | |
| "step": 14 | |
| }, | |
| { | |
| "epoch": 0.058765915768854066, | |
| "grad_norm": 3.230849027633667, | |
| "learning_rate": 9.98292246503335e-05, | |
| "loss": 1.1028, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.06268364348677767, | |
| "grad_norm": 3.521568775177002, | |
| "learning_rate": 9.975414512725057e-05, | |
| "loss": 1.3616, | |
| "step": 16 | |
| }, | |
| { | |
| "epoch": 0.06660137120470128, | |
| "grad_norm": 2.9588394165039062, | |
| "learning_rate": 9.966546331768191e-05, | |
| "loss": 1.1785, | |
| "step": 17 | |
| }, | |
| { | |
| "epoch": 0.07051909892262488, | |
| "grad_norm": 3.2159998416900635, | |
| "learning_rate": 9.956320346634876e-05, | |
| "loss": 1.3242, | |
| "step": 18 | |
| }, | |
| { | |
| "epoch": 0.07443682664054849, | |
| "grad_norm": 2.8295371532440186, | |
| "learning_rate": 9.944739353007344e-05, | |
| "loss": 1.0318, | |
| "step": 19 | |
| }, | |
| { | |
| "epoch": 0.07835455435847209, | |
| "grad_norm": 3.3692362308502197, | |
| "learning_rate": 9.931806517013612e-05, | |
| "loss": 1.3589, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.08227228207639568, | |
| "grad_norm": 2.8950631618499756, | |
| "learning_rate": 9.917525374361912e-05, | |
| "loss": 1.252, | |
| "step": 21 | |
| }, | |
| { | |
| "epoch": 0.08619000979431929, | |
| "grad_norm": 2.7280898094177246, | |
| "learning_rate": 9.901899829374047e-05, | |
| "loss": 1.1296, | |
| "step": 22 | |
| }, | |
| { | |
| "epoch": 0.0901077375122429, | |
| "grad_norm": 2.719750165939331, | |
| "learning_rate": 9.884934153917997e-05, | |
| "loss": 1.1225, | |
| "step": 23 | |
| }, | |
| { | |
| "epoch": 0.0940254652301665, | |
| "grad_norm": 2.513847827911377, | |
| "learning_rate": 9.86663298624003e-05, | |
| "loss": 1.1292, | |
| "step": 24 | |
| }, | |
| { | |
| "epoch": 0.0979431929480901, | |
| "grad_norm": 2.713838815689087, | |
| "learning_rate": 9.847001329696653e-05, | |
| "loss": 1.1088, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 0.10186092066601371, | |
| "grad_norm": 3.4801487922668457, | |
| "learning_rate": 9.826044551386744e-05, | |
| "loss": 1.2598, | |
| "step": 26 | |
| }, | |
| { | |
| "epoch": 0.10577864838393732, | |
| "grad_norm": 2.5598838329315186, | |
| "learning_rate": 9.803768380684242e-05, | |
| "loss": 0.9171, | |
| "step": 27 | |
| }, | |
| { | |
| "epoch": 0.10969637610186092, | |
| "grad_norm": 3.0595102310180664, | |
| "learning_rate": 9.780178907671789e-05, | |
| "loss": 1.1694, | |
| "step": 28 | |
| }, | |
| { | |
| "epoch": 0.11361410381978453, | |
| "grad_norm": 2.868380308151245, | |
| "learning_rate": 9.755282581475769e-05, | |
| "loss": 1.0886, | |
| "step": 29 | |
| }, | |
| { | |
| "epoch": 0.11753183153770813, | |
| "grad_norm": 3.10597562789917, | |
| "learning_rate": 9.729086208503174e-05, | |
| "loss": 1.1942, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.12144955925563174, | |
| "grad_norm": 2.72220778465271, | |
| "learning_rate": 9.701596950580806e-05, | |
| "loss": 0.9602, | |
| "step": 31 | |
| }, | |
| { | |
| "epoch": 0.12536728697355534, | |
| "grad_norm": 2.4036269187927246, | |
| "learning_rate": 9.672822322997305e-05, | |
| "loss": 0.9449, | |
| "step": 32 | |
| }, | |
| { | |
| "epoch": 0.12928501469147893, | |
| "grad_norm": 2.904219150543213, | |
| "learning_rate": 9.642770192448536e-05, | |
| "loss": 1.1555, | |
| "step": 33 | |
| }, | |
| { | |
| "epoch": 0.13320274240940255, | |
| "grad_norm": 2.7303013801574707, | |
| "learning_rate": 9.611448774886924e-05, | |
| "loss": 1.1668, | |
| "step": 34 | |
| }, | |
| { | |
| "epoch": 0.13712047012732614, | |
| "grad_norm": 2.835772752761841, | |
| "learning_rate": 9.578866633275288e-05, | |
| "loss": 1.0502, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 0.14103819784524976, | |
| "grad_norm": 2.830291271209717, | |
| "learning_rate": 9.545032675245813e-05, | |
| "loss": 1.1096, | |
| "step": 36 | |
| }, | |
| { | |
| "epoch": 0.14495592556317335, | |
| "grad_norm": 2.625466823577881, | |
| "learning_rate": 9.509956150664796e-05, | |
| "loss": 1.037, | |
| "step": 37 | |
| }, | |
| { | |
| "epoch": 0.14887365328109697, | |
| "grad_norm": 3.175342559814453, | |
| "learning_rate": 9.473646649103818e-05, | |
| "loss": 1.1339, | |
| "step": 38 | |
| }, | |
| { | |
| "epoch": 0.15279138099902057, | |
| "grad_norm": 2.816603422164917, | |
| "learning_rate": 9.43611409721806e-05, | |
| "loss": 0.8884, | |
| "step": 39 | |
| }, | |
| { | |
| "epoch": 0.15670910871694418, | |
| "grad_norm": 3.1322615146636963, | |
| "learning_rate": 9.397368756032445e-05, | |
| "loss": 1.0401, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.16062683643486778, | |
| "grad_norm": 3.353376626968384, | |
| "learning_rate": 9.357421218136386e-05, | |
| "loss": 1.1617, | |
| "step": 41 | |
| }, | |
| { | |
| "epoch": 0.16454456415279137, | |
| "grad_norm": 2.9151480197906494, | |
| "learning_rate": 9.316282404787871e-05, | |
| "loss": 1.036, | |
| "step": 42 | |
| }, | |
| { | |
| "epoch": 0.168462291870715, | |
| "grad_norm": 2.935025453567505, | |
| "learning_rate": 9.273963562927695e-05, | |
| "loss": 1.0222, | |
| "step": 43 | |
| }, | |
| { | |
| "epoch": 0.17238001958863858, | |
| "grad_norm": 2.736802339553833, | |
| "learning_rate": 9.230476262104677e-05, | |
| "loss": 1.0799, | |
| "step": 44 | |
| }, | |
| { | |
| "epoch": 0.1762977473065622, | |
| "grad_norm": 3.0634095668792725, | |
| "learning_rate": 9.185832391312644e-05, | |
| "loss": 1.0179, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 0.1802154750244858, | |
| "grad_norm": 2.914475202560425, | |
| "learning_rate": 9.140044155740101e-05, | |
| "loss": 0.8634, | |
| "step": 46 | |
| }, | |
| { | |
| "epoch": 0.1841332027424094, | |
| "grad_norm": 3.3175463676452637, | |
| "learning_rate": 9.093124073433463e-05, | |
| "loss": 0.9203, | |
| "step": 47 | |
| }, | |
| { | |
| "epoch": 0.188050930460333, | |
| "grad_norm": 3.209653377532959, | |
| "learning_rate": 9.045084971874738e-05, | |
| "loss": 1.0183, | |
| "step": 48 | |
| }, | |
| { | |
| "epoch": 0.19196865817825662, | |
| "grad_norm": 2.7022249698638916, | |
| "learning_rate": 8.995939984474624e-05, | |
| "loss": 0.7421, | |
| "step": 49 | |
| }, | |
| { | |
| "epoch": 0.1958863858961802, | |
| "grad_norm": 3.0036754608154297, | |
| "learning_rate": 8.945702546981969e-05, | |
| "loss": 0.7637, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.1958863858961802, | |
| "eval_loss": 0.3203330934047699, | |
| "eval_runtime": 31.6732, | |
| "eval_samples_per_second": 13.576, | |
| "eval_steps_per_second": 3.41, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.19980411361410383, | |
| "grad_norm": 6.922638893127441, | |
| "learning_rate": 8.894386393810563e-05, | |
| "loss": 1.5597, | |
| "step": 51 | |
| }, | |
| { | |
| "epoch": 0.20372184133202742, | |
| "grad_norm": 4.306941032409668, | |
| "learning_rate": 8.842005554284296e-05, | |
| "loss": 1.3601, | |
| "step": 52 | |
| }, | |
| { | |
| "epoch": 0.20763956904995104, | |
| "grad_norm": 2.745713949203491, | |
| "learning_rate": 8.788574348801675e-05, | |
| "loss": 1.3413, | |
| "step": 53 | |
| }, | |
| { | |
| "epoch": 0.21155729676787463, | |
| "grad_norm": 2.2029619216918945, | |
| "learning_rate": 8.73410738492077e-05, | |
| "loss": 1.1808, | |
| "step": 54 | |
| }, | |
| { | |
| "epoch": 0.21547502448579825, | |
| "grad_norm": 2.272630214691162, | |
| "learning_rate": 8.678619553365659e-05, | |
| "loss": 1.0573, | |
| "step": 55 | |
| }, | |
| { | |
| "epoch": 0.21939275220372184, | |
| "grad_norm": 2.146240711212158, | |
| "learning_rate": 8.622126023955446e-05, | |
| "loss": 1.0678, | |
| "step": 56 | |
| }, | |
| { | |
| "epoch": 0.22331047992164543, | |
| "grad_norm": 2.200601577758789, | |
| "learning_rate": 8.564642241456986e-05, | |
| "loss": 1.1393, | |
| "step": 57 | |
| }, | |
| { | |
| "epoch": 0.22722820763956905, | |
| "grad_norm": 2.271404981613159, | |
| "learning_rate": 8.506183921362443e-05, | |
| "loss": 1.1989, | |
| "step": 58 | |
| }, | |
| { | |
| "epoch": 0.23114593535749264, | |
| "grad_norm": 2.0966899394989014, | |
| "learning_rate": 8.44676704559283e-05, | |
| "loss": 1.2637, | |
| "step": 59 | |
| }, | |
| { | |
| "epoch": 0.23506366307541626, | |
| "grad_norm": 1.988345980644226, | |
| "learning_rate": 8.386407858128706e-05, | |
| "loss": 1.1106, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.23898139079333985, | |
| "grad_norm": 2.0976436138153076, | |
| "learning_rate": 8.32512286056924e-05, | |
| "loss": 0.9696, | |
| "step": 61 | |
| }, | |
| { | |
| "epoch": 0.24289911851126347, | |
| "grad_norm": 2.122800350189209, | |
| "learning_rate": 8.262928807620843e-05, | |
| "loss": 1.1271, | |
| "step": 62 | |
| }, | |
| { | |
| "epoch": 0.24681684622918706, | |
| "grad_norm": 2.22821044921875, | |
| "learning_rate": 8.199842702516583e-05, | |
| "loss": 1.1168, | |
| "step": 63 | |
| }, | |
| { | |
| "epoch": 0.2507345739471107, | |
| "grad_norm": 2.1739425659179688, | |
| "learning_rate": 8.135881792367686e-05, | |
| "loss": 1.1319, | |
| "step": 64 | |
| }, | |
| { | |
| "epoch": 0.2546523016650343, | |
| "grad_norm": 2.167677879333496, | |
| "learning_rate": 8.07106356344834e-05, | |
| "loss": 1.1428, | |
| "step": 65 | |
| }, | |
| { | |
| "epoch": 0.25857002938295787, | |
| "grad_norm": 2.0300333499908447, | |
| "learning_rate": 8.005405736415126e-05, | |
| "loss": 1.0347, | |
| "step": 66 | |
| }, | |
| { | |
| "epoch": 0.2624877571008815, | |
| "grad_norm": 2.1151342391967773, | |
| "learning_rate": 7.938926261462366e-05, | |
| "loss": 1.095, | |
| "step": 67 | |
| }, | |
| { | |
| "epoch": 0.2664054848188051, | |
| "grad_norm": 2.0412073135375977, | |
| "learning_rate": 7.871643313414718e-05, | |
| "loss": 1.1526, | |
| "step": 68 | |
| }, | |
| { | |
| "epoch": 0.2703232125367287, | |
| "grad_norm": 2.337883710861206, | |
| "learning_rate": 7.803575286758364e-05, | |
| "loss": 1.0398, | |
| "step": 69 | |
| }, | |
| { | |
| "epoch": 0.2742409402546523, | |
| "grad_norm": 2.6651041507720947, | |
| "learning_rate": 7.734740790612136e-05, | |
| "loss": 0.9876, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.2781586679725759, | |
| "grad_norm": 2.350996732711792, | |
| "learning_rate": 7.66515864363997e-05, | |
| "loss": 0.8954, | |
| "step": 71 | |
| }, | |
| { | |
| "epoch": 0.2820763956904995, | |
| "grad_norm": 2.271243095397949, | |
| "learning_rate": 7.594847868906076e-05, | |
| "loss": 0.99, | |
| "step": 72 | |
| }, | |
| { | |
| "epoch": 0.2859941234084231, | |
| "grad_norm": 2.2934367656707764, | |
| "learning_rate": 7.52382768867422e-05, | |
| "loss": 1.0287, | |
| "step": 73 | |
| }, | |
| { | |
| "epoch": 0.2899118511263467, | |
| "grad_norm": 2.434135913848877, | |
| "learning_rate": 7.452117519152542e-05, | |
| "loss": 1.1248, | |
| "step": 74 | |
| }, | |
| { | |
| "epoch": 0.2938295788442703, | |
| "grad_norm": 2.567089557647705, | |
| "learning_rate": 7.379736965185368e-05, | |
| "loss": 1.1565, | |
| "step": 75 | |
| }, | |
| { | |
| "epoch": 0.29774730656219395, | |
| "grad_norm": 2.120260238647461, | |
| "learning_rate": 7.30670581489344e-05, | |
| "loss": 1.0654, | |
| "step": 76 | |
| }, | |
| { | |
| "epoch": 0.30166503428011754, | |
| "grad_norm": 2.736607313156128, | |
| "learning_rate": 7.233044034264034e-05, | |
| "loss": 1.3288, | |
| "step": 77 | |
| }, | |
| { | |
| "epoch": 0.30558276199804113, | |
| "grad_norm": 2.2041382789611816, | |
| "learning_rate": 7.158771761692464e-05, | |
| "loss": 1.0499, | |
| "step": 78 | |
| }, | |
| { | |
| "epoch": 0.3095004897159647, | |
| "grad_norm": 2.29158616065979, | |
| "learning_rate": 7.083909302476453e-05, | |
| "loss": 1.1707, | |
| "step": 79 | |
| }, | |
| { | |
| "epoch": 0.31341821743388837, | |
| "grad_norm": 2.2305917739868164, | |
| "learning_rate": 7.008477123264848e-05, | |
| "loss": 0.9847, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.31733594515181196, | |
| "grad_norm": 3.056943893432617, | |
| "learning_rate": 6.932495846462261e-05, | |
| "loss": 1.166, | |
| "step": 81 | |
| }, | |
| { | |
| "epoch": 0.32125367286973555, | |
| "grad_norm": 2.7335922718048096, | |
| "learning_rate": 6.855986244591104e-05, | |
| "loss": 0.9707, | |
| "step": 82 | |
| }, | |
| { | |
| "epoch": 0.32517140058765914, | |
| "grad_norm": 2.3004262447357178, | |
| "learning_rate": 6.778969234612584e-05, | |
| "loss": 0.9733, | |
| "step": 83 | |
| }, | |
| { | |
| "epoch": 0.32908912830558273, | |
| "grad_norm": 2.531168222427368, | |
| "learning_rate": 6.701465872208216e-05, | |
| "loss": 1.1586, | |
| "step": 84 | |
| }, | |
| { | |
| "epoch": 0.3330068560235064, | |
| "grad_norm": 2.658029317855835, | |
| "learning_rate": 6.623497346023418e-05, | |
| "loss": 0.924, | |
| "step": 85 | |
| }, | |
| { | |
| "epoch": 0.33692458374143, | |
| "grad_norm": 2.634483575820923, | |
| "learning_rate": 6.545084971874738e-05, | |
| "loss": 0.8868, | |
| "step": 86 | |
| }, | |
| { | |
| "epoch": 0.34084231145935356, | |
| "grad_norm": 2.426271915435791, | |
| "learning_rate": 6.466250186922325e-05, | |
| "loss": 0.9485, | |
| "step": 87 | |
| }, | |
| { | |
| "epoch": 0.34476003917727716, | |
| "grad_norm": 2.598158836364746, | |
| "learning_rate": 6.387014543809223e-05, | |
| "loss": 0.9368, | |
| "step": 88 | |
| }, | |
| { | |
| "epoch": 0.3486777668952008, | |
| "grad_norm": 2.328049421310425, | |
| "learning_rate": 6.307399704769099e-05, | |
| "loss": 0.8935, | |
| "step": 89 | |
| }, | |
| { | |
| "epoch": 0.3525954946131244, | |
| "grad_norm": 3.068272590637207, | |
| "learning_rate": 6.227427435703997e-05, | |
| "loss": 1.2512, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.356513222331048, | |
| "grad_norm": 2.496204137802124, | |
| "learning_rate": 6.147119600233758e-05, | |
| "loss": 0.9942, | |
| "step": 91 | |
| }, | |
| { | |
| "epoch": 0.3604309500489716, | |
| "grad_norm": 2.955332040786743, | |
| "learning_rate": 6.066498153718735e-05, | |
| "loss": 1.029, | |
| "step": 92 | |
| }, | |
| { | |
| "epoch": 0.3643486777668952, | |
| "grad_norm": 2.5348706245422363, | |
| "learning_rate": 5.985585137257401e-05, | |
| "loss": 0.9778, | |
| "step": 93 | |
| }, | |
| { | |
| "epoch": 0.3682664054848188, | |
| "grad_norm": 2.6345531940460205, | |
| "learning_rate": 5.90440267166055e-05, | |
| "loss": 1.0212, | |
| "step": 94 | |
| }, | |
| { | |
| "epoch": 0.3721841332027424, | |
| "grad_norm": 2.6471123695373535, | |
| "learning_rate": 5.8229729514036705e-05, | |
| "loss": 1.0014, | |
| "step": 95 | |
| }, | |
| { | |
| "epoch": 0.376101860920666, | |
| "grad_norm": 3.0204238891601562, | |
| "learning_rate": 5.74131823855921e-05, | |
| "loss": 0.8708, | |
| "step": 96 | |
| }, | |
| { | |
| "epoch": 0.38001958863858964, | |
| "grad_norm": 2.308615207672119, | |
| "learning_rate": 5.6594608567103456e-05, | |
| "loss": 0.868, | |
| "step": 97 | |
| }, | |
| { | |
| "epoch": 0.38393731635651324, | |
| "grad_norm": 2.2340455055236816, | |
| "learning_rate": 5.577423184847932e-05, | |
| "loss": 0.7931, | |
| "step": 98 | |
| }, | |
| { | |
| "epoch": 0.3878550440744368, | |
| "grad_norm": 2.526561737060547, | |
| "learning_rate": 5.495227651252315e-05, | |
| "loss": 0.6846, | |
| "step": 99 | |
| }, | |
| { | |
| "epoch": 0.3917727717923604, | |
| "grad_norm": 3.3785598278045654, | |
| "learning_rate": 5.4128967273616625e-05, | |
| "loss": 0.8847, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.3917727717923604, | |
| "eval_loss": 0.2694668471813202, | |
| "eval_runtime": 31.6768, | |
| "eval_samples_per_second": 13.575, | |
| "eval_steps_per_second": 3.409, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.395690499510284, | |
| "grad_norm": 3.07016658782959, | |
| "learning_rate": 5.330452921628497e-05, | |
| "loss": 1.4631, | |
| "step": 101 | |
| }, | |
| { | |
| "epoch": 0.39960822722820766, | |
| "grad_norm": 2.533738136291504, | |
| "learning_rate": 5.247918773366112e-05, | |
| "loss": 1.1278, | |
| "step": 102 | |
| }, | |
| { | |
| "epoch": 0.40352595494613125, | |
| "grad_norm": 2.074531316757202, | |
| "learning_rate": 5.165316846586541e-05, | |
| "loss": 1.1451, | |
| "step": 103 | |
| }, | |
| { | |
| "epoch": 0.40744368266405484, | |
| "grad_norm": 1.5933665037155151, | |
| "learning_rate": 5.0826697238317935e-05, | |
| "loss": 0.9272, | |
| "step": 104 | |
| }, | |
| { | |
| "epoch": 0.41136141038197843, | |
| "grad_norm": 1.8569263219833374, | |
| "learning_rate": 5e-05, | |
| "loss": 1.1566, | |
| "step": 105 | |
| }, | |
| { | |
| "epoch": 0.4152791380999021, | |
| "grad_norm": 1.7498559951782227, | |
| "learning_rate": 4.917330276168208e-05, | |
| "loss": 1.1107, | |
| "step": 106 | |
| }, | |
| { | |
| "epoch": 0.41919686581782567, | |
| "grad_norm": 1.6929476261138916, | |
| "learning_rate": 4.834683153413459e-05, | |
| "loss": 1.0605, | |
| "step": 107 | |
| }, | |
| { | |
| "epoch": 0.42311459353574926, | |
| "grad_norm": 1.9647297859191895, | |
| "learning_rate": 4.7520812266338885e-05, | |
| "loss": 1.0904, | |
| "step": 108 | |
| }, | |
| { | |
| "epoch": 0.42703232125367285, | |
| "grad_norm": 2.053180456161499, | |
| "learning_rate": 4.669547078371504e-05, | |
| "loss": 1.2116, | |
| "step": 109 | |
| }, | |
| { | |
| "epoch": 0.4309500489715965, | |
| "grad_norm": 8.551095008850098, | |
| "learning_rate": 4.5871032726383386e-05, | |
| "loss": 1.1253, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.4348677766895201, | |
| "grad_norm": 1.8820927143096924, | |
| "learning_rate": 4.504772348747687e-05, | |
| "loss": 0.9762, | |
| "step": 111 | |
| }, | |
| { | |
| "epoch": 0.4387855044074437, | |
| "grad_norm": 2.170362949371338, | |
| "learning_rate": 4.4225768151520694e-05, | |
| "loss": 1.0922, | |
| "step": 112 | |
| }, | |
| { | |
| "epoch": 0.4427032321253673, | |
| "grad_norm": 1.9470332860946655, | |
| "learning_rate": 4.3405391432896555e-05, | |
| "loss": 0.964, | |
| "step": 113 | |
| }, | |
| { | |
| "epoch": 0.44662095984329087, | |
| "grad_norm": 1.9972456693649292, | |
| "learning_rate": 4.2586817614407895e-05, | |
| "loss": 1.2283, | |
| "step": 114 | |
| }, | |
| { | |
| "epoch": 0.4505386875612145, | |
| "grad_norm": 1.8343126773834229, | |
| "learning_rate": 4.17702704859633e-05, | |
| "loss": 0.9953, | |
| "step": 115 | |
| }, | |
| { | |
| "epoch": 0.4544564152791381, | |
| "grad_norm": 1.8403739929199219, | |
| "learning_rate": 4.095597328339452e-05, | |
| "loss": 0.9655, | |
| "step": 116 | |
| }, | |
| { | |
| "epoch": 0.4583741429970617, | |
| "grad_norm": 2.1382968425750732, | |
| "learning_rate": 4.0144148627425993e-05, | |
| "loss": 1.1016, | |
| "step": 117 | |
| }, | |
| { | |
| "epoch": 0.4622918707149853, | |
| "grad_norm": 1.901031255722046, | |
| "learning_rate": 3.933501846281267e-05, | |
| "loss": 1.0008, | |
| "step": 118 | |
| }, | |
| { | |
| "epoch": 0.46620959843290893, | |
| "grad_norm": 1.9865580797195435, | |
| "learning_rate": 3.852880399766243e-05, | |
| "loss": 1.0154, | |
| "step": 119 | |
| }, | |
| { | |
| "epoch": 0.4701273261508325, | |
| "grad_norm": 1.8943233489990234, | |
| "learning_rate": 3.772572564296005e-05, | |
| "loss": 0.8704, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.4740450538687561, | |
| "grad_norm": 2.244124174118042, | |
| "learning_rate": 3.6926002952309016e-05, | |
| "loss": 0.971, | |
| "step": 121 | |
| }, | |
| { | |
| "epoch": 0.4779627815866797, | |
| "grad_norm": 2.311457633972168, | |
| "learning_rate": 3.612985456190778e-05, | |
| "loss": 1.2189, | |
| "step": 122 | |
| }, | |
| { | |
| "epoch": 0.48188050930460335, | |
| "grad_norm": 1.9182300567626953, | |
| "learning_rate": 3.533749813077677e-05, | |
| "loss": 0.8777, | |
| "step": 123 | |
| }, | |
| { | |
| "epoch": 0.48579823702252695, | |
| "grad_norm": 2.1501009464263916, | |
| "learning_rate": 3.4549150281252636e-05, | |
| "loss": 0.9504, | |
| "step": 124 | |
| }, | |
| { | |
| "epoch": 0.48971596474045054, | |
| "grad_norm": 1.9254357814788818, | |
| "learning_rate": 3.3765026539765834e-05, | |
| "loss": 0.8981, | |
| "step": 125 | |
| }, | |
| { | |
| "epoch": 0.49363369245837413, | |
| "grad_norm": 2.3888661861419678, | |
| "learning_rate": 3.298534127791785e-05, | |
| "loss": 1.2164, | |
| "step": 126 | |
| }, | |
| { | |
| "epoch": 0.4975514201762977, | |
| "grad_norm": 1.9912840127944946, | |
| "learning_rate": 3.221030765387417e-05, | |
| "loss": 1.0323, | |
| "step": 127 | |
| }, | |
| { | |
| "epoch": 0.5014691478942214, | |
| "grad_norm": 2.0007553100585938, | |
| "learning_rate": 3.144013755408895e-05, | |
| "loss": 1.0059, | |
| "step": 128 | |
| }, | |
| { | |
| "epoch": 0.5053868756121449, | |
| "grad_norm": 2.337552309036255, | |
| "learning_rate": 3.0675041535377405e-05, | |
| "loss": 1.1343, | |
| "step": 129 | |
| }, | |
| { | |
| "epoch": 0.5093046033300686, | |
| "grad_norm": 2.038306951522827, | |
| "learning_rate": 2.991522876735154e-05, | |
| "loss": 0.8762, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.5132223310479922, | |
| "grad_norm": 2.274177074432373, | |
| "learning_rate": 2.916090697523549e-05, | |
| "loss": 0.9796, | |
| "step": 131 | |
| }, | |
| { | |
| "epoch": 0.5171400587659157, | |
| "grad_norm": 2.1572694778442383, | |
| "learning_rate": 2.8412282383075363e-05, | |
| "loss": 0.7931, | |
| "step": 132 | |
| }, | |
| { | |
| "epoch": 0.5210577864838394, | |
| "grad_norm": 2.419728994369507, | |
| "learning_rate": 2.766955965735968e-05, | |
| "loss": 0.8772, | |
| "step": 133 | |
| }, | |
| { | |
| "epoch": 0.524975514201763, | |
| "grad_norm": 2.1187241077423096, | |
| "learning_rate": 2.693294185106562e-05, | |
| "loss": 0.9569, | |
| "step": 134 | |
| }, | |
| { | |
| "epoch": 0.5288932419196866, | |
| "grad_norm": 2.109083890914917, | |
| "learning_rate": 2.6202630348146324e-05, | |
| "loss": 0.9681, | |
| "step": 135 | |
| }, | |
| { | |
| "epoch": 0.5328109696376102, | |
| "grad_norm": 2.372102975845337, | |
| "learning_rate": 2.547882480847461e-05, | |
| "loss": 0.9457, | |
| "step": 136 | |
| }, | |
| { | |
| "epoch": 0.5367286973555337, | |
| "grad_norm": 2.407362699508667, | |
| "learning_rate": 2.476172311325783e-05, | |
| "loss": 0.9488, | |
| "step": 137 | |
| }, | |
| { | |
| "epoch": 0.5406464250734574, | |
| "grad_norm": 2.4459962844848633, | |
| "learning_rate": 2.405152131093926e-05, | |
| "loss": 0.9936, | |
| "step": 138 | |
| }, | |
| { | |
| "epoch": 0.544564152791381, | |
| "grad_norm": 2.218838930130005, | |
| "learning_rate": 2.3348413563600325e-05, | |
| "loss": 0.9295, | |
| "step": 139 | |
| }, | |
| { | |
| "epoch": 0.5484818805093046, | |
| "grad_norm": 2.2406013011932373, | |
| "learning_rate": 2.2652592093878666e-05, | |
| "loss": 1.0423, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.5523996082272282, | |
| "grad_norm": 2.0880637168884277, | |
| "learning_rate": 2.196424713241637e-05, | |
| "loss": 0.8297, | |
| "step": 141 | |
| }, | |
| { | |
| "epoch": 0.5563173359451518, | |
| "grad_norm": 2.784471273422241, | |
| "learning_rate": 2.128356686585282e-05, | |
| "loss": 0.9793, | |
| "step": 142 | |
| }, | |
| { | |
| "epoch": 0.5602350636630754, | |
| "grad_norm": 2.300804853439331, | |
| "learning_rate": 2.061073738537635e-05, | |
| "loss": 0.9287, | |
| "step": 143 | |
| }, | |
| { | |
| "epoch": 0.564152791380999, | |
| "grad_norm": 2.6218957901000977, | |
| "learning_rate": 1.9945942635848748e-05, | |
| "loss": 0.8821, | |
| "step": 144 | |
| }, | |
| { | |
| "epoch": 0.5680705190989226, | |
| "grad_norm": 2.166187047958374, | |
| "learning_rate": 1.928936436551661e-05, | |
| "loss": 0.8848, | |
| "step": 145 | |
| }, | |
| { | |
| "epoch": 0.5719882468168462, | |
| "grad_norm": 2.3030903339385986, | |
| "learning_rate": 1.8641182076323148e-05, | |
| "loss": 0.803, | |
| "step": 146 | |
| }, | |
| { | |
| "epoch": 0.5759059745347699, | |
| "grad_norm": 2.4375343322753906, | |
| "learning_rate": 1.800157297483417e-05, | |
| "loss": 0.8343, | |
| "step": 147 | |
| }, | |
| { | |
| "epoch": 0.5798237022526934, | |
| "grad_norm": 2.1112802028656006, | |
| "learning_rate": 1.7370711923791567e-05, | |
| "loss": 0.678, | |
| "step": 148 | |
| }, | |
| { | |
| "epoch": 0.5837414299706171, | |
| "grad_norm": 2.634140729904175, | |
| "learning_rate": 1.6748771394307585e-05, | |
| "loss": 0.7468, | |
| "step": 149 | |
| }, | |
| { | |
| "epoch": 0.5876591576885406, | |
| "grad_norm": 2.989452362060547, | |
| "learning_rate": 1.6135921418712956e-05, | |
| "loss": 0.7439, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.5876591576885406, | |
| "eval_loss": 0.24679341912269592, | |
| "eval_runtime": 31.6625, | |
| "eval_samples_per_second": 13.581, | |
| "eval_steps_per_second": 3.411, | |
| "step": 150 | |
| } | |
| ], | |
| "logging_steps": 1, | |
| "max_steps": 200, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 1, | |
| "save_steps": 50, | |
| "stateful_callbacks": { | |
| "EarlyStoppingCallback": { | |
| "args": { | |
| "early_stopping_patience": 5, | |
| "early_stopping_threshold": 0.0 | |
| }, | |
| "attributes": { | |
| "early_stopping_patience_counter": 0 | |
| } | |
| }, | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 2.147424726417408e+17, | |
| "train_batch_size": 8, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |