Instructions to use cimol/a42dd542-bf68-4fe7-a1e2-05d672b8be5f with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use cimol/a42dd542-bf68-4fe7-a1e2-05d672b8be5f with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen2.5-Coder-7B-Instruct") model = PeftModel.from_pretrained(base_model, "cimol/a42dd542-bf68-4fe7-a1e2-05d672b8be5f") - Notebooks
- Google Colab
- Kaggle
Download last-checkpoint/trainer_state.json from cimol/a42dd542-bf68-4fe7-a1e2-05d672b8be5f: direct link, hf CLI and curl.
- Browser
- Download file 36.6 kB
-
https://huggingface.co/cimol/a42dd542-bf68-4fe7-a1e2-05d672b8be5f/resolve/main/last-checkpoint/trainer_state.json
- Command line
-
hf download hf://cimol/a42dd542-bf68-4fe7-a1e2-05d672b8be5f/last-checkpoint/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/cimol/a42dd542-bf68-4fe7-a1e2-05d672b8be5f/resolve/main/last-checkpoint/trainer_state.json
36.6 kB
| { | |
| "best_metric": 2.171226978302002, | |
| "best_model_checkpoint": "miner_id_24/checkpoint-200", | |
| "epoch": 0.13710368466152528, | |
| "eval_steps": 50, | |
| "global_step": 200, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.0006855184233076263, | |
| "grad_norm": 0.4368946850299835, | |
| "learning_rate": 7e-06, | |
| "loss": 2.3639, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.0006855184233076263, | |
| "eval_loss": 2.760465145111084, | |
| "eval_runtime": 174.3986, | |
| "eval_samples_per_second": 14.088, | |
| "eval_steps_per_second": 3.526, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.0013710368466152527, | |
| "grad_norm": 0.47925180196762085, | |
| "learning_rate": 1.4e-05, | |
| "loss": 2.4098, | |
| "step": 2 | |
| }, | |
| { | |
| "epoch": 0.002056555269922879, | |
| "grad_norm": 0.537840723991394, | |
| "learning_rate": 2.1e-05, | |
| "loss": 2.5733, | |
| "step": 3 | |
| }, | |
| { | |
| "epoch": 0.0027420736932305054, | |
| "grad_norm": 0.5270087122917175, | |
| "learning_rate": 2.8e-05, | |
| "loss": 2.5575, | |
| "step": 4 | |
| }, | |
| { | |
| "epoch": 0.003427592116538132, | |
| "grad_norm": 0.5146782994270325, | |
| "learning_rate": 3.5e-05, | |
| "loss": 2.5588, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.004113110539845758, | |
| "grad_norm": 0.5258235931396484, | |
| "learning_rate": 4.2e-05, | |
| "loss": 2.5312, | |
| "step": 6 | |
| }, | |
| { | |
| "epoch": 0.004798628963153384, | |
| "grad_norm": 0.47052979469299316, | |
| "learning_rate": 4.899999999999999e-05, | |
| "loss": 2.4014, | |
| "step": 7 | |
| }, | |
| { | |
| "epoch": 0.005484147386461011, | |
| "grad_norm": 0.4933658838272095, | |
| "learning_rate": 5.6e-05, | |
| "loss": 2.5448, | |
| "step": 8 | |
| }, | |
| { | |
| "epoch": 0.006169665809768638, | |
| "grad_norm": 0.43804389238357544, | |
| "learning_rate": 6.3e-05, | |
| "loss": 2.3859, | |
| "step": 9 | |
| }, | |
| { | |
| "epoch": 0.006855184233076264, | |
| "grad_norm": 0.41613465547561646, | |
| "learning_rate": 7e-05, | |
| "loss": 2.3303, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.007540702656383891, | |
| "grad_norm": 0.44592195749282837, | |
| "learning_rate": 6.999521567473641e-05, | |
| "loss": 2.3207, | |
| "step": 11 | |
| }, | |
| { | |
| "epoch": 0.008226221079691516, | |
| "grad_norm": 0.5341413617134094, | |
| "learning_rate": 6.998086400693241e-05, | |
| "loss": 2.2716, | |
| "step": 12 | |
| }, | |
| { | |
| "epoch": 0.008911739502999142, | |
| "grad_norm": 0.5610725283622742, | |
| "learning_rate": 6.995694892019065e-05, | |
| "loss": 2.4044, | |
| "step": 13 | |
| }, | |
| { | |
| "epoch": 0.009597257926306769, | |
| "grad_norm": 0.5976554155349731, | |
| "learning_rate": 6.99234769526571e-05, | |
| "loss": 2.3527, | |
| "step": 14 | |
| }, | |
| { | |
| "epoch": 0.010282776349614395, | |
| "grad_norm": 0.5752586722373962, | |
| "learning_rate": 6.988045725523343e-05, | |
| "loss": 2.3368, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.010968294772922021, | |
| "grad_norm": 0.5358518958091736, | |
| "learning_rate": 6.982790158907539e-05, | |
| "loss": 2.2973, | |
| "step": 16 | |
| }, | |
| { | |
| "epoch": 0.011653813196229648, | |
| "grad_norm": 0.5135672688484192, | |
| "learning_rate": 6.976582432237733e-05, | |
| "loss": 2.2767, | |
| "step": 17 | |
| }, | |
| { | |
| "epoch": 0.012339331619537276, | |
| "grad_norm": 0.5366981029510498, | |
| "learning_rate": 6.969424242644413e-05, | |
| "loss": 2.3117, | |
| "step": 18 | |
| }, | |
| { | |
| "epoch": 0.013024850042844902, | |
| "grad_norm": 0.5566994547843933, | |
| "learning_rate": 6.961317547105138e-05, | |
| "loss": 2.1326, | |
| "step": 19 | |
| }, | |
| { | |
| "epoch": 0.013710368466152529, | |
| "grad_norm": 0.5392957329750061, | |
| "learning_rate": 6.952264561909527e-05, | |
| "loss": 2.3174, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.014395886889460155, | |
| "grad_norm": 0.6095602512359619, | |
| "learning_rate": 6.942267762053337e-05, | |
| "loss": 2.3736, | |
| "step": 21 | |
| }, | |
| { | |
| "epoch": 0.015081405312767781, | |
| "grad_norm": 0.56801837682724, | |
| "learning_rate": 6.931329880561832e-05, | |
| "loss": 2.2813, | |
| "step": 22 | |
| }, | |
| { | |
| "epoch": 0.015766923736075408, | |
| "grad_norm": 0.5793257355690002, | |
| "learning_rate": 6.919453907742597e-05, | |
| "loss": 2.2916, | |
| "step": 23 | |
| }, | |
| { | |
| "epoch": 0.016452442159383032, | |
| "grad_norm": 0.601382315158844, | |
| "learning_rate": 6.90664309036802e-05, | |
| "loss": 2.2621, | |
| "step": 24 | |
| }, | |
| { | |
| "epoch": 0.01713796058269066, | |
| "grad_norm": 0.5628184676170349, | |
| "learning_rate": 6.892900930787656e-05, | |
| "loss": 2.3218, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 0.017823479005998285, | |
| "grad_norm": 0.5984280109405518, | |
| "learning_rate": 6.87823118597072e-05, | |
| "loss": 2.1779, | |
| "step": 26 | |
| }, | |
| { | |
| "epoch": 0.018508997429305913, | |
| "grad_norm": 0.6334589719772339, | |
| "learning_rate": 6.862637866478969e-05, | |
| "loss": 2.1783, | |
| "step": 27 | |
| }, | |
| { | |
| "epoch": 0.019194515852613538, | |
| "grad_norm": 0.6164846420288086, | |
| "learning_rate": 6.846125235370252e-05, | |
| "loss": 2.2444, | |
| "step": 28 | |
| }, | |
| { | |
| "epoch": 0.019880034275921166, | |
| "grad_norm": 0.6727099418640137, | |
| "learning_rate": 6.828697807033038e-05, | |
| "loss": 2.3452, | |
| "step": 29 | |
| }, | |
| { | |
| "epoch": 0.02056555269922879, | |
| "grad_norm": 0.7243219614028931, | |
| "learning_rate": 6.81036034595222e-05, | |
| "loss": 2.4162, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.02125107112253642, | |
| "grad_norm": 0.6986833214759827, | |
| "learning_rate": 6.791117865406564e-05, | |
| "loss": 2.3235, | |
| "step": 31 | |
| }, | |
| { | |
| "epoch": 0.021936589545844043, | |
| "grad_norm": 0.7440876364707947, | |
| "learning_rate": 6.770975626098112e-05, | |
| "loss": 2.2856, | |
| "step": 32 | |
| }, | |
| { | |
| "epoch": 0.02262210796915167, | |
| "grad_norm": 0.7108097076416016, | |
| "learning_rate": 6.749939134713974e-05, | |
| "loss": 2.1324, | |
| "step": 33 | |
| }, | |
| { | |
| "epoch": 0.023307626392459296, | |
| "grad_norm": 0.740744411945343, | |
| "learning_rate": 6.728014142420846e-05, | |
| "loss": 2.2058, | |
| "step": 34 | |
| }, | |
| { | |
| "epoch": 0.023993144815766924, | |
| "grad_norm": 0.767447829246521, | |
| "learning_rate": 6.7052066432927e-05, | |
| "loss": 2.3296, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 0.024678663239074552, | |
| "grad_norm": 0.8943948745727539, | |
| "learning_rate": 6.681522872672069e-05, | |
| "loss": 2.5074, | |
| "step": 36 | |
| }, | |
| { | |
| "epoch": 0.025364181662382176, | |
| "grad_norm": 0.8490576148033142, | |
| "learning_rate": 6.656969305465356e-05, | |
| "loss": 2.3597, | |
| "step": 37 | |
| }, | |
| { | |
| "epoch": 0.026049700085689804, | |
| "grad_norm": 0.815835177898407, | |
| "learning_rate": 6.631552654372672e-05, | |
| "loss": 2.3154, | |
| "step": 38 | |
| }, | |
| { | |
| "epoch": 0.02673521850899743, | |
| "grad_norm": 0.877070963382721, | |
| "learning_rate": 6.60527986805264e-05, | |
| "loss": 2.3273, | |
| "step": 39 | |
| }, | |
| { | |
| "epoch": 0.027420736932305057, | |
| "grad_norm": 1.040402889251709, | |
| "learning_rate": 6.578158129222711e-05, | |
| "loss": 2.3329, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.028106255355612682, | |
| "grad_norm": 0.9905269742012024, | |
| "learning_rate": 6.550194852695469e-05, | |
| "loss": 2.2696, | |
| "step": 41 | |
| }, | |
| { | |
| "epoch": 0.02879177377892031, | |
| "grad_norm": 1.0185683965682983, | |
| "learning_rate": 6.521397683351509e-05, | |
| "loss": 2.1854, | |
| "step": 42 | |
| }, | |
| { | |
| "epoch": 0.029477292202227934, | |
| "grad_norm": 1.1009654998779297, | |
| "learning_rate": 6.491774494049386e-05, | |
| "loss": 2.3268, | |
| "step": 43 | |
| }, | |
| { | |
| "epoch": 0.030162810625535563, | |
| "grad_norm": 1.1542255878448486, | |
| "learning_rate": 6.461333383473272e-05, | |
| "loss": 2.1718, | |
| "step": 44 | |
| }, | |
| { | |
| "epoch": 0.030848329048843187, | |
| "grad_norm": 1.752344012260437, | |
| "learning_rate": 6.430082673918849e-05, | |
| "loss": 2.0563, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 0.031533847472150815, | |
| "grad_norm": 1.4921183586120605, | |
| "learning_rate": 6.398030909018069e-05, | |
| "loss": 2.4322, | |
| "step": 46 | |
| }, | |
| { | |
| "epoch": 0.03221936589545844, | |
| "grad_norm": 1.5160636901855469, | |
| "learning_rate": 6.365186851403423e-05, | |
| "loss": 2.2883, | |
| "step": 47 | |
| }, | |
| { | |
| "epoch": 0.032904884318766064, | |
| "grad_norm": 1.7056834697723389, | |
| "learning_rate": 6.331559480312315e-05, | |
| "loss": 2.2562, | |
| "step": 48 | |
| }, | |
| { | |
| "epoch": 0.03359040274207369, | |
| "grad_norm": 1.8772844076156616, | |
| "learning_rate": 6.297157989132236e-05, | |
| "loss": 2.1196, | |
| "step": 49 | |
| }, | |
| { | |
| "epoch": 0.03427592116538132, | |
| "grad_norm": 2.8052427768707275, | |
| "learning_rate": 6.261991782887377e-05, | |
| "loss": 2.1031, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.03427592116538132, | |
| "eval_loss": 2.4464125633239746, | |
| "eval_runtime": 175.5835, | |
| "eval_samples_per_second": 13.993, | |
| "eval_steps_per_second": 3.503, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.03496143958868895, | |
| "grad_norm": 1.1087968349456787, | |
| "learning_rate": 6.226070475667393e-05, | |
| "loss": 2.2777, | |
| "step": 51 | |
| }, | |
| { | |
| "epoch": 0.03564695801199657, | |
| "grad_norm": 1.1224277019500732, | |
| "learning_rate": 6.189403887999006e-05, | |
| "loss": 2.4869, | |
| "step": 52 | |
| }, | |
| { | |
| "epoch": 0.0363324764353042, | |
| "grad_norm": 0.9294334650039673, | |
| "learning_rate": 6.152002044161171e-05, | |
| "loss": 2.4066, | |
| "step": 53 | |
| }, | |
| { | |
| "epoch": 0.037017994858611826, | |
| "grad_norm": 0.7608312964439392, | |
| "learning_rate": 6.113875169444539e-05, | |
| "loss": 2.3264, | |
| "step": 54 | |
| }, | |
| { | |
| "epoch": 0.037703513281919454, | |
| "grad_norm": 0.5293958187103271, | |
| "learning_rate": 6.0750336873559605e-05, | |
| "loss": 2.2976, | |
| "step": 55 | |
| }, | |
| { | |
| "epoch": 0.038389031705227075, | |
| "grad_norm": 0.5001978874206543, | |
| "learning_rate": 6.035488216768811e-05, | |
| "loss": 2.2571, | |
| "step": 56 | |
| }, | |
| { | |
| "epoch": 0.0390745501285347, | |
| "grad_norm": 0.49879032373428345, | |
| "learning_rate": 5.9952495690198894e-05, | |
| "loss": 2.2304, | |
| "step": 57 | |
| }, | |
| { | |
| "epoch": 0.03976006855184233, | |
| "grad_norm": 0.5498774647712708, | |
| "learning_rate": 5.954328744953709e-05, | |
| "loss": 2.2506, | |
| "step": 58 | |
| }, | |
| { | |
| "epoch": 0.04044558697514996, | |
| "grad_norm": 0.5348244905471802, | |
| "learning_rate": 5.91273693191498e-05, | |
| "loss": 2.2631, | |
| "step": 59 | |
| }, | |
| { | |
| "epoch": 0.04113110539845758, | |
| "grad_norm": 0.5236966013908386, | |
| "learning_rate": 5.870485500690094e-05, | |
| "loss": 2.2917, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.04181662382176521, | |
| "grad_norm": 0.5978161096572876, | |
| "learning_rate": 5.827586002398468e-05, | |
| "loss": 2.17, | |
| "step": 61 | |
| }, | |
| { | |
| "epoch": 0.04250214224507284, | |
| "grad_norm": 0.522532045841217, | |
| "learning_rate": 5.784050165334589e-05, | |
| "loss": 2.2009, | |
| "step": 62 | |
| }, | |
| { | |
| "epoch": 0.043187660668380465, | |
| "grad_norm": 0.5356420874595642, | |
| "learning_rate": 5.739889891761608e-05, | |
| "loss": 2.2726, | |
| "step": 63 | |
| }, | |
| { | |
| "epoch": 0.043873179091688086, | |
| "grad_norm": 0.5630861520767212, | |
| "learning_rate": 5.6951172546573794e-05, | |
| "loss": 2.3207, | |
| "step": 64 | |
| }, | |
| { | |
| "epoch": 0.044558697514995714, | |
| "grad_norm": 0.5354762673377991, | |
| "learning_rate": 5.6497444944138376e-05, | |
| "loss": 2.2271, | |
| "step": 65 | |
| }, | |
| { | |
| "epoch": 0.04524421593830334, | |
| "grad_norm": 0.5601297616958618, | |
| "learning_rate": 5.603784015490587e-05, | |
| "loss": 2.096, | |
| "step": 66 | |
| }, | |
| { | |
| "epoch": 0.04592973436161097, | |
| "grad_norm": 0.5554521083831787, | |
| "learning_rate": 5.557248383023655e-05, | |
| "loss": 2.1702, | |
| "step": 67 | |
| }, | |
| { | |
| "epoch": 0.04661525278491859, | |
| "grad_norm": 0.5609880089759827, | |
| "learning_rate": 5.510150319390302e-05, | |
| "loss": 2.2266, | |
| "step": 68 | |
| }, | |
| { | |
| "epoch": 0.04730077120822622, | |
| "grad_norm": 0.621530294418335, | |
| "learning_rate": 5.4625027007308546e-05, | |
| "loss": 2.2195, | |
| "step": 69 | |
| }, | |
| { | |
| "epoch": 0.04798628963153385, | |
| "grad_norm": 0.590848982334137, | |
| "learning_rate": 5.414318553428494e-05, | |
| "loss": 2.234, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.048671808054841476, | |
| "grad_norm": 0.6129457354545593, | |
| "learning_rate": 5.3656110505479776e-05, | |
| "loss": 2.1827, | |
| "step": 71 | |
| }, | |
| { | |
| "epoch": 0.049357326478149104, | |
| "grad_norm": 0.6404921412467957, | |
| "learning_rate": 5.316393508234253e-05, | |
| "loss": 2.1911, | |
| "step": 72 | |
| }, | |
| { | |
| "epoch": 0.050042844901456725, | |
| "grad_norm": 0.6630199551582336, | |
| "learning_rate": 5.266679382071953e-05, | |
| "loss": 2.1965, | |
| "step": 73 | |
| }, | |
| { | |
| "epoch": 0.05072836332476435, | |
| "grad_norm": 0.6906275749206543, | |
| "learning_rate": 5.216482263406778e-05, | |
| "loss": 2.2031, | |
| "step": 74 | |
| }, | |
| { | |
| "epoch": 0.05141388174807198, | |
| "grad_norm": 0.6980186700820923, | |
| "learning_rate": 5.1658158756297576e-05, | |
| "loss": 2.3229, | |
| "step": 75 | |
| }, | |
| { | |
| "epoch": 0.05209940017137961, | |
| "grad_norm": 0.7447901368141174, | |
| "learning_rate": 5.114694070425407e-05, | |
| "loss": 2.1765, | |
| "step": 76 | |
| }, | |
| { | |
| "epoch": 0.05278491859468723, | |
| "grad_norm": 0.7455515265464783, | |
| "learning_rate": 5.063130823984823e-05, | |
| "loss": 2.3248, | |
| "step": 77 | |
| }, | |
| { | |
| "epoch": 0.05347043701799486, | |
| "grad_norm": 0.7050088047981262, | |
| "learning_rate": 5.011140233184724e-05, | |
| "loss": 2.1526, | |
| "step": 78 | |
| }, | |
| { | |
| "epoch": 0.054155955441302486, | |
| "grad_norm": 0.7368468642234802, | |
| "learning_rate": 4.958736511733516e-05, | |
| "loss": 2.238, | |
| "step": 79 | |
| }, | |
| { | |
| "epoch": 0.054841473864610114, | |
| "grad_norm": 0.7982569932937622, | |
| "learning_rate": 4.905933986285393e-05, | |
| "loss": 2.2873, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.055526992287917736, | |
| "grad_norm": 0.8542561531066895, | |
| "learning_rate": 4.8527470925235824e-05, | |
| "loss": 2.3207, | |
| "step": 81 | |
| }, | |
| { | |
| "epoch": 0.056212510711225364, | |
| "grad_norm": 0.8338145017623901, | |
| "learning_rate": 4.799190371213772e-05, | |
| "loss": 2.2727, | |
| "step": 82 | |
| }, | |
| { | |
| "epoch": 0.05689802913453299, | |
| "grad_norm": 0.8469993472099304, | |
| "learning_rate": 4.745278464228808e-05, | |
| "loss": 2.2099, | |
| "step": 83 | |
| }, | |
| { | |
| "epoch": 0.05758354755784062, | |
| "grad_norm": 0.9120200872421265, | |
| "learning_rate": 4.69102611054575e-05, | |
| "loss": 2.3549, | |
| "step": 84 | |
| }, | |
| { | |
| "epoch": 0.05826906598114824, | |
| "grad_norm": 0.9229581356048584, | |
| "learning_rate": 4.6364481422163926e-05, | |
| "loss": 2.3835, | |
| "step": 85 | |
| }, | |
| { | |
| "epoch": 0.05895458440445587, | |
| "grad_norm": 0.9590667486190796, | |
| "learning_rate": 4.581559480312316e-05, | |
| "loss": 2.3397, | |
| "step": 86 | |
| }, | |
| { | |
| "epoch": 0.0596401028277635, | |
| "grad_norm": 0.9578183889389038, | |
| "learning_rate": 4.526375130845627e-05, | |
| "loss": 2.179, | |
| "step": 87 | |
| }, | |
| { | |
| "epoch": 0.060325621251071125, | |
| "grad_norm": 1.023118019104004, | |
| "learning_rate": 4.4709101806664554e-05, | |
| "loss": 2.1585, | |
| "step": 88 | |
| }, | |
| { | |
| "epoch": 0.061011139674378746, | |
| "grad_norm": 1.091235876083374, | |
| "learning_rate": 4.4151797933383685e-05, | |
| "loss": 2.3241, | |
| "step": 89 | |
| }, | |
| { | |
| "epoch": 0.061696658097686374, | |
| "grad_norm": 1.0647417306900024, | |
| "learning_rate": 4.359199204992797e-05, | |
| "loss": 2.2472, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.062382176520994, | |
| "grad_norm": 1.1701401472091675, | |
| "learning_rate": 4.30298372016363e-05, | |
| "loss": 2.2542, | |
| "step": 91 | |
| }, | |
| { | |
| "epoch": 0.06306769494430163, | |
| "grad_norm": 1.0854402780532837, | |
| "learning_rate": 4.246548707603114e-05, | |
| "loss": 1.9001, | |
| "step": 92 | |
| }, | |
| { | |
| "epoch": 0.06375321336760925, | |
| "grad_norm": 1.2632555961608887, | |
| "learning_rate": 4.1899095960801805e-05, | |
| "loss": 2.338, | |
| "step": 93 | |
| }, | |
| { | |
| "epoch": 0.06443873179091689, | |
| "grad_norm": 1.3740942478179932, | |
| "learning_rate": 4.133081870162385e-05, | |
| "loss": 2.1948, | |
| "step": 94 | |
| }, | |
| { | |
| "epoch": 0.06512425021422451, | |
| "grad_norm": 1.461464524269104, | |
| "learning_rate": 4.076081065982569e-05, | |
| "loss": 2.1339, | |
| "step": 95 | |
| }, | |
| { | |
| "epoch": 0.06580976863753213, | |
| "grad_norm": 1.557289719581604, | |
| "learning_rate": 4.018922766991447e-05, | |
| "loss": 2.2991, | |
| "step": 96 | |
| }, | |
| { | |
| "epoch": 0.06649528706083976, | |
| "grad_norm": 1.6103731393814087, | |
| "learning_rate": 3.961622599697241e-05, | |
| "loss": 2.2176, | |
| "step": 97 | |
| }, | |
| { | |
| "epoch": 0.06718080548414739, | |
| "grad_norm": 1.867081880569458, | |
| "learning_rate": 3.9041962293935516e-05, | |
| "loss": 1.9107, | |
| "step": 98 | |
| }, | |
| { | |
| "epoch": 0.067866323907455, | |
| "grad_norm": 1.955940842628479, | |
| "learning_rate": 3.84665935587662e-05, | |
| "loss": 2.034, | |
| "step": 99 | |
| }, | |
| { | |
| "epoch": 0.06855184233076264, | |
| "grad_norm": 2.633854866027832, | |
| "learning_rate": 3.7890277091531636e-05, | |
| "loss": 2.1461, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.06855184233076264, | |
| "eval_loss": 2.311368703842163, | |
| "eval_runtime": 175.9506, | |
| "eval_samples_per_second": 13.964, | |
| "eval_steps_per_second": 3.495, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.06923736075407026, | |
| "grad_norm": 1.004059910774231, | |
| "learning_rate": 3.7313170451399475e-05, | |
| "loss": 2.2891, | |
| "step": 101 | |
| }, | |
| { | |
| "epoch": 0.0699228791773779, | |
| "grad_norm": 0.9783998131752014, | |
| "learning_rate": 3.673543141356278e-05, | |
| "loss": 2.395, | |
| "step": 102 | |
| }, | |
| { | |
| "epoch": 0.07060839760068552, | |
| "grad_norm": 1.0185178518295288, | |
| "learning_rate": 3.6157217926105783e-05, | |
| "loss": 2.2837, | |
| "step": 103 | |
| }, | |
| { | |
| "epoch": 0.07129391602399314, | |
| "grad_norm": 0.9947238564491272, | |
| "learning_rate": 3.557868806682255e-05, | |
| "loss": 2.3518, | |
| "step": 104 | |
| }, | |
| { | |
| "epoch": 0.07197943444730077, | |
| "grad_norm": 0.9198320508003235, | |
| "learning_rate": 3.5e-05, | |
| "loss": 2.3111, | |
| "step": 105 | |
| }, | |
| { | |
| "epoch": 0.0726649528706084, | |
| "grad_norm": 0.9715070724487305, | |
| "learning_rate": 3.442131193317745e-05, | |
| "loss": 2.3173, | |
| "step": 106 | |
| }, | |
| { | |
| "epoch": 0.07335047129391603, | |
| "grad_norm": 0.8154056072235107, | |
| "learning_rate": 3.384278207389421e-05, | |
| "loss": 2.258, | |
| "step": 107 | |
| }, | |
| { | |
| "epoch": 0.07403598971722365, | |
| "grad_norm": 0.7482956647872925, | |
| "learning_rate": 3.3264568586437216e-05, | |
| "loss": 2.2164, | |
| "step": 108 | |
| }, | |
| { | |
| "epoch": 0.07472150814053127, | |
| "grad_norm": 0.6518887281417847, | |
| "learning_rate": 3.268682954860052e-05, | |
| "loss": 2.1861, | |
| "step": 109 | |
| }, | |
| { | |
| "epoch": 0.07540702656383891, | |
| "grad_norm": 0.5987632274627686, | |
| "learning_rate": 3.210972290846837e-05, | |
| "loss": 2.1645, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.07609254498714653, | |
| "grad_norm": 0.622978150844574, | |
| "learning_rate": 3.15334064412338e-05, | |
| "loss": 2.2316, | |
| "step": 111 | |
| }, | |
| { | |
| "epoch": 0.07677806341045415, | |
| "grad_norm": 0.6469138860702515, | |
| "learning_rate": 3.0958037706064485e-05, | |
| "loss": 2.119, | |
| "step": 112 | |
| }, | |
| { | |
| "epoch": 0.07746358183376179, | |
| "grad_norm": 0.6277498602867126, | |
| "learning_rate": 3.038377400302758e-05, | |
| "loss": 2.1299, | |
| "step": 113 | |
| }, | |
| { | |
| "epoch": 0.0781491002570694, | |
| "grad_norm": 0.6326158046722412, | |
| "learning_rate": 2.9810772330085524e-05, | |
| "loss": 2.0619, | |
| "step": 114 | |
| }, | |
| { | |
| "epoch": 0.07883461868037704, | |
| "grad_norm": 0.6318880915641785, | |
| "learning_rate": 2.9239189340174306e-05, | |
| "loss": 2.1612, | |
| "step": 115 | |
| }, | |
| { | |
| "epoch": 0.07952013710368466, | |
| "grad_norm": 0.6449387669563293, | |
| "learning_rate": 2.8669181298376163e-05, | |
| "loss": 2.2981, | |
| "step": 116 | |
| }, | |
| { | |
| "epoch": 0.08020565552699228, | |
| "grad_norm": 0.6613906621932983, | |
| "learning_rate": 2.8100904039198193e-05, | |
| "loss": 2.1068, | |
| "step": 117 | |
| }, | |
| { | |
| "epoch": 0.08089117395029992, | |
| "grad_norm": 0.6904773116111755, | |
| "learning_rate": 2.7534512923968863e-05, | |
| "loss": 2.1431, | |
| "step": 118 | |
| }, | |
| { | |
| "epoch": 0.08157669237360754, | |
| "grad_norm": 0.6582241058349609, | |
| "learning_rate": 2.6970162798363695e-05, | |
| "loss": 2.1572, | |
| "step": 119 | |
| }, | |
| { | |
| "epoch": 0.08226221079691516, | |
| "grad_norm": 0.6912741661071777, | |
| "learning_rate": 2.640800795007203e-05, | |
| "loss": 2.2768, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.0829477292202228, | |
| "grad_norm": 0.7034792900085449, | |
| "learning_rate": 2.5848202066616305e-05, | |
| "loss": 2.1116, | |
| "step": 121 | |
| }, | |
| { | |
| "epoch": 0.08363324764353042, | |
| "grad_norm": 0.6760836839675903, | |
| "learning_rate": 2.5290898193335446e-05, | |
| "loss": 2.1501, | |
| "step": 122 | |
| }, | |
| { | |
| "epoch": 0.08431876606683805, | |
| "grad_norm": 0.72305828332901, | |
| "learning_rate": 2.4736248691543736e-05, | |
| "loss": 2.2179, | |
| "step": 123 | |
| }, | |
| { | |
| "epoch": 0.08500428449014567, | |
| "grad_norm": 0.7351564168930054, | |
| "learning_rate": 2.4184405196876842e-05, | |
| "loss": 2.1684, | |
| "step": 124 | |
| }, | |
| { | |
| "epoch": 0.0856898029134533, | |
| "grad_norm": 0.7352508306503296, | |
| "learning_rate": 2.363551857783608e-05, | |
| "loss": 2.0446, | |
| "step": 125 | |
| }, | |
| { | |
| "epoch": 0.08637532133676093, | |
| "grad_norm": 0.7216020226478577, | |
| "learning_rate": 2.308973889454249e-05, | |
| "loss": 2.0774, | |
| "step": 126 | |
| }, | |
| { | |
| "epoch": 0.08706083976006855, | |
| "grad_norm": 0.7965324521064758, | |
| "learning_rate": 2.2547215357711918e-05, | |
| "loss": 2.2296, | |
| "step": 127 | |
| }, | |
| { | |
| "epoch": 0.08774635818337617, | |
| "grad_norm": 0.7469529509544373, | |
| "learning_rate": 2.2008096287862266e-05, | |
| "loss": 2.0849, | |
| "step": 128 | |
| }, | |
| { | |
| "epoch": 0.0884318766066838, | |
| "grad_norm": 0.8050830364227295, | |
| "learning_rate": 2.1472529074764177e-05, | |
| "loss": 2.2615, | |
| "step": 129 | |
| }, | |
| { | |
| "epoch": 0.08911739502999143, | |
| "grad_norm": 0.7781040668487549, | |
| "learning_rate": 2.0940660137146074e-05, | |
| "loss": 2.1893, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.08980291345329906, | |
| "grad_norm": 0.8017018437385559, | |
| "learning_rate": 2.041263488266484e-05, | |
| "loss": 2.1451, | |
| "step": 131 | |
| }, | |
| { | |
| "epoch": 0.09048843187660668, | |
| "grad_norm": 0.8338680267333984, | |
| "learning_rate": 1.988859766815275e-05, | |
| "loss": 2.1856, | |
| "step": 132 | |
| }, | |
| { | |
| "epoch": 0.0911739502999143, | |
| "grad_norm": 0.8552135825157166, | |
| "learning_rate": 1.9368691760151773e-05, | |
| "loss": 2.1477, | |
| "step": 133 | |
| }, | |
| { | |
| "epoch": 0.09185946872322194, | |
| "grad_norm": 0.8702215552330017, | |
| "learning_rate": 1.885305929574593e-05, | |
| "loss": 2.099, | |
| "step": 134 | |
| }, | |
| { | |
| "epoch": 0.09254498714652956, | |
| "grad_norm": 0.9290527105331421, | |
| "learning_rate": 1.8341841243702424e-05, | |
| "loss": 2.1671, | |
| "step": 135 | |
| }, | |
| { | |
| "epoch": 0.09323050556983718, | |
| "grad_norm": 0.9839473366737366, | |
| "learning_rate": 1.7835177365932225e-05, | |
| "loss": 2.3551, | |
| "step": 136 | |
| }, | |
| { | |
| "epoch": 0.09391602399314482, | |
| "grad_norm": 1.0397003889083862, | |
| "learning_rate": 1.7333206179280478e-05, | |
| "loss": 2.2355, | |
| "step": 137 | |
| }, | |
| { | |
| "epoch": 0.09460154241645244, | |
| "grad_norm": 0.9706122875213623, | |
| "learning_rate": 1.6836064917657478e-05, | |
| "loss": 2.2157, | |
| "step": 138 | |
| }, | |
| { | |
| "epoch": 0.09528706083976007, | |
| "grad_norm": 0.985498309135437, | |
| "learning_rate": 1.6343889494520224e-05, | |
| "loss": 2.1513, | |
| "step": 139 | |
| }, | |
| { | |
| "epoch": 0.0959725792630677, | |
| "grad_norm": 1.0459221601486206, | |
| "learning_rate": 1.5856814465715064e-05, | |
| "loss": 2.0888, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.09665809768637532, | |
| "grad_norm": 1.1182277202606201, | |
| "learning_rate": 1.5374972992691458e-05, | |
| "loss": 2.2288, | |
| "step": 141 | |
| }, | |
| { | |
| "epoch": 0.09734361610968295, | |
| "grad_norm": 1.1353912353515625, | |
| "learning_rate": 1.4898496806096974e-05, | |
| "loss": 2.1112, | |
| "step": 142 | |
| }, | |
| { | |
| "epoch": 0.09802913453299057, | |
| "grad_norm": 1.694881796836853, | |
| "learning_rate": 1.4427516169763444e-05, | |
| "loss": 2.229, | |
| "step": 143 | |
| }, | |
| { | |
| "epoch": 0.09871465295629821, | |
| "grad_norm": 1.2565925121307373, | |
| "learning_rate": 1.396215984509412e-05, | |
| "loss": 2.1144, | |
| "step": 144 | |
| }, | |
| { | |
| "epoch": 0.09940017137960583, | |
| "grad_norm": 1.3687129020690918, | |
| "learning_rate": 1.3502555055861625e-05, | |
| "loss": 2.0882, | |
| "step": 145 | |
| }, | |
| { | |
| "epoch": 0.10008568980291345, | |
| "grad_norm": 1.4564164876937866, | |
| "learning_rate": 1.3048827453426203e-05, | |
| "loss": 2.121, | |
| "step": 146 | |
| }, | |
| { | |
| "epoch": 0.10077120822622108, | |
| "grad_norm": 1.76753830909729, | |
| "learning_rate": 1.2601101082383917e-05, | |
| "loss": 2.2216, | |
| "step": 147 | |
| }, | |
| { | |
| "epoch": 0.1014567266495287, | |
| "grad_norm": 1.7149028778076172, | |
| "learning_rate": 1.2159498346654094e-05, | |
| "loss": 2.1153, | |
| "step": 148 | |
| }, | |
| { | |
| "epoch": 0.10214224507283633, | |
| "grad_norm": 1.9386048316955566, | |
| "learning_rate": 1.1724139976015306e-05, | |
| "loss": 2.046, | |
| "step": 149 | |
| }, | |
| { | |
| "epoch": 0.10282776349614396, | |
| "grad_norm": 2.64031720161438, | |
| "learning_rate": 1.1295144993099068e-05, | |
| "loss": 2.0785, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.10282776349614396, | |
| "eval_loss": 2.1849844455718994, | |
| "eval_runtime": 175.9852, | |
| "eval_samples_per_second": 13.961, | |
| "eval_steps_per_second": 3.495, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.10351328191945158, | |
| "grad_norm": 0.4782308042049408, | |
| "learning_rate": 1.0872630680850196e-05, | |
| "loss": 2.0454, | |
| "step": 151 | |
| }, | |
| { | |
| "epoch": 0.10419880034275922, | |
| "grad_norm": 0.5835824608802795, | |
| "learning_rate": 1.0456712550462898e-05, | |
| "loss": 2.1698, | |
| "step": 152 | |
| }, | |
| { | |
| "epoch": 0.10488431876606684, | |
| "grad_norm": 0.5484896898269653, | |
| "learning_rate": 1.0047504309801104e-05, | |
| "loss": 2.16, | |
| "step": 153 | |
| }, | |
| { | |
| "epoch": 0.10556983718937446, | |
| "grad_norm": 0.6225553154945374, | |
| "learning_rate": 9.645117832311886e-06, | |
| "loss": 2.2452, | |
| "step": 154 | |
| }, | |
| { | |
| "epoch": 0.1062553556126821, | |
| "grad_norm": 0.6657689213752747, | |
| "learning_rate": 9.249663126440394e-06, | |
| "loss": 2.2186, | |
| "step": 155 | |
| }, | |
| { | |
| "epoch": 0.10694087403598972, | |
| "grad_norm": 0.6836898922920227, | |
| "learning_rate": 8.861248305554624e-06, | |
| "loss": 2.1986, | |
| "step": 156 | |
| }, | |
| { | |
| "epoch": 0.10762639245929734, | |
| "grad_norm": 0.6431835293769836, | |
| "learning_rate": 8.47997955838829e-06, | |
| "loss": 2.253, | |
| "step": 157 | |
| }, | |
| { | |
| "epoch": 0.10831191088260497, | |
| "grad_norm": 0.7124049067497253, | |
| "learning_rate": 8.10596112000994e-06, | |
| "loss": 2.3133, | |
| "step": 158 | |
| }, | |
| { | |
| "epoch": 0.1089974293059126, | |
| "grad_norm": 0.6604753136634827, | |
| "learning_rate": 7.739295243326067e-06, | |
| "loss": 2.1704, | |
| "step": 159 | |
| }, | |
| { | |
| "epoch": 0.10968294772922023, | |
| "grad_norm": 0.6440089344978333, | |
| "learning_rate": 7.380082171126228e-06, | |
| "loss": 2.1077, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.11036846615252785, | |
| "grad_norm": 0.7280560731887817, | |
| "learning_rate": 7.028420108677635e-06, | |
| "loss": 2.2385, | |
| "step": 161 | |
| }, | |
| { | |
| "epoch": 0.11105398457583547, | |
| "grad_norm": 0.6944758892059326, | |
| "learning_rate": 6.684405196876842e-06, | |
| "loss": 2.2893, | |
| "step": 162 | |
| }, | |
| { | |
| "epoch": 0.1117395029991431, | |
| "grad_norm": 0.7794995903968811, | |
| "learning_rate": 6.3481314859657675e-06, | |
| "loss": 2.2727, | |
| "step": 163 | |
| }, | |
| { | |
| "epoch": 0.11242502142245073, | |
| "grad_norm": 0.7010950446128845, | |
| "learning_rate": 6.019690909819298e-06, | |
| "loss": 2.1636, | |
| "step": 164 | |
| }, | |
| { | |
| "epoch": 0.11311053984575835, | |
| "grad_norm": 0.7227156162261963, | |
| "learning_rate": 5.6991732608115e-06, | |
| "loss": 2.1365, | |
| "step": 165 | |
| }, | |
| { | |
| "epoch": 0.11379605826906598, | |
| "grad_norm": 0.7752686738967896, | |
| "learning_rate": 5.386666165267256e-06, | |
| "loss": 2.2113, | |
| "step": 166 | |
| }, | |
| { | |
| "epoch": 0.1144815766923736, | |
| "grad_norm": 0.7315618395805359, | |
| "learning_rate": 5.08225505950613e-06, | |
| "loss": 2.2679, | |
| "step": 167 | |
| }, | |
| { | |
| "epoch": 0.11516709511568124, | |
| "grad_norm": 0.7621932625770569, | |
| "learning_rate": 4.786023166484913e-06, | |
| "loss": 2.0931, | |
| "step": 168 | |
| }, | |
| { | |
| "epoch": 0.11585261353898886, | |
| "grad_norm": 0.7386559247970581, | |
| "learning_rate": 4.498051473045291e-06, | |
| "loss": 2.0621, | |
| "step": 169 | |
| }, | |
| { | |
| "epoch": 0.11653813196229648, | |
| "grad_norm": 0.7826411724090576, | |
| "learning_rate": 4.218418707772886e-06, | |
| "loss": 2.105, | |
| "step": 170 | |
| }, | |
| { | |
| "epoch": 0.11722365038560412, | |
| "grad_norm": 0.7282847166061401, | |
| "learning_rate": 3.947201319473587e-06, | |
| "loss": 2.1473, | |
| "step": 171 | |
| }, | |
| { | |
| "epoch": 0.11790916880891174, | |
| "grad_norm": 0.765394389629364, | |
| "learning_rate": 3.684473456273278e-06, | |
| "loss": 2.1247, | |
| "step": 172 | |
| }, | |
| { | |
| "epoch": 0.11859468723221936, | |
| "grad_norm": 0.7690032720565796, | |
| "learning_rate": 3.4303069453464383e-06, | |
| "loss": 2.221, | |
| "step": 173 | |
| }, | |
| { | |
| "epoch": 0.119280205655527, | |
| "grad_norm": 0.7550768852233887, | |
| "learning_rate": 3.184771273279312e-06, | |
| "loss": 2.1277, | |
| "step": 174 | |
| }, | |
| { | |
| "epoch": 0.11996572407883462, | |
| "grad_norm": 0.7658005952835083, | |
| "learning_rate": 2.947933567072987e-06, | |
| "loss": 2.1159, | |
| "step": 175 | |
| }, | |
| { | |
| "epoch": 0.12065124250214225, | |
| "grad_norm": 0.769408106803894, | |
| "learning_rate": 2.719858575791534e-06, | |
| "loss": 2.0933, | |
| "step": 176 | |
| }, | |
| { | |
| "epoch": 0.12133676092544987, | |
| "grad_norm": 0.791779637336731, | |
| "learning_rate": 2.500608652860256e-06, | |
| "loss": 2.1937, | |
| "step": 177 | |
| }, | |
| { | |
| "epoch": 0.12202227934875749, | |
| "grad_norm": 0.8091723918914795, | |
| "learning_rate": 2.2902437390188737e-06, | |
| "loss": 2.2312, | |
| "step": 178 | |
| }, | |
| { | |
| "epoch": 0.12270779777206513, | |
| "grad_norm": 0.9017377495765686, | |
| "learning_rate": 2.0888213459343587e-06, | |
| "loss": 2.1079, | |
| "step": 179 | |
| }, | |
| { | |
| "epoch": 0.12339331619537275, | |
| "grad_norm": 0.7710647583007812, | |
| "learning_rate": 1.8963965404777875e-06, | |
| "loss": 1.9247, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.12407883461868038, | |
| "grad_norm": 0.8581042289733887, | |
| "learning_rate": 1.7130219296696263e-06, | |
| "loss": 2.1444, | |
| "step": 181 | |
| }, | |
| { | |
| "epoch": 0.124764353041988, | |
| "grad_norm": 0.9161327481269836, | |
| "learning_rate": 1.5387476462974824e-06, | |
| "loss": 2.1693, | |
| "step": 182 | |
| }, | |
| { | |
| "epoch": 0.12544987146529563, | |
| "grad_norm": 0.8806943297386169, | |
| "learning_rate": 1.3736213352103147e-06, | |
| "loss": 2.1161, | |
| "step": 183 | |
| }, | |
| { | |
| "epoch": 0.12613538988860326, | |
| "grad_norm": 0.9642539620399475, | |
| "learning_rate": 1.2176881402928002e-06, | |
| "loss": 2.1925, | |
| "step": 184 | |
| }, | |
| { | |
| "epoch": 0.1268209083119109, | |
| "grad_norm": 0.9556791186332703, | |
| "learning_rate": 1.0709906921234367e-06, | |
| "loss": 2.1456, | |
| "step": 185 | |
| }, | |
| { | |
| "epoch": 0.1275064267352185, | |
| "grad_norm": 0.9510772824287415, | |
| "learning_rate": 9.33569096319799e-07, | |
| "loss": 2.2513, | |
| "step": 186 | |
| }, | |
| { | |
| "epoch": 0.12819194515852614, | |
| "grad_norm": 1.027150273323059, | |
| "learning_rate": 8.054609225740255e-07, | |
| "loss": 2.2014, | |
| "step": 187 | |
| }, | |
| { | |
| "epoch": 0.12887746358183377, | |
| "grad_norm": 1.0660666227340698, | |
| "learning_rate": 6.867011943816724e-07, | |
| "loss": 2.1472, | |
| "step": 188 | |
| }, | |
| { | |
| "epoch": 0.12956298200514138, | |
| "grad_norm": 1.01221764087677, | |
| "learning_rate": 5.77322379466617e-07, | |
| "loss": 2.2024, | |
| "step": 189 | |
| }, | |
| { | |
| "epoch": 0.13024850042844902, | |
| "grad_norm": 1.144607424736023, | |
| "learning_rate": 4.773543809047186e-07, | |
| "loss": 2.1921, | |
| "step": 190 | |
| }, | |
| { | |
| "epoch": 0.13093401885175665, | |
| "grad_norm": 1.1356172561645508, | |
| "learning_rate": 3.868245289486027e-07, | |
| "loss": 2.1876, | |
| "step": 191 | |
| }, | |
| { | |
| "epoch": 0.13161953727506426, | |
| "grad_norm": 1.0943187475204468, | |
| "learning_rate": 3.0575757355586817e-07, | |
| "loss": 2.0193, | |
| "step": 192 | |
| }, | |
| { | |
| "epoch": 0.1323050556983719, | |
| "grad_norm": 1.297936201095581, | |
| "learning_rate": 2.3417567762266497e-07, | |
| "loss": 2.13, | |
| "step": 193 | |
| }, | |
| { | |
| "epoch": 0.13299057412167953, | |
| "grad_norm": 1.3492588996887207, | |
| "learning_rate": 1.7209841092460043e-07, | |
| "loss": 2.289, | |
| "step": 194 | |
| }, | |
| { | |
| "epoch": 0.13367609254498714, | |
| "grad_norm": 1.3177716732025146, | |
| "learning_rate": 1.1954274476655534e-07, | |
| "loss": 2.2274, | |
| "step": 195 | |
| }, | |
| { | |
| "epoch": 0.13436161096829477, | |
| "grad_norm": 1.4273556470870972, | |
| "learning_rate": 7.652304734289127e-08, | |
| "loss": 2.1752, | |
| "step": 196 | |
| }, | |
| { | |
| "epoch": 0.1350471293916024, | |
| "grad_norm": 1.7116583585739136, | |
| "learning_rate": 4.30510798093342e-08, | |
| "loss": 2.1852, | |
| "step": 197 | |
| }, | |
| { | |
| "epoch": 0.13573264781491, | |
| "grad_norm": 1.9014217853546143, | |
| "learning_rate": 1.9135993067588284e-08, | |
| "loss": 1.9634, | |
| "step": 198 | |
| }, | |
| { | |
| "epoch": 0.13641816623821765, | |
| "grad_norm": 2.3818442821502686, | |
| "learning_rate": 4.784325263584854e-09, | |
| "loss": 1.9621, | |
| "step": 199 | |
| }, | |
| { | |
| "epoch": 0.13710368466152528, | |
| "grad_norm": 3.527977705001831, | |
| "learning_rate": 0.0, | |
| "loss": 2.2093, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.13710368466152528, | |
| "eval_loss": 2.171226978302002, | |
| "eval_runtime": 175.9338, | |
| "eval_samples_per_second": 13.965, | |
| "eval_steps_per_second": 3.496, | |
| "step": 200 | |
| } | |
| ], | |
| "logging_steps": 1, | |
| "max_steps": 200, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 1, | |
| "save_steps": 50, | |
| "stateful_callbacks": { | |
| "EarlyStoppingCallback": { | |
| "args": { | |
| "early_stopping_patience": 4, | |
| "early_stopping_threshold": 0.0 | |
| }, | |
| "attributes": { | |
| "early_stopping_patience_counter": 0 | |
| } | |
| }, | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 2.8579962575388672e+17, | |
| "train_batch_size": 8, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |