Instructions to use sn56a6/edf1df8b-bf6c-4fc0-96e6-8e191e0da18f with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use sn56a6/edf1df8b-bf6c-4fc0-96e6-8e191e0da18f with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("peft-internal-testing/tiny-dummy-qwen2") model = PeftModel.from_pretrained(base_model, "sn56a6/edf1df8b-bf6c-4fc0-96e6-8e191e0da18f") - Notebooks
- Google Colab
- Kaggle
Download last-checkpoint/trainer_state.json from sn56a6/edf1df8b-bf6c-4fc0-96e6-8e191e0da18f: direct link, hf CLI and curl.
- Browser
- Download file 32.4 kB
-
https://huggingface.co/sn56a6/edf1df8b-bf6c-4fc0-96e6-8e191e0da18f/resolve/main/last-checkpoint/trainer_state.json
- Command line
-
hf download hf://sn56a6/edf1df8b-bf6c-4fc0-96e6-8e191e0da18f/last-checkpoint/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/sn56a6/edf1df8b-bf6c-4fc0-96e6-8e191e0da18f/resolve/main/last-checkpoint/trainer_state.json
32.4 kB
| { | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 1.1757789535567313, | |
| "eval_steps": 42, | |
| "global_step": 500, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.0023515579071134627, | |
| "eval_loss": 11.93134593963623, | |
| "eval_runtime": 10.5205, | |
| "eval_samples_per_second": 272.23, | |
| "eval_steps_per_second": 8.555, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.007054673721340388, | |
| "grad_norm": 0.012463300488889217, | |
| "learning_rate": 3e-05, | |
| "loss": 11.9313, | |
| "step": 3 | |
| }, | |
| { | |
| "epoch": 0.014109347442680775, | |
| "grad_norm": 0.015666738152503967, | |
| "learning_rate": 6e-05, | |
| "loss": 11.9317, | |
| "step": 6 | |
| }, | |
| { | |
| "epoch": 0.021164021164021163, | |
| "grad_norm": 0.021637123078107834, | |
| "learning_rate": 9e-05, | |
| "loss": 11.9311, | |
| "step": 9 | |
| }, | |
| { | |
| "epoch": 0.02821869488536155, | |
| "grad_norm": 0.01769455522298813, | |
| "learning_rate": 9.999588943391597e-05, | |
| "loss": 11.9313, | |
| "step": 12 | |
| }, | |
| { | |
| "epoch": 0.03527336860670194, | |
| "grad_norm": 0.018525123596191406, | |
| "learning_rate": 9.99743108100344e-05, | |
| "loss": 11.9307, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.042328042328042326, | |
| "grad_norm": 0.022749891504645348, | |
| "learning_rate": 9.993424445916923e-05, | |
| "loss": 11.9299, | |
| "step": 18 | |
| }, | |
| { | |
| "epoch": 0.04938271604938271, | |
| "grad_norm": 0.033247560262680054, | |
| "learning_rate": 9.987570520365104e-05, | |
| "loss": 11.9297, | |
| "step": 21 | |
| }, | |
| { | |
| "epoch": 0.0564373897707231, | |
| "grad_norm": 0.03264958783984184, | |
| "learning_rate": 9.979871469976196e-05, | |
| "loss": 11.9297, | |
| "step": 24 | |
| }, | |
| { | |
| "epoch": 0.06349206349206349, | |
| "grad_norm": 0.03716759756207466, | |
| "learning_rate": 9.970330142972401e-05, | |
| "loss": 11.9294, | |
| "step": 27 | |
| }, | |
| { | |
| "epoch": 0.07054673721340388, | |
| "grad_norm": 0.06127162277698517, | |
| "learning_rate": 9.95895006911623e-05, | |
| "loss": 11.9286, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.07760141093474426, | |
| "grad_norm": 0.04590151831507683, | |
| "learning_rate": 9.945735458404681e-05, | |
| "loss": 11.9284, | |
| "step": 33 | |
| }, | |
| { | |
| "epoch": 0.08465608465608465, | |
| "grad_norm": 0.056038279086351395, | |
| "learning_rate": 9.930691199511775e-05, | |
| "loss": 11.9269, | |
| "step": 36 | |
| }, | |
| { | |
| "epoch": 0.09171075837742504, | |
| "grad_norm": 0.05783422291278839, | |
| "learning_rate": 9.91382285798002e-05, | |
| "loss": 11.927, | |
| "step": 39 | |
| }, | |
| { | |
| "epoch": 0.09876543209876543, | |
| "grad_norm": 0.053165242075920105, | |
| "learning_rate": 9.895136674161465e-05, | |
| "loss": 11.9254, | |
| "step": 42 | |
| }, | |
| { | |
| "epoch": 0.09876543209876543, | |
| "eval_loss": 11.925036430358887, | |
| "eval_runtime": 10.6532, | |
| "eval_samples_per_second": 268.841, | |
| "eval_steps_per_second": 8.448, | |
| "step": 42 | |
| }, | |
| { | |
| "epoch": 0.10582010582010581, | |
| "grad_norm": 0.038137730211019516, | |
| "learning_rate": 9.874639560909117e-05, | |
| "loss": 11.9253, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 0.1128747795414462, | |
| "grad_norm": 0.047474514693021774, | |
| "learning_rate": 9.852339101019574e-05, | |
| "loss": 11.9246, | |
| "step": 48 | |
| }, | |
| { | |
| "epoch": 0.11992945326278659, | |
| "grad_norm": 0.040923334658145905, | |
| "learning_rate": 9.828243544427796e-05, | |
| "loss": 11.924, | |
| "step": 51 | |
| }, | |
| { | |
| "epoch": 0.12698412698412698, | |
| "grad_norm": 0.04106362909078598, | |
| "learning_rate": 9.802361805155097e-05, | |
| "loss": 11.923, | |
| "step": 54 | |
| }, | |
| { | |
| "epoch": 0.13403880070546736, | |
| "grad_norm": 0.032283883541822433, | |
| "learning_rate": 9.774703458011453e-05, | |
| "loss": 11.9225, | |
| "step": 57 | |
| }, | |
| { | |
| "epoch": 0.14109347442680775, | |
| "grad_norm": 0.06524904817342758, | |
| "learning_rate": 9.745278735053343e-05, | |
| "loss": 11.9216, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.14814814814814814, | |
| "grad_norm": 0.030542707070708275, | |
| "learning_rate": 9.714098521798465e-05, | |
| "loss": 11.9207, | |
| "step": 63 | |
| }, | |
| { | |
| "epoch": 0.15520282186948853, | |
| "grad_norm": 0.02981523610651493, | |
| "learning_rate": 9.681174353198687e-05, | |
| "loss": 11.9213, | |
| "step": 66 | |
| }, | |
| { | |
| "epoch": 0.16225749559082892, | |
| "grad_norm": 0.03491352126002312, | |
| "learning_rate": 9.64651840937276e-05, | |
| "loss": 11.921, | |
| "step": 69 | |
| }, | |
| { | |
| "epoch": 0.1693121693121693, | |
| "grad_norm": 0.029117964208126068, | |
| "learning_rate": 9.610143511100354e-05, | |
| "loss": 11.9203, | |
| "step": 72 | |
| }, | |
| { | |
| "epoch": 0.1763668430335097, | |
| "grad_norm": 0.03240862861275673, | |
| "learning_rate": 9.572063115079063e-05, | |
| "loss": 11.9196, | |
| "step": 75 | |
| }, | |
| { | |
| "epoch": 0.18342151675485008, | |
| "grad_norm": 0.022400949150323868, | |
| "learning_rate": 9.53229130894619e-05, | |
| "loss": 11.9196, | |
| "step": 78 | |
| }, | |
| { | |
| "epoch": 0.19047619047619047, | |
| "grad_norm": 0.022678203880786896, | |
| "learning_rate": 9.490842806067095e-05, | |
| "loss": 11.9205, | |
| "step": 81 | |
| }, | |
| { | |
| "epoch": 0.19753086419753085, | |
| "grad_norm": 0.03054334782063961, | |
| "learning_rate": 9.44773294009206e-05, | |
| "loss": 11.9196, | |
| "step": 84 | |
| }, | |
| { | |
| "epoch": 0.19753086419753085, | |
| "eval_loss": 11.919329643249512, | |
| "eval_runtime": 10.8695, | |
| "eval_samples_per_second": 263.49, | |
| "eval_steps_per_second": 8.28, | |
| "step": 84 | |
| }, | |
| { | |
| "epoch": 0.20458553791887124, | |
| "grad_norm": 0.02172757126390934, | |
| "learning_rate": 9.40297765928369e-05, | |
| "loss": 11.9193, | |
| "step": 87 | |
| }, | |
| { | |
| "epoch": 0.21164021164021163, | |
| "grad_norm": 0.023694947361946106, | |
| "learning_rate": 9.356593520616948e-05, | |
| "loss": 11.919, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.21869488536155202, | |
| "grad_norm": 0.03806081414222717, | |
| "learning_rate": 9.308597683653975e-05, | |
| "loss": 11.9175, | |
| "step": 93 | |
| }, | |
| { | |
| "epoch": 0.2257495590828924, | |
| "grad_norm": 0.02170073799788952, | |
| "learning_rate": 9.259007904196023e-05, | |
| "loss": 11.9193, | |
| "step": 96 | |
| }, | |
| { | |
| "epoch": 0.2328042328042328, | |
| "grad_norm": 0.02128906175494194, | |
| "learning_rate": 9.207842527714767e-05, | |
| "loss": 11.9188, | |
| "step": 99 | |
| }, | |
| { | |
| "epoch": 0.23985890652557318, | |
| "grad_norm": 0.02206624485552311, | |
| "learning_rate": 9.155120482565521e-05, | |
| "loss": 11.9186, | |
| "step": 102 | |
| }, | |
| { | |
| "epoch": 0.24691358024691357, | |
| "grad_norm": 0.021899454295635223, | |
| "learning_rate": 9.10086127298478e-05, | |
| "loss": 11.918, | |
| "step": 105 | |
| }, | |
| { | |
| "epoch": 0.25396825396825395, | |
| "grad_norm": 0.030206996947526932, | |
| "learning_rate": 9.045084971874738e-05, | |
| "loss": 11.9182, | |
| "step": 108 | |
| }, | |
| { | |
| "epoch": 0.26102292768959434, | |
| "grad_norm": 0.02562589943408966, | |
| "learning_rate": 8.987812213377424e-05, | |
| "loss": 11.9188, | |
| "step": 111 | |
| }, | |
| { | |
| "epoch": 0.26807760141093473, | |
| "grad_norm": 0.021965835243463516, | |
| "learning_rate": 8.929064185241213e-05, | |
| "loss": 11.9183, | |
| "step": 114 | |
| }, | |
| { | |
| "epoch": 0.2751322751322751, | |
| "grad_norm": 0.02135823480784893, | |
| "learning_rate": 8.868862620982534e-05, | |
| "loss": 11.9182, | |
| "step": 117 | |
| }, | |
| { | |
| "epoch": 0.2821869488536155, | |
| "grad_norm": 0.029081886634230614, | |
| "learning_rate": 8.807229791845673e-05, | |
| "loss": 11.9185, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.2892416225749559, | |
| "grad_norm": 0.01876562461256981, | |
| "learning_rate": 8.744188498563641e-05, | |
| "loss": 11.9178, | |
| "step": 123 | |
| }, | |
| { | |
| "epoch": 0.2962962962962963, | |
| "grad_norm": 0.026901056990027428, | |
| "learning_rate": 8.679762062923175e-05, | |
| "loss": 11.9171, | |
| "step": 126 | |
| }, | |
| { | |
| "epoch": 0.2962962962962963, | |
| "eval_loss": 11.917308807373047, | |
| "eval_runtime": 10.6998, | |
| "eval_samples_per_second": 267.668, | |
| "eval_steps_per_second": 8.411, | |
| "step": 126 | |
| }, | |
| { | |
| "epoch": 0.30335097001763667, | |
| "grad_norm": 0.02258964441716671, | |
| "learning_rate": 8.613974319136958e-05, | |
| "loss": 11.9175, | |
| "step": 129 | |
| }, | |
| { | |
| "epoch": 0.31040564373897706, | |
| "grad_norm": 0.019902532920241356, | |
| "learning_rate": 8.54684960502629e-05, | |
| "loss": 11.9189, | |
| "step": 132 | |
| }, | |
| { | |
| "epoch": 0.31746031746031744, | |
| "grad_norm": 0.020657015964388847, | |
| "learning_rate": 8.478412753017433e-05, | |
| "loss": 11.9177, | |
| "step": 135 | |
| }, | |
| { | |
| "epoch": 0.32451499118165783, | |
| "grad_norm": 0.02019345760345459, | |
| "learning_rate": 8.408689080954998e-05, | |
| "loss": 11.9179, | |
| "step": 138 | |
| }, | |
| { | |
| "epoch": 0.3315696649029982, | |
| "grad_norm": 0.020498735830187798, | |
| "learning_rate": 8.33770438273574e-05, | |
| "loss": 11.9175, | |
| "step": 141 | |
| }, | |
| { | |
| "epoch": 0.3386243386243386, | |
| "grad_norm": 0.02337423898279667, | |
| "learning_rate": 8.265484918766243e-05, | |
| "loss": 11.9178, | |
| "step": 144 | |
| }, | |
| { | |
| "epoch": 0.345679012345679, | |
| "grad_norm": 0.02081083133816719, | |
| "learning_rate": 8.192057406248028e-05, | |
| "loss": 11.9165, | |
| "step": 147 | |
| }, | |
| { | |
| "epoch": 0.3527336860670194, | |
| "grad_norm": 0.01860121265053749, | |
| "learning_rate": 8.117449009293668e-05, | |
| "loss": 11.9177, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.35978835978835977, | |
| "grad_norm": 0.02201470360159874, | |
| "learning_rate": 8.041687328877567e-05, | |
| "loss": 11.9165, | |
| "step": 153 | |
| }, | |
| { | |
| "epoch": 0.36684303350970016, | |
| "grad_norm": 0.024001609534025192, | |
| "learning_rate": 7.964800392625129e-05, | |
| "loss": 11.9168, | |
| "step": 156 | |
| }, | |
| { | |
| "epoch": 0.37389770723104054, | |
| "grad_norm": 0.02816389873623848, | |
| "learning_rate": 7.886816644444098e-05, | |
| "loss": 11.9174, | |
| "step": 159 | |
| }, | |
| { | |
| "epoch": 0.38095238095238093, | |
| "grad_norm": 0.01967223547399044, | |
| "learning_rate": 7.807764934001874e-05, | |
| "loss": 11.9147, | |
| "step": 162 | |
| }, | |
| { | |
| "epoch": 0.3880070546737213, | |
| "grad_norm": 0.033686283975839615, | |
| "learning_rate": 7.727674506052743e-05, | |
| "loss": 11.9176, | |
| "step": 165 | |
| }, | |
| { | |
| "epoch": 0.3950617283950617, | |
| "grad_norm": 0.029333915561437607, | |
| "learning_rate": 7.646574989618938e-05, | |
| "loss": 11.9173, | |
| "step": 168 | |
| }, | |
| { | |
| "epoch": 0.3950617283950617, | |
| "eval_loss": 11.916316032409668, | |
| "eval_runtime": 10.8242, | |
| "eval_samples_per_second": 264.592, | |
| "eval_steps_per_second": 8.315, | |
| "step": 168 | |
| }, | |
| { | |
| "epoch": 0.4021164021164021, | |
| "grad_norm": 0.020618749782443047, | |
| "learning_rate": 7.564496387029532e-05, | |
| "loss": 11.9169, | |
| "step": 171 | |
| }, | |
| { | |
| "epoch": 0.4091710758377425, | |
| "grad_norm": 0.02999415621161461, | |
| "learning_rate": 7.481469062821252e-05, | |
| "loss": 11.9169, | |
| "step": 174 | |
| }, | |
| { | |
| "epoch": 0.41622574955908287, | |
| "grad_norm": 0.01824590004980564, | |
| "learning_rate": 7.39752373250527e-05, | |
| "loss": 11.9167, | |
| "step": 177 | |
| }, | |
| { | |
| "epoch": 0.42328042328042326, | |
| "grad_norm": 0.05701108276844025, | |
| "learning_rate": 7.312691451204178e-05, | |
| "loss": 11.9158, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.43033509700176364, | |
| "grad_norm": 0.022198380902409554, | |
| "learning_rate": 7.227003602163295e-05, | |
| "loss": 11.9165, | |
| "step": 183 | |
| }, | |
| { | |
| "epoch": 0.43738977072310403, | |
| "grad_norm": 0.022028392180800438, | |
| "learning_rate": 7.14049188514063e-05, | |
| "loss": 11.9166, | |
| "step": 186 | |
| }, | |
| { | |
| "epoch": 0.4444444444444444, | |
| "grad_norm": 0.03448006883263588, | |
| "learning_rate": 7.05318830467969e-05, | |
| "loss": 11.9165, | |
| "step": 189 | |
| }, | |
| { | |
| "epoch": 0.4514991181657848, | |
| "grad_norm": 0.023939242586493492, | |
| "learning_rate": 6.965125158269619e-05, | |
| "loss": 11.9164, | |
| "step": 192 | |
| }, | |
| { | |
| "epoch": 0.4585537918871252, | |
| "grad_norm": 0.01913815177977085, | |
| "learning_rate": 6.876335024396872e-05, | |
| "loss": 11.9166, | |
| "step": 195 | |
| }, | |
| { | |
| "epoch": 0.4656084656084656, | |
| "grad_norm": 0.02663358673453331, | |
| "learning_rate": 6.786850750493006e-05, | |
| "loss": 11.917, | |
| "step": 198 | |
| }, | |
| { | |
| "epoch": 0.47266313932980597, | |
| "grad_norm": 0.020899388939142227, | |
| "learning_rate": 6.696705440782938e-05, | |
| "loss": 11.9154, | |
| "step": 201 | |
| }, | |
| { | |
| "epoch": 0.47971781305114636, | |
| "grad_norm": 0.023673586547374725, | |
| "learning_rate": 6.605932444038229e-05, | |
| "loss": 11.9157, | |
| "step": 204 | |
| }, | |
| { | |
| "epoch": 0.48677248677248675, | |
| "grad_norm": 0.025094488635659218, | |
| "learning_rate": 6.514565341239861e-05, | |
| "loss": 11.9159, | |
| "step": 207 | |
| }, | |
| { | |
| "epoch": 0.49382716049382713, | |
| "grad_norm": 0.02767755836248398, | |
| "learning_rate": 6.422637933155162e-05, | |
| "loss": 11.9158, | |
| "step": 210 | |
| }, | |
| { | |
| "epoch": 0.49382716049382713, | |
| "eval_loss": 11.915718078613281, | |
| "eval_runtime": 10.5682, | |
| "eval_samples_per_second": 271.003, | |
| "eval_steps_per_second": 8.516, | |
| "step": 210 | |
| }, | |
| { | |
| "epoch": 0.5008818342151675, | |
| "grad_norm": 0.02221379242837429, | |
| "learning_rate": 6.330184227833376e-05, | |
| "loss": 11.9166, | |
| "step": 213 | |
| }, | |
| { | |
| "epoch": 0.5079365079365079, | |
| "grad_norm": 0.021963153034448624, | |
| "learning_rate": 6.237238428024572e-05, | |
| "loss": 11.9161, | |
| "step": 216 | |
| }, | |
| { | |
| "epoch": 0.5149911816578483, | |
| "grad_norm": 0.026923993602395058, | |
| "learning_rate": 6.143834918526527e-05, | |
| "loss": 11.9146, | |
| "step": 219 | |
| }, | |
| { | |
| "epoch": 0.5220458553791887, | |
| "grad_norm": 0.035227347165346146, | |
| "learning_rate": 6.0500082534642464e-05, | |
| "loss": 11.9154, | |
| "step": 222 | |
| }, | |
| { | |
| "epoch": 0.5291005291005291, | |
| "grad_norm": 0.021038008853793144, | |
| "learning_rate": 5.955793143506863e-05, | |
| "loss": 11.9158, | |
| "step": 225 | |
| }, | |
| { | |
| "epoch": 0.5361552028218695, | |
| "grad_norm": 0.01937274821102619, | |
| "learning_rate": 5.861224443026595e-05, | |
| "loss": 11.9162, | |
| "step": 228 | |
| }, | |
| { | |
| "epoch": 0.5432098765432098, | |
| "grad_norm": 0.019016142934560776, | |
| "learning_rate": 5.766337137204579e-05, | |
| "loss": 11.9163, | |
| "step": 231 | |
| }, | |
| { | |
| "epoch": 0.5502645502645502, | |
| "grad_norm": 0.027969468384981155, | |
| "learning_rate": 5.6711663290882776e-05, | |
| "loss": 11.9162, | |
| "step": 234 | |
| }, | |
| { | |
| "epoch": 0.5573192239858906, | |
| "grad_norm": 0.017542190849781036, | |
| "learning_rate": 5.575747226605298e-05, | |
| "loss": 11.917, | |
| "step": 237 | |
| }, | |
| { | |
| "epoch": 0.564373897707231, | |
| "grad_norm": 0.023203639313578606, | |
| "learning_rate": 5.480115129538409e-05, | |
| "loss": 11.9157, | |
| "step": 240 | |
| }, | |
| { | |
| "epoch": 0.5714285714285714, | |
| "grad_norm": 0.014958036132156849, | |
| "learning_rate": 5.384305416466584e-05, | |
| "loss": 11.9154, | |
| "step": 243 | |
| }, | |
| { | |
| "epoch": 0.5784832451499118, | |
| "grad_norm": 0.027328725904226303, | |
| "learning_rate": 5.288353531676873e-05, | |
| "loss": 11.9163, | |
| "step": 246 | |
| }, | |
| { | |
| "epoch": 0.5855379188712522, | |
| "grad_norm": 0.03369058296084404, | |
| "learning_rate": 5.192294972051992e-05, | |
| "loss": 11.9157, | |
| "step": 249 | |
| }, | |
| { | |
| "epoch": 0.5925925925925926, | |
| "grad_norm": 0.019253922626376152, | |
| "learning_rate": 5.0961652739384356e-05, | |
| "loss": 11.9147, | |
| "step": 252 | |
| }, | |
| { | |
| "epoch": 0.5925925925925926, | |
| "eval_loss": 11.915318489074707, | |
| "eval_runtime": 10.3985, | |
| "eval_samples_per_second": 275.425, | |
| "eval_steps_per_second": 8.655, | |
| "step": 252 | |
| }, | |
| { | |
| "epoch": 0.599647266313933, | |
| "grad_norm": 0.024502824991941452, | |
| "learning_rate": 5e-05, | |
| "loss": 11.9151, | |
| "step": 255 | |
| }, | |
| { | |
| "epoch": 0.6067019400352733, | |
| "grad_norm": 0.03233085200190544, | |
| "learning_rate": 4.903834726061565e-05, | |
| "loss": 11.9151, | |
| "step": 258 | |
| }, | |
| { | |
| "epoch": 0.6137566137566137, | |
| "grad_norm": 0.02394239790737629, | |
| "learning_rate": 4.807705027948008e-05, | |
| "loss": 11.915, | |
| "step": 261 | |
| }, | |
| { | |
| "epoch": 0.6208112874779541, | |
| "grad_norm": 0.01676931232213974, | |
| "learning_rate": 4.711646468323129e-05, | |
| "loss": 11.9164, | |
| "step": 264 | |
| }, | |
| { | |
| "epoch": 0.6278659611992945, | |
| "grad_norm": 0.02555139549076557, | |
| "learning_rate": 4.6156945835334184e-05, | |
| "loss": 11.9157, | |
| "step": 267 | |
| }, | |
| { | |
| "epoch": 0.6349206349206349, | |
| "grad_norm": 0.02273687720298767, | |
| "learning_rate": 4.5198848704615914e-05, | |
| "loss": 11.9159, | |
| "step": 270 | |
| }, | |
| { | |
| "epoch": 0.6419753086419753, | |
| "grad_norm": 0.0266796313226223, | |
| "learning_rate": 4.424252773394704e-05, | |
| "loss": 11.9163, | |
| "step": 273 | |
| }, | |
| { | |
| "epoch": 0.6490299823633157, | |
| "grad_norm": 0.025090264156460762, | |
| "learning_rate": 4.328833670911724e-05, | |
| "loss": 11.9164, | |
| "step": 276 | |
| }, | |
| { | |
| "epoch": 0.656084656084656, | |
| "grad_norm": 0.02523481287062168, | |
| "learning_rate": 4.23366286279542e-05, | |
| "loss": 11.9157, | |
| "step": 279 | |
| }, | |
| { | |
| "epoch": 0.6631393298059964, | |
| "grad_norm": 0.021461475640535355, | |
| "learning_rate": 4.138775556973406e-05, | |
| "loss": 11.9147, | |
| "step": 282 | |
| }, | |
| { | |
| "epoch": 0.6701940035273368, | |
| "grad_norm": 0.023472610861063004, | |
| "learning_rate": 4.04420685649314e-05, | |
| "loss": 11.9161, | |
| "step": 285 | |
| }, | |
| { | |
| "epoch": 0.6772486772486772, | |
| "grad_norm": 0.024858828634023666, | |
| "learning_rate": 3.9499917465357534e-05, | |
| "loss": 11.9152, | |
| "step": 288 | |
| }, | |
| { | |
| "epoch": 0.6843033509700176, | |
| "grad_norm": 0.020553085952997208, | |
| "learning_rate": 3.856165081473474e-05, | |
| "loss": 11.9139, | |
| "step": 291 | |
| }, | |
| { | |
| "epoch": 0.691358024691358, | |
| "grad_norm": 0.02018040418624878, | |
| "learning_rate": 3.762761571975429e-05, | |
| "loss": 11.9156, | |
| "step": 294 | |
| }, | |
| { | |
| "epoch": 0.691358024691358, | |
| "eval_loss": 11.915068626403809, | |
| "eval_runtime": 10.6749, | |
| "eval_samples_per_second": 268.292, | |
| "eval_steps_per_second": 8.431, | |
| "step": 294 | |
| }, | |
| { | |
| "epoch": 0.6984126984126984, | |
| "grad_norm": 0.01783376932144165, | |
| "learning_rate": 3.6698157721666246e-05, | |
| "loss": 11.9134, | |
| "step": 297 | |
| }, | |
| { | |
| "epoch": 0.7054673721340388, | |
| "grad_norm": 0.028524892404675484, | |
| "learning_rate": 3.5773620668448384e-05, | |
| "loss": 11.9158, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.7125220458553791, | |
| "grad_norm": 0.02065563015639782, | |
| "learning_rate": 3.48543465876014e-05, | |
| "loss": 11.9146, | |
| "step": 303 | |
| }, | |
| { | |
| "epoch": 0.7195767195767195, | |
| "grad_norm": 0.045134805142879486, | |
| "learning_rate": 3.3940675559617724e-05, | |
| "loss": 11.9159, | |
| "step": 306 | |
| }, | |
| { | |
| "epoch": 0.7266313932980599, | |
| "grad_norm": 0.023695290088653564, | |
| "learning_rate": 3.303294559217063e-05, | |
| "loss": 11.9153, | |
| "step": 309 | |
| }, | |
| { | |
| "epoch": 0.7336860670194003, | |
| "grad_norm": 0.01704533025622368, | |
| "learning_rate": 3.213149249506997e-05, | |
| "loss": 11.915, | |
| "step": 312 | |
| }, | |
| { | |
| "epoch": 0.7407407407407407, | |
| "grad_norm": 0.02306770719587803, | |
| "learning_rate": 3.12366497560313e-05, | |
| "loss": 11.9155, | |
| "step": 315 | |
| }, | |
| { | |
| "epoch": 0.7477954144620811, | |
| "grad_norm": 0.023183325305581093, | |
| "learning_rate": 3.0348748417303823e-05, | |
| "loss": 11.9152, | |
| "step": 318 | |
| }, | |
| { | |
| "epoch": 0.7548500881834215, | |
| "grad_norm": 0.01855415478348732, | |
| "learning_rate": 2.9468116953203107e-05, | |
| "loss": 11.9163, | |
| "step": 321 | |
| }, | |
| { | |
| "epoch": 0.7619047619047619, | |
| "grad_norm": 0.02062196470797062, | |
| "learning_rate": 2.8595081148593738e-05, | |
| "loss": 11.9154, | |
| "step": 324 | |
| }, | |
| { | |
| "epoch": 0.7689594356261023, | |
| "grad_norm": 0.025098076090216637, | |
| "learning_rate": 2.772996397836704e-05, | |
| "loss": 11.9162, | |
| "step": 327 | |
| }, | |
| { | |
| "epoch": 0.7760141093474426, | |
| "grad_norm": 0.018340418115258217, | |
| "learning_rate": 2.687308548795825e-05, | |
| "loss": 11.9156, | |
| "step": 330 | |
| }, | |
| { | |
| "epoch": 0.783068783068783, | |
| "grad_norm": 0.02159620076417923, | |
| "learning_rate": 2.6024762674947313e-05, | |
| "loss": 11.9144, | |
| "step": 333 | |
| }, | |
| { | |
| "epoch": 0.7901234567901234, | |
| "grad_norm": 0.02330397628247738, | |
| "learning_rate": 2.5185309371787513e-05, | |
| "loss": 11.9156, | |
| "step": 336 | |
| }, | |
| { | |
| "epoch": 0.7901234567901234, | |
| "eval_loss": 11.914894104003906, | |
| "eval_runtime": 10.4255, | |
| "eval_samples_per_second": 274.71, | |
| "eval_steps_per_second": 8.633, | |
| "step": 336 | |
| }, | |
| { | |
| "epoch": 0.7971781305114638, | |
| "grad_norm": 0.023469174280762672, | |
| "learning_rate": 2.43550361297047e-05, | |
| "loss": 11.9143, | |
| "step": 339 | |
| }, | |
| { | |
| "epoch": 0.8042328042328042, | |
| "grad_norm": 0.01915036514401436, | |
| "learning_rate": 2.353425010381063e-05, | |
| "loss": 11.917, | |
| "step": 342 | |
| }, | |
| { | |
| "epoch": 0.8112874779541446, | |
| "grad_norm": 0.021919777616858482, | |
| "learning_rate": 2.272325493947257e-05, | |
| "loss": 11.9137, | |
| "step": 345 | |
| }, | |
| { | |
| "epoch": 0.818342151675485, | |
| "grad_norm": 0.03252611309289932, | |
| "learning_rate": 2.192235065998126e-05, | |
| "loss": 11.9149, | |
| "step": 348 | |
| }, | |
| { | |
| "epoch": 0.8253968253968254, | |
| "grad_norm": 0.021312009543180466, | |
| "learning_rate": 2.1131833555559037e-05, | |
| "loss": 11.9156, | |
| "step": 351 | |
| }, | |
| { | |
| "epoch": 0.8324514991181657, | |
| "grad_norm": 0.02721288986504078, | |
| "learning_rate": 2.0351996073748713e-05, | |
| "loss": 11.9151, | |
| "step": 354 | |
| }, | |
| { | |
| "epoch": 0.8395061728395061, | |
| "grad_norm": 0.019871938973665237, | |
| "learning_rate": 1.9583126711224343e-05, | |
| "loss": 11.9153, | |
| "step": 357 | |
| }, | |
| { | |
| "epoch": 0.8465608465608465, | |
| "grad_norm": 0.02446158602833748, | |
| "learning_rate": 1.8825509907063327e-05, | |
| "loss": 11.9161, | |
| "step": 360 | |
| }, | |
| { | |
| "epoch": 0.8536155202821869, | |
| "grad_norm": 0.028117047622799873, | |
| "learning_rate": 1.807942593751973e-05, | |
| "loss": 11.9168, | |
| "step": 363 | |
| }, | |
| { | |
| "epoch": 0.8606701940035273, | |
| "grad_norm": 0.02583438903093338, | |
| "learning_rate": 1.7345150812337564e-05, | |
| "loss": 11.9152, | |
| "step": 366 | |
| }, | |
| { | |
| "epoch": 0.8677248677248677, | |
| "grad_norm": 0.026459960266947746, | |
| "learning_rate": 1.66229561726426e-05, | |
| "loss": 11.9158, | |
| "step": 369 | |
| }, | |
| { | |
| "epoch": 0.8747795414462081, | |
| "grad_norm": 0.01762359030544758, | |
| "learning_rate": 1.5913109190450032e-05, | |
| "loss": 11.9162, | |
| "step": 372 | |
| }, | |
| { | |
| "epoch": 0.8818342151675485, | |
| "grad_norm": 0.02677866630256176, | |
| "learning_rate": 1.5215872469825682e-05, | |
| "loss": 11.9157, | |
| "step": 375 | |
| }, | |
| { | |
| "epoch": 0.8888888888888888, | |
| "grad_norm": 0.03301187977194786, | |
| "learning_rate": 1.4531503949737108e-05, | |
| "loss": 11.9147, | |
| "step": 378 | |
| }, | |
| { | |
| "epoch": 0.8888888888888888, | |
| "eval_loss": 11.914801597595215, | |
| "eval_runtime": 10.6272, | |
| "eval_samples_per_second": 269.497, | |
| "eval_steps_per_second": 8.469, | |
| "step": 378 | |
| }, | |
| { | |
| "epoch": 0.8959435626102292, | |
| "grad_norm": 0.019808808341622353, | |
| "learning_rate": 1.3860256808630428e-05, | |
| "loss": 11.916, | |
| "step": 381 | |
| }, | |
| { | |
| "epoch": 0.9029982363315696, | |
| "grad_norm": 0.023428916931152344, | |
| "learning_rate": 1.3202379370768252e-05, | |
| "loss": 11.916, | |
| "step": 384 | |
| }, | |
| { | |
| "epoch": 0.91005291005291, | |
| "grad_norm": 0.018417634069919586, | |
| "learning_rate": 1.2558115014363592e-05, | |
| "loss": 11.9163, | |
| "step": 387 | |
| }, | |
| { | |
| "epoch": 0.9171075837742504, | |
| "grad_norm": 0.020826270803809166, | |
| "learning_rate": 1.1927702081543279e-05, | |
| "loss": 11.9157, | |
| "step": 390 | |
| }, | |
| { | |
| "epoch": 0.9241622574955908, | |
| "grad_norm": 0.02966841123998165, | |
| "learning_rate": 1.1311373790174657e-05, | |
| "loss": 11.9141, | |
| "step": 393 | |
| }, | |
| { | |
| "epoch": 0.9312169312169312, | |
| "grad_norm": 0.018212970346212387, | |
| "learning_rate": 1.0709358147587884e-05, | |
| "loss": 11.9149, | |
| "step": 396 | |
| }, | |
| { | |
| "epoch": 0.9382716049382716, | |
| "grad_norm": 0.022182345390319824, | |
| "learning_rate": 1.0121877866225781e-05, | |
| "loss": 11.9147, | |
| "step": 399 | |
| }, | |
| { | |
| "epoch": 0.9453262786596119, | |
| "grad_norm": 0.0208954568952322, | |
| "learning_rate": 9.549150281252633e-06, | |
| "loss": 11.9151, | |
| "step": 402 | |
| }, | |
| { | |
| "epoch": 0.9523809523809523, | |
| "grad_norm": 0.022049158811569214, | |
| "learning_rate": 8.991387270152201e-06, | |
| "loss": 11.9143, | |
| "step": 405 | |
| }, | |
| { | |
| "epoch": 0.9594356261022927, | |
| "grad_norm": 0.024066656827926636, | |
| "learning_rate": 8.448795174344804e-06, | |
| "loss": 11.9147, | |
| "step": 408 | |
| }, | |
| { | |
| "epoch": 0.9664902998236331, | |
| "grad_norm": 0.0210318211466074, | |
| "learning_rate": 7.921574722852343e-06, | |
| "loss": 11.9155, | |
| "step": 411 | |
| }, | |
| { | |
| "epoch": 0.9735449735449735, | |
| "grad_norm": 0.028868673369288445, | |
| "learning_rate": 7.409920958039795e-06, | |
| "loss": 11.9127, | |
| "step": 414 | |
| }, | |
| { | |
| "epoch": 0.9805996472663139, | |
| "grad_norm": 0.023974774405360222, | |
| "learning_rate": 6.9140231634602485e-06, | |
| "loss": 11.9156, | |
| "step": 417 | |
| }, | |
| { | |
| "epoch": 0.9876543209876543, | |
| "grad_norm": 0.028788629919290543, | |
| "learning_rate": 6.43406479383053e-06, | |
| "loss": 11.9142, | |
| "step": 420 | |
| }, | |
| { | |
| "epoch": 0.9876543209876543, | |
| "eval_loss": 11.914738655090332, | |
| "eval_runtime": 10.0919, | |
| "eval_samples_per_second": 283.793, | |
| "eval_steps_per_second": 8.918, | |
| "step": 420 | |
| }, | |
| { | |
| "epoch": 0.9947089947089947, | |
| "grad_norm": 0.023696357384324074, | |
| "learning_rate": 5.9702234071631e-06, | |
| "loss": 11.9155, | |
| "step": 423 | |
| }, | |
| { | |
| "epoch": 1.001763668430335, | |
| "grad_norm": 0.03524714708328247, | |
| "learning_rate": 5.5226705990794155e-06, | |
| "loss": 14.6697, | |
| "step": 426 | |
| }, | |
| { | |
| "epoch": 1.0088183421516754, | |
| "grad_norm": 0.019016897305846214, | |
| "learning_rate": 5.091571939329048e-06, | |
| "loss": 12.1316, | |
| "step": 429 | |
| }, | |
| { | |
| "epoch": 1.0158730158730158, | |
| "grad_norm": 0.021377887576818466, | |
| "learning_rate": 4.677086910538092e-06, | |
| "loss": 12.2558, | |
| "step": 432 | |
| }, | |
| { | |
| "epoch": 1.0229276895943562, | |
| "grad_norm": 0.02245590090751648, | |
| "learning_rate": 4.279368849209381e-06, | |
| "loss": 11.7679, | |
| "step": 435 | |
| }, | |
| { | |
| "epoch": 1.0299823633156966, | |
| "grad_norm": 0.016183845698833466, | |
| "learning_rate": 3.898564888996476e-06, | |
| "loss": 11.9971, | |
| "step": 438 | |
| }, | |
| { | |
| "epoch": 1.037037037037037, | |
| "grad_norm": 0.017965037375688553, | |
| "learning_rate": 3.534815906272404e-06, | |
| "loss": 11.6156, | |
| "step": 441 | |
| }, | |
| { | |
| "epoch": 1.0440917107583774, | |
| "grad_norm": 0.020000630989670753, | |
| "learning_rate": 3.18825646801314e-06, | |
| "loss": 12.1001, | |
| "step": 444 | |
| }, | |
| { | |
| "epoch": 1.0511463844797178, | |
| "grad_norm": 0.021238798275589943, | |
| "learning_rate": 2.8590147820153513e-06, | |
| "loss": 11.8095, | |
| "step": 447 | |
| }, | |
| { | |
| "epoch": 1.0582010582010581, | |
| "grad_norm": 0.025096235796809196, | |
| "learning_rate": 2.547212649466568e-06, | |
| "loss": 11.8153, | |
| "step": 450 | |
| }, | |
| { | |
| "epoch": 1.0652557319223985, | |
| "grad_norm": 0.02049741894006729, | |
| "learning_rate": 2.2529654198854835e-06, | |
| "loss": 11.9978, | |
| "step": 453 | |
| }, | |
| { | |
| "epoch": 1.072310405643739, | |
| "grad_norm": 0.025018183514475822, | |
| "learning_rate": 1.9763819484490355e-06, | |
| "loss": 11.9496, | |
| "step": 456 | |
| }, | |
| { | |
| "epoch": 1.0793650793650793, | |
| "grad_norm": 0.021498849615454674, | |
| "learning_rate": 1.7175645557220566e-06, | |
| "loss": 11.7952, | |
| "step": 459 | |
| }, | |
| { | |
| "epoch": 1.0864197530864197, | |
| "grad_norm": 0.02332150749862194, | |
| "learning_rate": 1.4766089898042678e-06, | |
| "loss": 11.7297, | |
| "step": 462 | |
| }, | |
| { | |
| "epoch": 1.0864197530864197, | |
| "eval_loss": 11.914722442626953, | |
| "eval_runtime": 10.3803, | |
| "eval_samples_per_second": 275.908, | |
| "eval_steps_per_second": 8.67, | |
| "step": 462 | |
| }, | |
| { | |
| "epoch": 1.09347442680776, | |
| "grad_norm": 0.028014004230499268, | |
| "learning_rate": 1.2536043909088191e-06, | |
| "loss": 12.2078, | |
| "step": 465 | |
| }, | |
| { | |
| "epoch": 1.1005291005291005, | |
| "grad_norm": 0.019710175693035126, | |
| "learning_rate": 1.0486332583853563e-06, | |
| "loss": 11.932, | |
| "step": 468 | |
| }, | |
| { | |
| "epoch": 1.1075837742504409, | |
| "grad_norm": 0.0251897145062685, | |
| "learning_rate": 8.617714201998084e-07, | |
| "loss": 11.9246, | |
| "step": 471 | |
| }, | |
| { | |
| "epoch": 1.1146384479717812, | |
| "grad_norm": 0.020719952881336212, | |
| "learning_rate": 6.93088004882253e-07, | |
| "loss": 11.9967, | |
| "step": 474 | |
| }, | |
| { | |
| "epoch": 1.1216931216931216, | |
| "grad_norm": 0.01866576448082924, | |
| "learning_rate": 5.426454159531913e-07, | |
| "loss": 11.7563, | |
| "step": 477 | |
| }, | |
| { | |
| "epoch": 1.128747795414462, | |
| "grad_norm": 0.029010610654950142, | |
| "learning_rate": 4.104993088376974e-07, | |
| "loss": 11.9562, | |
| "step": 480 | |
| }, | |
| { | |
| "epoch": 1.1358024691358024, | |
| "grad_norm": 0.020120302215218544, | |
| "learning_rate": 2.966985702759828e-07, | |
| "loss": 12.0342, | |
| "step": 483 | |
| }, | |
| { | |
| "epoch": 1.1428571428571428, | |
| "grad_norm": 0.02001938596367836, | |
| "learning_rate": 2.012853002380466e-07, | |
| "loss": 11.7064, | |
| "step": 486 | |
| }, | |
| { | |
| "epoch": 1.1499118165784832, | |
| "grad_norm": 0.021438777446746826, | |
| "learning_rate": 1.2429479634897267e-07, | |
| "loss": 11.9928, | |
| "step": 489 | |
| }, | |
| { | |
| "epoch": 1.1569664902998236, | |
| "grad_norm": 0.016329078003764153, | |
| "learning_rate": 6.575554083078084e-08, | |
| "loss": 11.9815, | |
| "step": 492 | |
| }, | |
| { | |
| "epoch": 1.164021164021164, | |
| "grad_norm": 0.05567369982600212, | |
| "learning_rate": 2.568918996560532e-08, | |
| "loss": 11.9687, | |
| "step": 495 | |
| }, | |
| { | |
| "epoch": 1.1710758377425043, | |
| "grad_norm": 0.026132524013519287, | |
| "learning_rate": 4.110566084036816e-09, | |
| "loss": 11.8083, | |
| "step": 498 | |
| } | |
| ], | |
| "logging_steps": 3, | |
| "max_steps": 500, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 2, | |
| "save_steps": 42, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 13720076943360.0, | |
| "train_batch_size": 8, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |