Instructions to use cimol/b16d9612-9267-485c-a7c6-526ffb20af02 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use cimol/b16d9612-9267-485c-a7c6-526ffb20af02 with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen2.5-7B-Instruct") model = PeftModel.from_pretrained(base_model, "cimol/b16d9612-9267-485c-a7c6-526ffb20af02") - Notebooks
- Google Colab
- Kaggle
Download last-checkpoint/trainer_state.json from cimol/b16d9612-9267-485c-a7c6-526ffb20af02: direct link, hf CLI and curl.
- Browser
- Download file 15.1 kB
-
https://huggingface.co/cimol/b16d9612-9267-485c-a7c6-526ffb20af02/resolve/main/last-checkpoint/trainer_state.json
- Command line
-
hf download hf://cimol/b16d9612-9267-485c-a7c6-526ffb20af02/last-checkpoint/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/cimol/b16d9612-9267-485c-a7c6-526ffb20af02/resolve/main/last-checkpoint/trainer_state.json
15.1 kB
| { | |
| "best_metric": 1.7665985822677612, | |
| "best_model_checkpoint": "miner_id_24/checkpoint-50", | |
| "epoch": 3.0, | |
| "eval_steps": 50, | |
| "global_step": 81, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.037037037037037035, | |
| "grad_norm": 3.0762956142425537, | |
| "learning_rate": 7e-06, | |
| "loss": 2.8658, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.037037037037037035, | |
| "eval_loss": 3.5639960765838623, | |
| "eval_runtime": 3.2908, | |
| "eval_samples_per_second": 13.978, | |
| "eval_steps_per_second": 3.647, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.07407407407407407, | |
| "grad_norm": 3.893179416656494, | |
| "learning_rate": 1.4e-05, | |
| "loss": 3.6387, | |
| "step": 2 | |
| }, | |
| { | |
| "epoch": 0.1111111111111111, | |
| "grad_norm": 3.5411174297332764, | |
| "learning_rate": 2.1e-05, | |
| "loss": 3.4665, | |
| "step": 3 | |
| }, | |
| { | |
| "epoch": 0.14814814814814814, | |
| "grad_norm": 3.4620625972747803, | |
| "learning_rate": 2.8e-05, | |
| "loss": 3.4607, | |
| "step": 4 | |
| }, | |
| { | |
| "epoch": 0.18518518518518517, | |
| "grad_norm": 2.918036460876465, | |
| "learning_rate": 3.5e-05, | |
| "loss": 3.4412, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.2222222222222222, | |
| "grad_norm": 3.1347150802612305, | |
| "learning_rate": 4.2e-05, | |
| "loss": 3.4438, | |
| "step": 6 | |
| }, | |
| { | |
| "epoch": 0.25925925925925924, | |
| "grad_norm": 2.0844814777374268, | |
| "learning_rate": 4.899999999999999e-05, | |
| "loss": 2.9182, | |
| "step": 7 | |
| }, | |
| { | |
| "epoch": 0.2962962962962963, | |
| "grad_norm": 2.5144269466400146, | |
| "learning_rate": 5.6e-05, | |
| "loss": 2.7481, | |
| "step": 8 | |
| }, | |
| { | |
| "epoch": 0.3333333333333333, | |
| "grad_norm": 2.326944351196289, | |
| "learning_rate": 6.3e-05, | |
| "loss": 2.8674, | |
| "step": 9 | |
| }, | |
| { | |
| "epoch": 0.37037037037037035, | |
| "grad_norm": 2.9930708408355713, | |
| "learning_rate": 7e-05, | |
| "loss": 2.5017, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.4074074074074074, | |
| "grad_norm": 3.013411045074463, | |
| "learning_rate": 6.996574292819907e-05, | |
| "loss": 2.485, | |
| "step": 11 | |
| }, | |
| { | |
| "epoch": 0.4444444444444444, | |
| "grad_norm": 1.9861669540405273, | |
| "learning_rate": 6.986303877262306e-05, | |
| "loss": 2.0643, | |
| "step": 12 | |
| }, | |
| { | |
| "epoch": 0.48148148148148145, | |
| "grad_norm": 1.5490922927856445, | |
| "learning_rate": 6.969208858147951e-05, | |
| "loss": 1.8952, | |
| "step": 13 | |
| }, | |
| { | |
| "epoch": 0.5185185185185185, | |
| "grad_norm": 1.5672123432159424, | |
| "learning_rate": 6.945322699779538e-05, | |
| "loss": 1.8413, | |
| "step": 14 | |
| }, | |
| { | |
| "epoch": 0.5555555555555556, | |
| "grad_norm": 1.6187469959259033, | |
| "learning_rate": 6.914692160433773e-05, | |
| "loss": 1.8835, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.5925925925925926, | |
| "grad_norm": 1.6432483196258545, | |
| "learning_rate": 6.877377200829835e-05, | |
| "loss": 1.981, | |
| "step": 16 | |
| }, | |
| { | |
| "epoch": 0.6296296296296297, | |
| "grad_norm": 1.4468789100646973, | |
| "learning_rate": 6.83345086675346e-05, | |
| "loss": 1.9871, | |
| "step": 17 | |
| }, | |
| { | |
| "epoch": 0.6666666666666666, | |
| "grad_norm": 1.4680477380752563, | |
| "learning_rate": 6.782999146066386e-05, | |
| "loss": 1.9639, | |
| "step": 18 | |
| }, | |
| { | |
| "epoch": 0.7037037037037037, | |
| "grad_norm": 1.1646554470062256, | |
| "learning_rate": 6.726120800381073e-05, | |
| "loss": 1.825, | |
| "step": 19 | |
| }, | |
| { | |
| "epoch": 0.7407407407407407, | |
| "grad_norm": 1.0831339359283447, | |
| "learning_rate": 6.662927171730213e-05, | |
| "loss": 1.645, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.7777777777777778, | |
| "grad_norm": 1.171111822128296, | |
| "learning_rate": 6.593541964609463e-05, | |
| "loss": 1.6865, | |
| "step": 21 | |
| }, | |
| { | |
| "epoch": 0.8148148148148148, | |
| "grad_norm": 1.1838645935058594, | |
| "learning_rate": 6.518101003820097e-05, | |
| "loss": 1.7186, | |
| "step": 22 | |
| }, | |
| { | |
| "epoch": 0.8518518518518519, | |
| "grad_norm": 1.165795087814331, | |
| "learning_rate": 6.43675196858557e-05, | |
| "loss": 1.9583, | |
| "step": 23 | |
| }, | |
| { | |
| "epoch": 0.8888888888888888, | |
| "grad_norm": 1.2042009830474854, | |
| "learning_rate": 6.34965410346251e-05, | |
| "loss": 2.0201, | |
| "step": 24 | |
| }, | |
| { | |
| "epoch": 0.9259259259259259, | |
| "grad_norm": 1.0208855867385864, | |
| "learning_rate": 6.256977906612013e-05, | |
| "loss": 1.6992, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 0.9629629629629629, | |
| "grad_norm": 1.0845801830291748, | |
| "learning_rate": 6.158904796041496e-05, | |
| "loss": 1.8454, | |
| "step": 26 | |
| }, | |
| { | |
| "epoch": 1.0, | |
| "grad_norm": 1.2094794511795044, | |
| "learning_rate": 6.05562675447042e-05, | |
| "loss": 1.5763, | |
| "step": 27 | |
| }, | |
| { | |
| "epoch": 1.037037037037037, | |
| "grad_norm": 0.9708940386772156, | |
| "learning_rate": 5.947345953515099e-05, | |
| "loss": 1.343, | |
| "step": 28 | |
| }, | |
| { | |
| "epoch": 1.074074074074074, | |
| "grad_norm": 1.04697585105896, | |
| "learning_rate": 5.834274357928277e-05, | |
| "loss": 1.4841, | |
| "step": 29 | |
| }, | |
| { | |
| "epoch": 1.1111111111111112, | |
| "grad_norm": 1.1268396377563477, | |
| "learning_rate": 5.7166333106681644e-05, | |
| "loss": 1.7666, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 1.1481481481481481, | |
| "grad_norm": 1.0179731845855713, | |
| "learning_rate": 5.594653099609201e-05, | |
| "loss": 1.6149, | |
| "step": 31 | |
| }, | |
| { | |
| "epoch": 1.1851851851851851, | |
| "grad_norm": 1.0401766300201416, | |
| "learning_rate": 5.468572506742732e-05, | |
| "loss": 1.4658, | |
| "step": 32 | |
| }, | |
| { | |
| "epoch": 1.2222222222222223, | |
| "grad_norm": 1.0923527479171753, | |
| "learning_rate": 5.338638340750045e-05, | |
| "loss": 1.5151, | |
| "step": 33 | |
| }, | |
| { | |
| "epoch": 1.2592592592592593, | |
| "grad_norm": 1.327430009841919, | |
| "learning_rate": 5.205104953862787e-05, | |
| "loss": 1.4007, | |
| "step": 34 | |
| }, | |
| { | |
| "epoch": 1.2962962962962963, | |
| "grad_norm": 1.1446799039840698, | |
| "learning_rate": 5.068233743956524e-05, | |
| "loss": 1.6688, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 1.3333333333333333, | |
| "grad_norm": 1.2676280736923218, | |
| "learning_rate": 4.9282926428521266e-05, | |
| "loss": 1.8306, | |
| "step": 36 | |
| }, | |
| { | |
| "epoch": 1.3703703703703702, | |
| "grad_norm": 1.1416800022125244, | |
| "learning_rate": 4.785555591826649e-05, | |
| "loss": 1.4489, | |
| "step": 37 | |
| }, | |
| { | |
| "epoch": 1.4074074074074074, | |
| "grad_norm": 1.2351667881011963, | |
| "learning_rate": 4.640302005360411e-05, | |
| "loss": 1.6123, | |
| "step": 38 | |
| }, | |
| { | |
| "epoch": 1.4444444444444444, | |
| "grad_norm": 1.3364763259887695, | |
| "learning_rate": 4.492816224170038e-05, | |
| "loss": 1.7015, | |
| "step": 39 | |
| }, | |
| { | |
| "epoch": 1.4814814814814814, | |
| "grad_norm": 1.1775413751602173, | |
| "learning_rate": 4.34338695859815e-05, | |
| "loss": 1.3681, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 1.5185185185185186, | |
| "grad_norm": 1.271789312362671, | |
| "learning_rate": 4.1923067234493104e-05, | |
| "loss": 1.4819, | |
| "step": 41 | |
| }, | |
| { | |
| "epoch": 1.5555555555555556, | |
| "grad_norm": 1.2430049180984497, | |
| "learning_rate": 4.0398712653785575e-05, | |
| "loss": 1.4588, | |
| "step": 42 | |
| }, | |
| { | |
| "epoch": 1.5925925925925926, | |
| "grad_norm": 1.2405871152877808, | |
| "learning_rate": 3.8863789839534444e-05, | |
| "loss": 1.4588, | |
| "step": 43 | |
| }, | |
| { | |
| "epoch": 1.6296296296296298, | |
| "grad_norm": 1.6753511428833008, | |
| "learning_rate": 3.7321303475228634e-05, | |
| "loss": 1.4858, | |
| "step": 44 | |
| }, | |
| { | |
| "epoch": 1.6666666666666665, | |
| "grad_norm": 1.3085812330245972, | |
| "learning_rate": 3.577427305036154e-05, | |
| "loss": 1.4578, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 1.7037037037037037, | |
| "grad_norm": 1.2051222324371338, | |
| "learning_rate": 3.4225726949638464e-05, | |
| "loss": 1.3037, | |
| "step": 46 | |
| }, | |
| { | |
| "epoch": 1.7407407407407407, | |
| "grad_norm": 1.1844594478607178, | |
| "learning_rate": 3.2678696524771367e-05, | |
| "loss": 1.3121, | |
| "step": 47 | |
| }, | |
| { | |
| "epoch": 1.7777777777777777, | |
| "grad_norm": 1.3920973539352417, | |
| "learning_rate": 3.113621016046556e-05, | |
| "loss": 1.5805, | |
| "step": 48 | |
| }, | |
| { | |
| "epoch": 1.8148148148148149, | |
| "grad_norm": 1.3390107154846191, | |
| "learning_rate": 2.960128734621441e-05, | |
| "loss": 1.2356, | |
| "step": 49 | |
| }, | |
| { | |
| "epoch": 1.8518518518518519, | |
| "grad_norm": 1.55970299243927, | |
| "learning_rate": 2.8076932765506893e-05, | |
| "loss": 1.445, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 1.8518518518518519, | |
| "eval_loss": 1.7665985822677612, | |
| "eval_runtime": 3.3723, | |
| "eval_samples_per_second": 13.64, | |
| "eval_steps_per_second": 3.558, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 1.8888888888888888, | |
| "grad_norm": 1.49898099899292, | |
| "learning_rate": 2.6566130414018495e-05, | |
| "loss": 1.427, | |
| "step": 51 | |
| }, | |
| { | |
| "epoch": 1.925925925925926, | |
| "grad_norm": 1.2517292499542236, | |
| "learning_rate": 2.5071837758299613e-05, | |
| "loss": 1.124, | |
| "step": 52 | |
| }, | |
| { | |
| "epoch": 1.9629629629629628, | |
| "grad_norm": 1.6222271919250488, | |
| "learning_rate": 2.359697994639589e-05, | |
| "loss": 1.318, | |
| "step": 53 | |
| }, | |
| { | |
| "epoch": 2.0, | |
| "grad_norm": 1.829454779624939, | |
| "learning_rate": 2.2144444081733517e-05, | |
| "loss": 1.7384, | |
| "step": 54 | |
| }, | |
| { | |
| "epoch": 2.037037037037037, | |
| "grad_norm": 1.2274819612503052, | |
| "learning_rate": 2.071707357147872e-05, | |
| "loss": 0.9252, | |
| "step": 55 | |
| }, | |
| { | |
| "epoch": 2.074074074074074, | |
| "grad_norm": 1.5283606052398682, | |
| "learning_rate": 1.931766256043475e-05, | |
| "loss": 1.0436, | |
| "step": 56 | |
| }, | |
| { | |
| "epoch": 2.111111111111111, | |
| "grad_norm": 1.5208336114883423, | |
| "learning_rate": 1.7948950461372128e-05, | |
| "loss": 1.2807, | |
| "step": 57 | |
| }, | |
| { | |
| "epoch": 2.148148148148148, | |
| "grad_norm": 1.5189377069473267, | |
| "learning_rate": 1.6613616592499547e-05, | |
| "loss": 0.9818, | |
| "step": 58 | |
| }, | |
| { | |
| "epoch": 2.185185185185185, | |
| "grad_norm": 1.881172776222229, | |
| "learning_rate": 1.5314274932572676e-05, | |
| "loss": 1.1691, | |
| "step": 59 | |
| }, | |
| { | |
| "epoch": 2.2222222222222223, | |
| "grad_norm": 1.8424291610717773, | |
| "learning_rate": 1.4053469003907992e-05, | |
| "loss": 1.246, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 2.259259259259259, | |
| "grad_norm": 1.9025232791900635, | |
| "learning_rate": 1.2833666893318349e-05, | |
| "loss": 1.0384, | |
| "step": 61 | |
| }, | |
| { | |
| "epoch": 2.2962962962962963, | |
| "grad_norm": 1.7848095893859863, | |
| "learning_rate": 1.165725642071722e-05, | |
| "loss": 1.0459, | |
| "step": 62 | |
| }, | |
| { | |
| "epoch": 2.3333333333333335, | |
| "grad_norm": 1.7614284753799438, | |
| "learning_rate": 1.0526540464849008e-05, | |
| "loss": 1.1616, | |
| "step": 63 | |
| }, | |
| { | |
| "epoch": 2.3703703703703702, | |
| "grad_norm": 2.083524465560913, | |
| "learning_rate": 9.443732455295803e-06, | |
| "loss": 1.2212, | |
| "step": 64 | |
| }, | |
| { | |
| "epoch": 2.4074074074074074, | |
| "grad_norm": 2.120060682296753, | |
| "learning_rate": 8.410952039585034e-06, | |
| "loss": 1.2193, | |
| "step": 65 | |
| }, | |
| { | |
| "epoch": 2.4444444444444446, | |
| "grad_norm": 1.8416856527328491, | |
| "learning_rate": 7.430220933879868e-06, | |
| "loss": 1.0275, | |
| "step": 66 | |
| }, | |
| { | |
| "epoch": 2.4814814814814814, | |
| "grad_norm": 1.6970349550247192, | |
| "learning_rate": 6.503458965374907e-06, | |
| "loss": 0.9866, | |
| "step": 67 | |
| }, | |
| { | |
| "epoch": 2.5185185185185186, | |
| "grad_norm": 1.7311978340148926, | |
| "learning_rate": 5.632480314144302e-06, | |
| "loss": 1.1222, | |
| "step": 68 | |
| }, | |
| { | |
| "epoch": 2.5555555555555554, | |
| "grad_norm": 1.8647996187210083, | |
| "learning_rate": 4.818989961799024e-06, | |
| "loss": 1.1971, | |
| "step": 69 | |
| }, | |
| { | |
| "epoch": 2.5925925925925926, | |
| "grad_norm": 1.8562637567520142, | |
| "learning_rate": 4.064580353905361e-06, | |
| "loss": 1.0966, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 2.6296296296296298, | |
| "grad_norm": 2.1647939682006836, | |
| "learning_rate": 3.3707282826978684e-06, | |
| "loss": 1.2084, | |
| "step": 71 | |
| }, | |
| { | |
| "epoch": 2.6666666666666665, | |
| "grad_norm": 2.267409563064575, | |
| "learning_rate": 2.7387919961892603e-06, | |
| "loss": 1.2536, | |
| "step": 72 | |
| }, | |
| { | |
| "epoch": 2.7037037037037037, | |
| "grad_norm": 1.875108003616333, | |
| "learning_rate": 2.170008539336139e-06, | |
| "loss": 1.0385, | |
| "step": 73 | |
| }, | |
| { | |
| "epoch": 2.7407407407407405, | |
| "grad_norm": 1.690350890159607, | |
| "learning_rate": 1.665491332465404e-06, | |
| "loss": 0.9002, | |
| "step": 74 | |
| }, | |
| { | |
| "epoch": 2.7777777777777777, | |
| "grad_norm": 1.8636929988861084, | |
| "learning_rate": 1.2262279917016548e-06, | |
| "loss": 1.1853, | |
| "step": 75 | |
| }, | |
| { | |
| "epoch": 2.814814814814815, | |
| "grad_norm": 1.7024513483047485, | |
| "learning_rate": 8.530783956622628e-07, | |
| "loss": 1.1387, | |
| "step": 76 | |
| }, | |
| { | |
| "epoch": 2.851851851851852, | |
| "grad_norm": 1.7570838928222656, | |
| "learning_rate": 5.467730022046046e-07, | |
| "loss": 0.9747, | |
| "step": 77 | |
| }, | |
| { | |
| "epoch": 2.888888888888889, | |
| "grad_norm": 1.8747435808181763, | |
| "learning_rate": 3.0791141852049006e-07, | |
| "loss": 1.1896, | |
| "step": 78 | |
| }, | |
| { | |
| "epoch": 2.925925925925926, | |
| "grad_norm": 1.7294906377792358, | |
| "learning_rate": 1.369612273769316e-07, | |
| "loss": 0.9973, | |
| "step": 79 | |
| }, | |
| { | |
| "epoch": 2.962962962962963, | |
| "grad_norm": 1.958431363105774, | |
| "learning_rate": 3.4257071800923855e-08, | |
| "loss": 1.2993, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 3.0, | |
| "grad_norm": 1.917493462562561, | |
| "learning_rate": 0.0, | |
| "loss": 1.2325, | |
| "step": 81 | |
| } | |
| ], | |
| "logging_steps": 1, | |
| "max_steps": 81, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 3, | |
| "save_steps": 50, | |
| "stateful_callbacks": { | |
| "EarlyStoppingCallback": { | |
| "args": { | |
| "early_stopping_patience": 4, | |
| "early_stopping_threshold": 0.0 | |
| }, | |
| "attributes": { | |
| "early_stopping_patience_counter": 0 | |
| } | |
| }, | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 1.2072723618988032e+17, | |
| "train_batch_size": 8, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |