Instructions to use sindhusatish97/sparq-qwen2.5-14b-continued-sft with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use sindhusatish97/sparq-qwen2.5-14b-continued-sft with PEFT:
Base model is not found.
- Notebooks
- Google Colab
- Kaggle
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 0.37537285920166236, | |
| "eval_steps": 500, | |
| "global_step": 1400, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.005362469417166605, | |
| "grad_norm": 0.050072263926267624, | |
| "learning_rate": 1.4961796246648793e-05, | |
| "loss": 1.0673207283020019, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.01072493883433321, | |
| "grad_norm": 0.06825340539216995, | |
| "learning_rate": 1.4921581769436997e-05, | |
| "loss": 0.9185627937316895, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.016087408251499815, | |
| "grad_norm": 0.06827432662248611, | |
| "learning_rate": 1.48813672922252e-05, | |
| "loss": 0.7999343872070312, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.02144987766866642, | |
| "grad_norm": 0.05807405710220337, | |
| "learning_rate": 1.4841152815013404e-05, | |
| "loss": 0.7322770595550537, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.026812347085833025, | |
| "grad_norm": 0.06654328852891922, | |
| "learning_rate": 1.4800938337801608e-05, | |
| "loss": 0.7097890377044678, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.03217481650299963, | |
| "grad_norm": 0.09104783087968826, | |
| "learning_rate": 1.4760723860589812e-05, | |
| "loss": 0.6513629913330078, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.03753728592016624, | |
| "grad_norm": 0.10718850791454315, | |
| "learning_rate": 1.4720509383378015e-05, | |
| "loss": 0.678717851638794, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.04289975533733284, | |
| "grad_norm": 0.09187154471874237, | |
| "learning_rate": 1.4680294906166219e-05, | |
| "loss": 0.647278118133545, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.04826222475449945, | |
| "grad_norm": 0.07148946076631546, | |
| "learning_rate": 1.4640080428954423e-05, | |
| "loss": 0.6737877368927002, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.05362469417166605, | |
| "grad_norm": 0.08909227699041367, | |
| "learning_rate": 1.4599865951742626e-05, | |
| "loss": 0.6373191356658936, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.05898716358883266, | |
| "grad_norm": 0.07850278168916702, | |
| "learning_rate": 1.455965147453083e-05, | |
| "loss": 0.6020126819610596, | |
| "step": 220 | |
| }, | |
| { | |
| "epoch": 0.06434963300599926, | |
| "grad_norm": 0.09538089483976364, | |
| "learning_rate": 1.4519436997319034e-05, | |
| "loss": 0.6096773147583008, | |
| "step": 240 | |
| }, | |
| { | |
| "epoch": 0.06971210242316586, | |
| "grad_norm": 0.07478228211402893, | |
| "learning_rate": 1.447922252010724e-05, | |
| "loss": 0.6299086093902588, | |
| "step": 260 | |
| }, | |
| { | |
| "epoch": 0.07507457184033248, | |
| "grad_norm": 0.1514953374862671, | |
| "learning_rate": 1.4439008042895443e-05, | |
| "loss": 0.5591042518615723, | |
| "step": 280 | |
| }, | |
| { | |
| "epoch": 0.08043704125749908, | |
| "grad_norm": 0.08260886371135712, | |
| "learning_rate": 1.4398793565683647e-05, | |
| "loss": 0.6200376987457276, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.08579951067466568, | |
| "grad_norm": 0.17698714137077332, | |
| "learning_rate": 1.435857908847185e-05, | |
| "loss": 0.6023219585418701, | |
| "step": 320 | |
| }, | |
| { | |
| "epoch": 0.0911619800918323, | |
| "grad_norm": 0.06104859337210655, | |
| "learning_rate": 1.4318364611260054e-05, | |
| "loss": 0.6181454658508301, | |
| "step": 340 | |
| }, | |
| { | |
| "epoch": 0.0965244495089989, | |
| "grad_norm": 0.04990549385547638, | |
| "learning_rate": 1.4278150134048258e-05, | |
| "loss": 0.5593632698059082, | |
| "step": 360 | |
| }, | |
| { | |
| "epoch": 0.1018869189261655, | |
| "grad_norm": 0.09426380693912506, | |
| "learning_rate": 1.4237935656836461e-05, | |
| "loss": 0.5790591716766358, | |
| "step": 380 | |
| }, | |
| { | |
| "epoch": 0.1072493883433321, | |
| "grad_norm": 0.08783263713121414, | |
| "learning_rate": 1.4197721179624665e-05, | |
| "loss": 0.585063886642456, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.11261185776049872, | |
| "grad_norm": 0.06869607418775558, | |
| "learning_rate": 1.4157506702412869e-05, | |
| "loss": 0.5638764381408692, | |
| "step": 420 | |
| }, | |
| { | |
| "epoch": 0.11797432717766532, | |
| "grad_norm": 0.10537438839673996, | |
| "learning_rate": 1.4117292225201072e-05, | |
| "loss": 0.6060166835784913, | |
| "step": 440 | |
| }, | |
| { | |
| "epoch": 0.12333679659483192, | |
| "grad_norm": 0.09851580113172531, | |
| "learning_rate": 1.4077077747989278e-05, | |
| "loss": 0.5605969905853272, | |
| "step": 460 | |
| }, | |
| { | |
| "epoch": 0.12869926601199852, | |
| "grad_norm": 0.11954096704721451, | |
| "learning_rate": 1.4036863270777482e-05, | |
| "loss": 0.5549856662750244, | |
| "step": 480 | |
| }, | |
| { | |
| "epoch": 0.13406173542916514, | |
| "grad_norm": 0.13259431719779968, | |
| "learning_rate": 1.3996648793565685e-05, | |
| "loss": 0.5893547534942627, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 0.13942420484633172, | |
| "grad_norm": 0.11842650175094604, | |
| "learning_rate": 1.3956434316353889e-05, | |
| "loss": 0.6237683773040772, | |
| "step": 520 | |
| }, | |
| { | |
| "epoch": 0.14478667426349834, | |
| "grad_norm": 0.1204022690653801, | |
| "learning_rate": 1.3916219839142093e-05, | |
| "loss": 0.572803258895874, | |
| "step": 540 | |
| }, | |
| { | |
| "epoch": 0.15014914368066495, | |
| "grad_norm": 0.1345946341753006, | |
| "learning_rate": 1.3876005361930296e-05, | |
| "loss": 0.5632933139801025, | |
| "step": 560 | |
| }, | |
| { | |
| "epoch": 0.15551161309783154, | |
| "grad_norm": 0.11733393371105194, | |
| "learning_rate": 1.38357908847185e-05, | |
| "loss": 0.6197309494018555, | |
| "step": 580 | |
| }, | |
| { | |
| "epoch": 0.16087408251499816, | |
| "grad_norm": 0.0731734186410904, | |
| "learning_rate": 1.3795576407506704e-05, | |
| "loss": 0.5823808670043945, | |
| "step": 600 | |
| }, | |
| { | |
| "epoch": 0.16623655193216477, | |
| "grad_norm": 0.09452618658542633, | |
| "learning_rate": 1.3755361930294907e-05, | |
| "loss": 0.5599356651306152, | |
| "step": 620 | |
| }, | |
| { | |
| "epoch": 0.17159902134933136, | |
| "grad_norm": 0.09183815121650696, | |
| "learning_rate": 1.3715147453083111e-05, | |
| "loss": 0.5465828895568847, | |
| "step": 640 | |
| }, | |
| { | |
| "epoch": 0.17696149076649798, | |
| "grad_norm": 0.0953364372253418, | |
| "learning_rate": 1.3674932975871315e-05, | |
| "loss": 0.5516108989715576, | |
| "step": 660 | |
| }, | |
| { | |
| "epoch": 0.1823239601836646, | |
| "grad_norm": 0.11190114170312881, | |
| "learning_rate": 1.3634718498659519e-05, | |
| "loss": 0.5717048645019531, | |
| "step": 680 | |
| }, | |
| { | |
| "epoch": 0.18768642960083118, | |
| "grad_norm": 0.11502158641815186, | |
| "learning_rate": 1.3594504021447722e-05, | |
| "loss": 0.528355598449707, | |
| "step": 700 | |
| }, | |
| { | |
| "epoch": 0.1930488990179978, | |
| "grad_norm": 0.12480133026838303, | |
| "learning_rate": 1.3554289544235926e-05, | |
| "loss": 0.5860391616821289, | |
| "step": 720 | |
| }, | |
| { | |
| "epoch": 0.19841136843516438, | |
| "grad_norm": 0.14408785104751587, | |
| "learning_rate": 1.351407506702413e-05, | |
| "loss": 0.5422697544097901, | |
| "step": 740 | |
| }, | |
| { | |
| "epoch": 0.203773837852331, | |
| "grad_norm": 0.12405668199062347, | |
| "learning_rate": 1.3473860589812333e-05, | |
| "loss": 0.5876667499542236, | |
| "step": 760 | |
| }, | |
| { | |
| "epoch": 0.2091363072694976, | |
| "grad_norm": 0.12171291559934616, | |
| "learning_rate": 1.3433646112600537e-05, | |
| "loss": 0.563751220703125, | |
| "step": 780 | |
| }, | |
| { | |
| "epoch": 0.2144987766866642, | |
| "grad_norm": 0.10827518254518509, | |
| "learning_rate": 1.339343163538874e-05, | |
| "loss": 0.5700247764587403, | |
| "step": 800 | |
| }, | |
| { | |
| "epoch": 0.21986124610383082, | |
| "grad_norm": 0.08678701519966125, | |
| "learning_rate": 1.3353217158176944e-05, | |
| "loss": 0.5999309062957764, | |
| "step": 820 | |
| }, | |
| { | |
| "epoch": 0.22522371552099743, | |
| "grad_norm": 0.12222636491060257, | |
| "learning_rate": 1.3313002680965148e-05, | |
| "loss": 0.5421838760375977, | |
| "step": 840 | |
| }, | |
| { | |
| "epoch": 0.23058618493816402, | |
| "grad_norm": 0.11634483933448792, | |
| "learning_rate": 1.3272788203753352e-05, | |
| "loss": 0.6069926261901856, | |
| "step": 860 | |
| }, | |
| { | |
| "epoch": 0.23594865435533063, | |
| "grad_norm": 0.12163955718278885, | |
| "learning_rate": 1.3232573726541556e-05, | |
| "loss": 0.5558357238769531, | |
| "step": 880 | |
| }, | |
| { | |
| "epoch": 0.24131112377249722, | |
| "grad_norm": 0.13140572607517242, | |
| "learning_rate": 1.319235924932976e-05, | |
| "loss": 0.5537341117858887, | |
| "step": 900 | |
| }, | |
| { | |
| "epoch": 0.24667359318966384, | |
| "grad_norm": 0.1295424848794937, | |
| "learning_rate": 1.3152144772117963e-05, | |
| "loss": 0.5734247684478759, | |
| "step": 920 | |
| }, | |
| { | |
| "epoch": 0.2520360626068304, | |
| "grad_norm": 0.08855397999286652, | |
| "learning_rate": 1.3111930294906167e-05, | |
| "loss": 0.5499854564666748, | |
| "step": 940 | |
| }, | |
| { | |
| "epoch": 0.25739853202399704, | |
| "grad_norm": 0.10895389318466187, | |
| "learning_rate": 1.307171581769437e-05, | |
| "loss": 0.4994966506958008, | |
| "step": 960 | |
| }, | |
| { | |
| "epoch": 0.26276100144116366, | |
| "grad_norm": 0.10110122710466385, | |
| "learning_rate": 1.3031501340482574e-05, | |
| "loss": 0.5803254604339599, | |
| "step": 980 | |
| }, | |
| { | |
| "epoch": 0.26812347085833027, | |
| "grad_norm": 0.1323656141757965, | |
| "learning_rate": 1.2991286863270778e-05, | |
| "loss": 0.5268758773803711, | |
| "step": 1000 | |
| }, | |
| { | |
| "epoch": 0.2734859402754969, | |
| "grad_norm": 0.09068968147039413, | |
| "learning_rate": 1.2951072386058981e-05, | |
| "loss": 0.5150487899780274, | |
| "step": 1020 | |
| }, | |
| { | |
| "epoch": 0.27884840969266345, | |
| "grad_norm": 0.11400057375431061, | |
| "learning_rate": 1.2910857908847185e-05, | |
| "loss": 0.5365507125854492, | |
| "step": 1040 | |
| }, | |
| { | |
| "epoch": 0.28421087910983006, | |
| "grad_norm": 0.14133770763874054, | |
| "learning_rate": 1.2870643431635389e-05, | |
| "loss": 0.5134270668029786, | |
| "step": 1060 | |
| }, | |
| { | |
| "epoch": 0.2895733485269967, | |
| "grad_norm": 0.14621631801128387, | |
| "learning_rate": 1.2830428954423593e-05, | |
| "loss": 0.5870331287384033, | |
| "step": 1080 | |
| }, | |
| { | |
| "epoch": 0.2949358179441633, | |
| "grad_norm": 0.09397239238023758, | |
| "learning_rate": 1.2790214477211796e-05, | |
| "loss": 0.5265964984893798, | |
| "step": 1100 | |
| }, | |
| { | |
| "epoch": 0.3002982873613299, | |
| "grad_norm": 0.13457220792770386, | |
| "learning_rate": 1.275e-05, | |
| "loss": 0.541674280166626, | |
| "step": 1120 | |
| }, | |
| { | |
| "epoch": 0.3056607567784965, | |
| "grad_norm": 0.11553078144788742, | |
| "learning_rate": 1.2709785522788204e-05, | |
| "loss": 0.5721035003662109, | |
| "step": 1140 | |
| }, | |
| { | |
| "epoch": 0.3110232261956631, | |
| "grad_norm": 0.08464279770851135, | |
| "learning_rate": 1.2669571045576407e-05, | |
| "loss": 0.5242496967315674, | |
| "step": 1160 | |
| }, | |
| { | |
| "epoch": 0.3163856956128297, | |
| "grad_norm": 0.11578533798456192, | |
| "learning_rate": 1.2629356568364611e-05, | |
| "loss": 0.5268265724182128, | |
| "step": 1180 | |
| }, | |
| { | |
| "epoch": 0.3217481650299963, | |
| "grad_norm": 0.10422660410404205, | |
| "learning_rate": 1.2589142091152815e-05, | |
| "loss": 0.5755553722381592, | |
| "step": 1200 | |
| }, | |
| { | |
| "epoch": 0.32711063444716293, | |
| "grad_norm": 0.1601565182209015, | |
| "learning_rate": 1.2548927613941018e-05, | |
| "loss": 0.572784423828125, | |
| "step": 1220 | |
| }, | |
| { | |
| "epoch": 0.33247310386432954, | |
| "grad_norm": 0.1435895711183548, | |
| "learning_rate": 1.2508713136729222e-05, | |
| "loss": 0.4759331703186035, | |
| "step": 1240 | |
| }, | |
| { | |
| "epoch": 0.3378355732814961, | |
| "grad_norm": 0.13164320588111877, | |
| "learning_rate": 1.2468498659517426e-05, | |
| "loss": 0.5674447059631348, | |
| "step": 1260 | |
| }, | |
| { | |
| "epoch": 0.3431980426986627, | |
| "grad_norm": 0.17907585203647614, | |
| "learning_rate": 1.242828418230563e-05, | |
| "loss": 0.5384601593017578, | |
| "step": 1280 | |
| }, | |
| { | |
| "epoch": 0.34856051211582934, | |
| "grad_norm": 0.1515372097492218, | |
| "learning_rate": 1.2388069705093833e-05, | |
| "loss": 0.5154921531677246, | |
| "step": 1300 | |
| }, | |
| { | |
| "epoch": 0.35392298153299595, | |
| "grad_norm": 0.13605119287967682, | |
| "learning_rate": 1.2347855227882037e-05, | |
| "loss": 0.5586633205413818, | |
| "step": 1320 | |
| }, | |
| { | |
| "epoch": 0.35928545095016257, | |
| "grad_norm": 0.12003476917743683, | |
| "learning_rate": 1.230764075067024e-05, | |
| "loss": 0.5512509822845459, | |
| "step": 1340 | |
| }, | |
| { | |
| "epoch": 0.3646479203673292, | |
| "grad_norm": 0.11852169036865234, | |
| "learning_rate": 1.2267426273458444e-05, | |
| "loss": 0.5680348873138428, | |
| "step": 1360 | |
| }, | |
| { | |
| "epoch": 0.37001038978449574, | |
| "grad_norm": 0.16344694793224335, | |
| "learning_rate": 1.2227211796246648e-05, | |
| "loss": 0.5669443130493164, | |
| "step": 1380 | |
| }, | |
| { | |
| "epoch": 0.37537285920166236, | |
| "grad_norm": 0.11730384081602097, | |
| "learning_rate": 1.2186997319034852e-05, | |
| "loss": 0.5089732646942139, | |
| "step": 1400 | |
| } | |
| ], | |
| "logging_steps": 20, | |
| "max_steps": 7460, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 2, | |
| "save_steps": 200, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 1.716587745411932e+17, | |
| "train_batch_size": 1, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |