Download last-checkpoint/trainer_state.json from CodeIsAbstract/HybridModelScratch_testkaggle: direct link, hf CLI and curl.
- Browser
- Download file 15.6 kB
-
https://huggingface.co/CodeIsAbstract/HybridModelScratch_testkaggle/resolve/main/last-checkpoint/trainer_state.json
- Command line
-
hf download hf://CodeIsAbstract/HybridModelScratch_testkaggle/last-checkpoint/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/CodeIsAbstract/HybridModelScratch_testkaggle/resolve/main/last-checkpoint/trainer_state.json
15.6 kB
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 0.16, | |
| "eval_steps": 1000, | |
| "global_step": 8000, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": false, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.002, | |
| "grad_norm": 4.729869365692139, | |
| "learning_rate": 0.00033, | |
| "loss": 64.03521484375, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.004, | |
| "grad_norm": 7.427474021911621, | |
| "learning_rate": 0.0004999988080137436, | |
| "loss": 50.4660400390625, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.006, | |
| "grad_norm": 499.43450927734375, | |
| "learning_rate": 0.0004999889782951015, | |
| "loss": 55.620224609375, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.008, | |
| "grad_norm": 5.543099880218506, | |
| "learning_rate": 0.0004999692199574857, | |
| "loss": 55.512568359375, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.01, | |
| "grad_norm": 3.913767099380493, | |
| "learning_rate": 0.0004999395337856224, | |
| "loss": 54.5164111328125, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 0.012, | |
| "grad_norm": 3.7103848457336426, | |
| "learning_rate": 0.0004998999209585345, | |
| "loss": 48.3706103515625, | |
| "step": 600 | |
| }, | |
| { | |
| "epoch": 0.014, | |
| "grad_norm": 27.09828758239746, | |
| "learning_rate": 0.0004998503830494941, | |
| "loss": 48.7948681640625, | |
| "step": 700 | |
| }, | |
| { | |
| "epoch": 0.016, | |
| "grad_norm": 357.75579833984375, | |
| "learning_rate": 0.0004997909220259598, | |
| "loss": 47.264482421875, | |
| "step": 800 | |
| }, | |
| { | |
| "epoch": 0.018, | |
| "grad_norm": 21.32952117919922, | |
| "learning_rate": 0.0004997215402494993, | |
| "loss": 46.945810546875, | |
| "step": 900 | |
| }, | |
| { | |
| "epoch": 0.02, | |
| "grad_norm": 1339.14501953125, | |
| "learning_rate": 0.0004996422404756948, | |
| "loss": 50.6924609375, | |
| "step": 1000 | |
| }, | |
| { | |
| "epoch": 0.02, | |
| "eval_loss": 62.66343307495117, | |
| "eval_runtime": 11.6524, | |
| "eval_samples_per_second": 1.716, | |
| "eval_steps_per_second": 0.086, | |
| "step": 1000 | |
| }, | |
| { | |
| "epoch": 0.022, | |
| "grad_norm": 29.14947509765625, | |
| "learning_rate": 0.0004995530258540343, | |
| "loss": 49.645908203125, | |
| "step": 1100 | |
| }, | |
| { | |
| "epoch": 0.024, | |
| "grad_norm": 199.71112060546875, | |
| "learning_rate": 0.0004994538999277859, | |
| "loss": 46.0685693359375, | |
| "step": 1200 | |
| }, | |
| { | |
| "epoch": 0.026, | |
| "grad_norm": 26.30093002319336, | |
| "learning_rate": 0.0004993448666338572, | |
| "loss": 45.418486328125, | |
| "step": 1300 | |
| }, | |
| { | |
| "epoch": 0.028, | |
| "grad_norm": 72.15017700195312, | |
| "learning_rate": 0.0004992259303026396, | |
| "loss": 46.6232666015625, | |
| "step": 1400 | |
| }, | |
| { | |
| "epoch": 0.03, | |
| "grad_norm": 16.536649703979492, | |
| "learning_rate": 0.0004990970956578351, | |
| "loss": 52.807451171875, | |
| "step": 1500 | |
| }, | |
| { | |
| "epoch": 0.032, | |
| "grad_norm": 75235.25, | |
| "learning_rate": 0.0004989583678162697, | |
| "loss": 58.359990234375, | |
| "step": 1600 | |
| }, | |
| { | |
| "epoch": 0.034, | |
| "grad_norm": 7981.73974609375, | |
| "learning_rate": 0.0004988097522876898, | |
| "loss": 58.0731103515625, | |
| "step": 1700 | |
| }, | |
| { | |
| "epoch": 0.036, | |
| "grad_norm": 10026.5595703125, | |
| "learning_rate": 0.0004986512549745438, | |
| "loss": 59.05083984375, | |
| "step": 1800 | |
| }, | |
| { | |
| "epoch": 0.038, | |
| "grad_norm": 534.1368408203125, | |
| "learning_rate": 0.0004984828821717466, | |
| "loss": 54.8443310546875, | |
| "step": 1900 | |
| }, | |
| { | |
| "epoch": 0.04, | |
| "grad_norm": 423.7434997558594, | |
| "learning_rate": 0.0004983046405664306, | |
| "loss": 54.6559326171875, | |
| "step": 2000 | |
| }, | |
| { | |
| "epoch": 0.04, | |
| "eval_loss": 55.864173889160156, | |
| "eval_runtime": 1.8383, | |
| "eval_samples_per_second": 10.88, | |
| "eval_steps_per_second": 0.544, | |
| "step": 2000 | |
| }, | |
| { | |
| "epoch": 0.042, | |
| "grad_norm": 7411.40234375, | |
| "learning_rate": 0.0004981165372376802, | |
| "loss": 58.2853564453125, | |
| "step": 2100 | |
| }, | |
| { | |
| "epoch": 0.044, | |
| "grad_norm": 33.24531173706055, | |
| "learning_rate": 0.0004979185796562494, | |
| "loss": 57.3351123046875, | |
| "step": 2200 | |
| }, | |
| { | |
| "epoch": 0.046, | |
| "grad_norm": 1044.70458984375, | |
| "learning_rate": 0.0004977107756842668, | |
| "loss": 55.583466796875, | |
| "step": 2300 | |
| }, | |
| { | |
| "epoch": 0.048, | |
| "grad_norm": 488.1427001953125, | |
| "learning_rate": 0.0004974931335749219, | |
| "loss": 53.566806640625, | |
| "step": 2400 | |
| }, | |
| { | |
| "epoch": 0.05, | |
| "grad_norm": 949.36181640625, | |
| "learning_rate": 0.000497265661972138, | |
| "loss": 52.38724609375, | |
| "step": 2500 | |
| }, | |
| { | |
| "epoch": 0.052, | |
| "grad_norm": 1916.361328125, | |
| "learning_rate": 0.0004970283699102291, | |
| "loss": 55.593505859375, | |
| "step": 2600 | |
| }, | |
| { | |
| "epoch": 0.054, | |
| "grad_norm": 40184.4609375, | |
| "learning_rate": 0.0004967812668135405, | |
| "loss": 59.1807861328125, | |
| "step": 2700 | |
| }, | |
| { | |
| "epoch": 0.056, | |
| "grad_norm": 1653.9197998046875, | |
| "learning_rate": 0.0004965243624960747, | |
| "loss": 60.0856396484375, | |
| "step": 2800 | |
| }, | |
| { | |
| "epoch": 0.058, | |
| "grad_norm": 570.0091552734375, | |
| "learning_rate": 0.0004962576671611021, | |
| "loss": 55.3298486328125, | |
| "step": 2900 | |
| }, | |
| { | |
| "epoch": 0.06, | |
| "grad_norm": 767297.5625, | |
| "learning_rate": 0.000495981191400755, | |
| "loss": 59.248984375, | |
| "step": 3000 | |
| }, | |
| { | |
| "epoch": 0.06, | |
| "eval_loss": 65.7476806640625, | |
| "eval_runtime": 1.8469, | |
| "eval_samples_per_second": 10.829, | |
| "eval_steps_per_second": 0.541, | |
| "step": 3000 | |
| }, | |
| { | |
| "epoch": 0.062, | |
| "grad_norm": 1796894.75, | |
| "learning_rate": 0.0004956949461956074, | |
| "loss": 61.4278955078125, | |
| "step": 3100 | |
| }, | |
| { | |
| "epoch": 0.064, | |
| "grad_norm": 372332.40625, | |
| "learning_rate": 0.0004953989429142387, | |
| "loss": 58.2212158203125, | |
| "step": 3200 | |
| }, | |
| { | |
| "epoch": 0.066, | |
| "grad_norm": 839582.25, | |
| "learning_rate": 0.0004950931933127826, | |
| "loss": 60.9759326171875, | |
| "step": 3300 | |
| }, | |
| { | |
| "epoch": 0.068, | |
| "grad_norm": 456507.9375, | |
| "learning_rate": 0.0004947777095344596, | |
| "loss": 60.5664697265625, | |
| "step": 3400 | |
| }, | |
| { | |
| "epoch": 0.07, | |
| "grad_norm": 42298.0859375, | |
| "learning_rate": 0.0004944525041090948, | |
| "loss": 59.9312353515625, | |
| "step": 3500 | |
| }, | |
| { | |
| "epoch": 0.072, | |
| "grad_norm": 44746.13671875, | |
| "learning_rate": 0.0004941175899526208, | |
| "loss": 60.0662353515625, | |
| "step": 3600 | |
| }, | |
| { | |
| "epoch": 0.074, | |
| "grad_norm": 6649.8818359375, | |
| "learning_rate": 0.0004937729803665643, | |
| "loss": 61.2509716796875, | |
| "step": 3700 | |
| }, | |
| { | |
| "epoch": 0.076, | |
| "grad_norm": 15284.4833984375, | |
| "learning_rate": 0.0004934186890375175, | |
| "loss": 59.5713232421875, | |
| "step": 3800 | |
| }, | |
| { | |
| "epoch": 0.078, | |
| "grad_norm": 13539.5712890625, | |
| "learning_rate": 0.0004930547300365956, | |
| "loss": 58.5672265625, | |
| "step": 3900 | |
| }, | |
| { | |
| "epoch": 0.08, | |
| "grad_norm": 53576.8984375, | |
| "learning_rate": 0.0004926811178188765, | |
| "loss": 58.3544189453125, | |
| "step": 4000 | |
| }, | |
| { | |
| "epoch": 0.08, | |
| "eval_loss": 58.579750061035156, | |
| "eval_runtime": 1.8442, | |
| "eval_samples_per_second": 10.845, | |
| "eval_steps_per_second": 0.542, | |
| "step": 4000 | |
| }, | |
| { | |
| "epoch": 0.082, | |
| "grad_norm": 160393.71875, | |
| "learning_rate": 0.000492297867222828, | |
| "loss": 59.3321435546875, | |
| "step": 4100 | |
| }, | |
| { | |
| "epoch": 0.084, | |
| "grad_norm": 12930.431640625, | |
| "learning_rate": 0.0004919049934697177, | |
| "loss": 60.8889404296875, | |
| "step": 4200 | |
| }, | |
| { | |
| "epoch": 0.086, | |
| "grad_norm": 4836.6787109375, | |
| "learning_rate": 0.0004915025121630086, | |
| "loss": 61.493935546875, | |
| "step": 4300 | |
| }, | |
| { | |
| "epoch": 0.088, | |
| "grad_norm": 595.083984375, | |
| "learning_rate": 0.0004910904392877396, | |
| "loss": 58.8386474609375, | |
| "step": 4400 | |
| }, | |
| { | |
| "epoch": 0.09, | |
| "grad_norm": 12180.2373046875, | |
| "learning_rate": 0.0004906687912098906, | |
| "loss": 63.4604150390625, | |
| "step": 4500 | |
| }, | |
| { | |
| "epoch": 0.092, | |
| "grad_norm": 414.4292297363281, | |
| "learning_rate": 0.0004902375846757322, | |
| "loss": 59.018310546875, | |
| "step": 4600 | |
| }, | |
| { | |
| "epoch": 0.094, | |
| "grad_norm": 2297.899169921875, | |
| "learning_rate": 0.0004897968368111611, | |
| "loss": 57.6424853515625, | |
| "step": 4700 | |
| }, | |
| { | |
| "epoch": 0.096, | |
| "grad_norm": 2821.689697265625, | |
| "learning_rate": 0.0004893465651210193, | |
| "loss": 58.3983154296875, | |
| "step": 4800 | |
| }, | |
| { | |
| "epoch": 0.098, | |
| "grad_norm": 2559.970947265625, | |
| "learning_rate": 0.0004888867874883995, | |
| "loss": 57.3033935546875, | |
| "step": 4900 | |
| }, | |
| { | |
| "epoch": 0.1, | |
| "grad_norm": 8444.46484375, | |
| "learning_rate": 0.0004884175221739343, | |
| "loss": 56.6698193359375, | |
| "step": 5000 | |
| }, | |
| { | |
| "epoch": 0.1, | |
| "eval_loss": 57.39847946166992, | |
| "eval_runtime": 1.8416, | |
| "eval_samples_per_second": 10.86, | |
| "eval_steps_per_second": 0.543, | |
| "step": 5000 | |
| }, | |
| { | |
| "epoch": 0.102, | |
| "grad_norm": 6454.49658203125, | |
| "learning_rate": 0.0004879387878150716, | |
| "loss": 56.9357958984375, | |
| "step": 5100 | |
| }, | |
| { | |
| "epoch": 0.104, | |
| "grad_norm": 337538.0, | |
| "learning_rate": 0.00048745060342533363, | |
| "loss": 56.982119140625, | |
| "step": 5200 | |
| }, | |
| { | |
| "epoch": 0.106, | |
| "grad_norm": 68615.4609375, | |
| "learning_rate": 0.0004869529883935625, | |
| "loss": 56.92587890625, | |
| "step": 5300 | |
| }, | |
| { | |
| "epoch": 0.108, | |
| "grad_norm": 116072.59375, | |
| "learning_rate": 0.00048644596248314967, | |
| "loss": 57.190751953125, | |
| "step": 5400 | |
| }, | |
| { | |
| "epoch": 0.11, | |
| "grad_norm": 291359.625, | |
| "learning_rate": 0.0004859295458312511, | |
| "loss": 58.6766162109375, | |
| "step": 5500 | |
| }, | |
| { | |
| "epoch": 0.112, | |
| "grad_norm": 89412.578125, | |
| "learning_rate": 0.0004854037589479878, | |
| "loss": 58.688193359375, | |
| "step": 5600 | |
| }, | |
| { | |
| "epoch": 0.114, | |
| "grad_norm": 293593.4375, | |
| "learning_rate": 0.0004848686227156309, | |
| "loss": 59.063056640625, | |
| "step": 5700 | |
| }, | |
| { | |
| "epoch": 0.116, | |
| "grad_norm": 30479.330078125, | |
| "learning_rate": 0.0004843241583877724, | |
| "loss": 58.46580078125, | |
| "step": 5800 | |
| }, | |
| { | |
| "epoch": 0.118, | |
| "grad_norm": 89906.7734375, | |
| "learning_rate": 0.000483770387588481, | |
| "loss": 58.2409228515625, | |
| "step": 5900 | |
| }, | |
| { | |
| "epoch": 0.12, | |
| "grad_norm": 630418.5, | |
| "learning_rate": 0.00048320733231144354, | |
| "loss": 58.333916015625, | |
| "step": 6000 | |
| }, | |
| { | |
| "epoch": 0.12, | |
| "eval_loss": 59.336669921875, | |
| "eval_runtime": 1.8507, | |
| "eval_samples_per_second": 10.807, | |
| "eval_steps_per_second": 0.54, | |
| "step": 6000 | |
| }, | |
| { | |
| "epoch": 0.122, | |
| "grad_norm": 462631.5, | |
| "learning_rate": 0.000482635014919091, | |
| "loss": 59.005234375, | |
| "step": 6100 | |
| }, | |
| { | |
| "epoch": 0.124, | |
| "grad_norm": 273326.59375, | |
| "learning_rate": 0.00048205345814171074, | |
| "loss": 58.9213232421875, | |
| "step": 6200 | |
| }, | |
| { | |
| "epoch": 0.126, | |
| "grad_norm": 36951.60546875, | |
| "learning_rate": 0.00048146268507654377, | |
| "loss": 58.82025390625, | |
| "step": 6300 | |
| }, | |
| { | |
| "epoch": 0.128, | |
| "grad_norm": 103255.4765625, | |
| "learning_rate": 0.00048086271918686716, | |
| "loss": 58.84619140625, | |
| "step": 6400 | |
| }, | |
| { | |
| "epoch": 0.13, | |
| "grad_norm": 102904.96875, | |
| "learning_rate": 0.00048025358430106227, | |
| "loss": 58.514482421875, | |
| "step": 6500 | |
| }, | |
| { | |
| "epoch": 0.132, | |
| "grad_norm": 368401.625, | |
| "learning_rate": 0.00047963530461166826, | |
| "loss": 58.3533984375, | |
| "step": 6600 | |
| }, | |
| { | |
| "epoch": 0.134, | |
| "grad_norm": 307524.84375, | |
| "learning_rate": 0.0004790079046744218, | |
| "loss": 58.44912109375, | |
| "step": 6700 | |
| }, | |
| { | |
| "epoch": 0.136, | |
| "grad_norm": 218012.390625, | |
| "learning_rate": 0.000478371409407281, | |
| "loss": 58.45431640625, | |
| "step": 6800 | |
| }, | |
| { | |
| "epoch": 0.138, | |
| "grad_norm": 1243047.5, | |
| "learning_rate": 0.0004777258440894362, | |
| "loss": 58.24828125, | |
| "step": 6900 | |
| }, | |
| { | |
| "epoch": 0.14, | |
| "grad_norm": 158677.6875, | |
| "learning_rate": 0.0004770712343603062, | |
| "loss": 58.133662109375, | |
| "step": 7000 | |
| }, | |
| { | |
| "epoch": 0.14, | |
| "eval_loss": 58.2567138671875, | |
| "eval_runtime": 1.8382, | |
| "eval_samples_per_second": 10.88, | |
| "eval_steps_per_second": 0.544, | |
| "step": 7000 | |
| }, | |
| { | |
| "epoch": 0.142, | |
| "grad_norm": 237161.1875, | |
| "learning_rate": 0.0004764076062185194, | |
| "loss": 58.3987744140625, | |
| "step": 7100 | |
| }, | |
| { | |
| "epoch": 0.144, | |
| "grad_norm": 60140.4921875, | |
| "learning_rate": 0.00047573498602088154, | |
| "loss": 58.801220703125, | |
| "step": 7200 | |
| }, | |
| { | |
| "epoch": 0.146, | |
| "grad_norm": 494529.25, | |
| "learning_rate": 0.00047505340048132916, | |
| "loss": 58.503544921875, | |
| "step": 7300 | |
| }, | |
| { | |
| "epoch": 0.148, | |
| "grad_norm": 561465.625, | |
| "learning_rate": 0.00047436287666986803, | |
| "loss": 58.614169921875, | |
| "step": 7400 | |
| }, | |
| { | |
| "epoch": 0.15, | |
| "grad_norm": 572518.4375, | |
| "learning_rate": 0.00047366344201149856, | |
| "loss": 58.28416015625, | |
| "step": 7500 | |
| }, | |
| { | |
| "epoch": 0.152, | |
| "grad_norm": 1078630.125, | |
| "learning_rate": 0.0004729551242851264, | |
| "loss": 58.4676513671875, | |
| "step": 7600 | |
| }, | |
| { | |
| "epoch": 0.154, | |
| "grad_norm": 70299.53125, | |
| "learning_rate": 0.00047223795162245886, | |
| "loss": 58.200947265625, | |
| "step": 7700 | |
| }, | |
| { | |
| "epoch": 0.156, | |
| "grad_norm": 68182.265625, | |
| "learning_rate": 0.0004715119525068883, | |
| "loss": 58.55353515625, | |
| "step": 7800 | |
| }, | |
| { | |
| "epoch": 0.158, | |
| "grad_norm": 82698.640625, | |
| "learning_rate": 0.00047077715577236015, | |
| "loss": 58.704990234375, | |
| "step": 7900 | |
| }, | |
| { | |
| "epoch": 0.16, | |
| "grad_norm": 375052.03125, | |
| "learning_rate": 0.0004700335906022283, | |
| "loss": 58.521640625, | |
| "step": 8000 | |
| }, | |
| { | |
| "epoch": 0.16, | |
| "eval_loss": 57.7498779296875, | |
| "eval_runtime": 1.8363, | |
| "eval_samples_per_second": 10.892, | |
| "eval_steps_per_second": 0.545, | |
| "step": 8000 | |
| } | |
| ], | |
| "logging_steps": 100, | |
| "max_steps": 50000, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 9223372036854775807, | |
| "save_steps": 2000, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 5.699606151168e+16, | |
| "train_batch_size": 4, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |