diff --git "a/checkpoint-3960/trainer_state.json" "b/checkpoint-3960/trainer_state.json" new file mode 100644--- /dev/null +++ "b/checkpoint-3960/trainer_state.json" @@ -0,0 +1,6178 @@ +{ + "best_global_step": 3960, + "best_metric": 0.010845237411558628, + "best_model_checkpoint": "./experiments/unet_run_2026-06-25_17-20/checkpoint-3960", + "epoch": 60.0, + "eval_steps": 500, + "global_step": 3960, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.07575757575757576, + "grad_norm": 2.9941234588623047, + "learning_rate": 9.98989898989899e-05, + "loss": 1.0469184875488282, + "step": 5 + }, + { + "epoch": 0.15151515151515152, + "grad_norm": 2.392153263092041, + "learning_rate": 9.977272727272728e-05, + "loss": 0.8356879234313965, + "step": 10 + }, + { + "epoch": 0.22727272727272727, + "grad_norm": 2.106093406677246, + "learning_rate": 9.964646464646466e-05, + "loss": 0.6939716815948487, + "step": 15 + }, + { + "epoch": 0.30303030303030304, + "grad_norm": 1.7363940477371216, + "learning_rate": 9.952020202020202e-05, + "loss": 0.6017529487609863, + "step": 20 + }, + { + "epoch": 0.3787878787878788, + "grad_norm": 1.7888106107711792, + "learning_rate": 9.939393939393939e-05, + "loss": 0.5593678474426269, + "step": 25 + }, + { + "epoch": 0.45454545454545453, + "grad_norm": 1.6592128276824951, + "learning_rate": 9.926767676767678e-05, + "loss": 0.5081584930419922, + "step": 30 + }, + { + "epoch": 0.5303030303030303, + "grad_norm": 1.7191356420516968, + "learning_rate": 9.914141414141415e-05, + "loss": 0.4990970611572266, + "step": 35 + }, + { + "epoch": 0.6060606060606061, + "grad_norm": 1.633081316947937, + "learning_rate": 9.901515151515151e-05, + "loss": 0.4610342025756836, + "step": 40 + }, + { + "epoch": 0.6818181818181818, + "grad_norm": 1.472335934638977, + "learning_rate": 9.888888888888889e-05, + "loss": 0.44696502685546874, + "step": 45 + }, + { + "epoch": 0.7575757575757576, + "grad_norm": 1.6277743577957153, + "learning_rate": 9.876262626262627e-05, + "loss": 0.42172861099243164, + "step": 50 + }, + { + "epoch": 0.8333333333333334, + "grad_norm": 1.5494871139526367, + "learning_rate": 9.863636363636364e-05, + "loss": 0.3946859121322632, + "step": 55 + }, + { + "epoch": 0.9090909090909091, + "grad_norm": 1.525748610496521, + "learning_rate": 9.851010101010102e-05, + "loss": 0.3655247211456299, + "step": 60 + }, + { + "epoch": 0.9848484848484849, + "grad_norm": 1.3856412172317505, + "learning_rate": 9.838383838383838e-05, + "loss": 0.3523891448974609, + "step": 65 + }, + { + "epoch": 1.0, + "eval_loss": 0.33084386587142944, + "eval_mean_accuracy": 0.9162684080095905, + "eval_mean_iou": 0.8537441068515753, + "eval_runtime": 129.8738, + "eval_samples_per_second": 1.809, + "eval_steps_per_second": 0.231, + "step": 66 + }, + { + "epoch": 1.0606060606060606, + "grad_norm": 1.4051579236984253, + "learning_rate": 9.825757575757576e-05, + "loss": 0.3386881113052368, + "step": 70 + }, + { + "epoch": 1.1363636363636362, + "grad_norm": 1.385435700416565, + "learning_rate": 9.813131313131314e-05, + "loss": 0.3289308786392212, + "step": 75 + }, + { + "epoch": 1.2121212121212122, + "grad_norm": 1.254081130027771, + "learning_rate": 9.800505050505051e-05, + "loss": 0.30045642852783205, + "step": 80 + }, + { + "epoch": 1.2878787878787878, + "grad_norm": 1.2731844186782837, + "learning_rate": 9.787878787878789e-05, + "loss": 0.29194915294647217, + "step": 85 + }, + { + "epoch": 1.3636363636363638, + "grad_norm": 1.1707983016967773, + "learning_rate": 9.775252525252527e-05, + "loss": 0.27875852584838867, + "step": 90 + }, + { + "epoch": 1.4393939393939394, + "grad_norm": 1.4898347854614258, + "learning_rate": 9.762626262626263e-05, + "loss": 0.2846740961074829, + "step": 95 + }, + { + "epoch": 1.5151515151515151, + "grad_norm": 1.1377719640731812, + "learning_rate": 9.75e-05, + "loss": 0.2732773542404175, + "step": 100 + }, + { + "epoch": 1.5909090909090908, + "grad_norm": 1.131147861480713, + "learning_rate": 9.737373737373738e-05, + "loss": 0.26055796146392823, + "step": 105 + }, + { + "epoch": 1.6666666666666665, + "grad_norm": 1.123711347579956, + "learning_rate": 9.724747474747476e-05, + "loss": 0.24981226921081542, + "step": 110 + }, + { + "epoch": 1.7424242424242424, + "grad_norm": 1.0596073865890503, + "learning_rate": 9.712121212121212e-05, + "loss": 0.23202009201049806, + "step": 115 + }, + { + "epoch": 1.8181818181818183, + "grad_norm": 0.9440689086914062, + "learning_rate": 9.699494949494949e-05, + "loss": 0.22050716876983642, + "step": 120 + }, + { + "epoch": 1.893939393939394, + "grad_norm": 1.037160873413086, + "learning_rate": 9.686868686868688e-05, + "loss": 0.22513251304626464, + "step": 125 + }, + { + "epoch": 1.9696969696969697, + "grad_norm": 1.6623395681381226, + "learning_rate": 9.674242424242425e-05, + "loss": 0.22343740463256836, + "step": 130 + }, + { + "epoch": 2.0, + "eval_loss": 0.2221149206161499, + "eval_mean_accuracy": 0.9153114944563248, + "eval_mean_iou": 0.8548370081877293, + "eval_runtime": 128.12, + "eval_samples_per_second": 1.834, + "eval_steps_per_second": 0.234, + "step": 132 + }, + { + "epoch": 2.0454545454545454, + "grad_norm": 0.9342519044876099, + "learning_rate": 9.661616161616161e-05, + "loss": 0.20234076976776122, + "step": 135 + }, + { + "epoch": 2.121212121212121, + "grad_norm": 0.9801168441772461, + "learning_rate": 9.648989898989899e-05, + "loss": 0.20889129638671874, + "step": 140 + }, + { + "epoch": 2.196969696969697, + "grad_norm": 0.7967773079872131, + "learning_rate": 9.636363636363637e-05, + "loss": 0.19033305644989013, + "step": 145 + }, + { + "epoch": 2.2727272727272725, + "grad_norm": 0.8011311292648315, + "learning_rate": 9.623737373737374e-05, + "loss": 0.18722602128982543, + "step": 150 + }, + { + "epoch": 2.3484848484848486, + "grad_norm": 0.8544222712516785, + "learning_rate": 9.611111111111112e-05, + "loss": 0.17198851108551025, + "step": 155 + }, + { + "epoch": 2.4242424242424243, + "grad_norm": 0.876949667930603, + "learning_rate": 9.598484848484848e-05, + "loss": 0.1730184316635132, + "step": 160 + }, + { + "epoch": 2.5, + "grad_norm": 0.7883437871932983, + "learning_rate": 9.585858585858586e-05, + "loss": 0.15062739849090576, + "step": 165 + }, + { + "epoch": 2.5757575757575757, + "grad_norm": 0.9258979558944702, + "learning_rate": 9.573232323232324e-05, + "loss": 0.14979397058486937, + "step": 170 + }, + { + "epoch": 2.6515151515151514, + "grad_norm": 0.6856182813644409, + "learning_rate": 9.560606060606061e-05, + "loss": 0.13935924768447877, + "step": 175 + }, + { + "epoch": 2.7272727272727275, + "grad_norm": 0.8212957382202148, + "learning_rate": 9.547979797979797e-05, + "loss": 0.14868369102478027, + "step": 180 + }, + { + "epoch": 2.8030303030303028, + "grad_norm": 0.7166414260864258, + "learning_rate": 9.535353535353537e-05, + "loss": 0.13483697175979614, + "step": 185 + }, + { + "epoch": 2.878787878787879, + "grad_norm": 0.6200843453407288, + "learning_rate": 9.522727272727273e-05, + "loss": 0.13089344501495362, + "step": 190 + }, + { + "epoch": 2.9545454545454546, + "grad_norm": 1.2299199104309082, + "learning_rate": 9.51010101010101e-05, + "loss": 0.14360649585723878, + "step": 195 + }, + { + "epoch": 3.0, + "eval_loss": 0.12719328701496124, + "eval_mean_accuracy": 0.8923612222462977, + "eval_mean_iou": 0.8683020680152902, + "eval_runtime": 126.1594, + "eval_samples_per_second": 1.863, + "eval_steps_per_second": 0.238, + "step": 198 + }, + { + "epoch": 3.0303030303030303, + "grad_norm": 0.6092143654823303, + "learning_rate": 9.497474747474748e-05, + "loss": 0.1220658302307129, + "step": 200 + }, + { + "epoch": 3.106060606060606, + "grad_norm": 0.5944573879241943, + "learning_rate": 9.484848484848486e-05, + "loss": 0.1220181941986084, + "step": 205 + }, + { + "epoch": 3.1818181818181817, + "grad_norm": 0.5286777019500732, + "learning_rate": 9.472222222222222e-05, + "loss": 0.10843038558959961, + "step": 210 + }, + { + "epoch": 3.257575757575758, + "grad_norm": 0.5887173414230347, + "learning_rate": 9.45959595959596e-05, + "loss": 0.12465133666992187, + "step": 215 + }, + { + "epoch": 3.3333333333333335, + "grad_norm": 0.5597184896469116, + "learning_rate": 9.446969696969697e-05, + "loss": 0.10208001136779785, + "step": 220 + }, + { + "epoch": 3.409090909090909, + "grad_norm": 0.7133381962776184, + "learning_rate": 9.434343434343435e-05, + "loss": 0.10837376117706299, + "step": 225 + }, + { + "epoch": 3.484848484848485, + "grad_norm": 0.9375573396682739, + "learning_rate": 9.421717171717173e-05, + "loss": 0.0992659330368042, + "step": 230 + }, + { + "epoch": 3.5606060606060606, + "grad_norm": 0.5562649965286255, + "learning_rate": 9.40909090909091e-05, + "loss": 0.10355113744735718, + "step": 235 + }, + { + "epoch": 3.6363636363636362, + "grad_norm": 1.3427445888519287, + "learning_rate": 9.396464646464646e-05, + "loss": 0.09606725573539734, + "step": 240 + }, + { + "epoch": 3.712121212121212, + "grad_norm": 0.41196179389953613, + "learning_rate": 9.383838383838385e-05, + "loss": 0.09453063011169434, + "step": 245 + }, + { + "epoch": 3.787878787878788, + "grad_norm": 0.4183890223503113, + "learning_rate": 9.371212121212122e-05, + "loss": 0.09348648190498351, + "step": 250 + }, + { + "epoch": 3.8636363636363638, + "grad_norm": 0.5283163189888, + "learning_rate": 9.358585858585858e-05, + "loss": 0.09835280179977417, + "step": 255 + }, + { + "epoch": 3.9393939393939394, + "grad_norm": 0.9340290427207947, + "learning_rate": 9.345959595959596e-05, + "loss": 0.09048854112625122, + "step": 260 + }, + { + "epoch": 4.0, + "eval_loss": 0.08496074378490448, + "eval_mean_accuracy": 0.8858880236689162, + "eval_mean_iou": 0.8636583053187777, + "eval_runtime": 119.2785, + "eval_samples_per_second": 1.97, + "eval_steps_per_second": 0.252, + "step": 264 + }, + { + "epoch": 4.015151515151516, + "grad_norm": 0.4928303360939026, + "learning_rate": 9.333333333333334e-05, + "loss": 0.08872635960578919, + "step": 265 + }, + { + "epoch": 4.090909090909091, + "grad_norm": 0.5040572285652161, + "learning_rate": 9.320707070707071e-05, + "loss": 0.09237761497497558, + "step": 270 + }, + { + "epoch": 4.166666666666667, + "grad_norm": 0.42513513565063477, + "learning_rate": 9.308080808080809e-05, + "loss": 0.08510953783988953, + "step": 275 + }, + { + "epoch": 4.242424242424242, + "grad_norm": 0.44105979800224304, + "learning_rate": 9.295454545454545e-05, + "loss": 0.08392485976219177, + "step": 280 + }, + { + "epoch": 4.318181818181818, + "grad_norm": 0.44409239292144775, + "learning_rate": 9.282828282828283e-05, + "loss": 0.08064403533935546, + "step": 285 + }, + { + "epoch": 4.393939393939394, + "grad_norm": 0.6499812602996826, + "learning_rate": 9.27020202020202e-05, + "loss": 0.0818498432636261, + "step": 290 + }, + { + "epoch": 4.46969696969697, + "grad_norm": 0.6977149248123169, + "learning_rate": 9.257575757575758e-05, + "loss": 0.08164259791374207, + "step": 295 + }, + { + "epoch": 4.545454545454545, + "grad_norm": 0.37146201729774475, + "learning_rate": 9.244949494949496e-05, + "loss": 0.07308000326156616, + "step": 300 + }, + { + "epoch": 4.621212121212121, + "grad_norm": 0.34798547625541687, + "learning_rate": 9.232323232323232e-05, + "loss": 0.0734625518321991, + "step": 305 + }, + { + "epoch": 4.696969696969697, + "grad_norm": 0.4575636684894562, + "learning_rate": 9.21969696969697e-05, + "loss": 0.07146384716033935, + "step": 310 + }, + { + "epoch": 4.7727272727272725, + "grad_norm": 0.3662533760070801, + "learning_rate": 9.207070707070707e-05, + "loss": 0.06769375801086426, + "step": 315 + }, + { + "epoch": 4.848484848484849, + "grad_norm": 0.3551352024078369, + "learning_rate": 9.194444444444445e-05, + "loss": 0.06268389225006103, + "step": 320 + }, + { + "epoch": 4.924242424242424, + "grad_norm": 0.5918303728103638, + "learning_rate": 9.181818181818183e-05, + "loss": 0.06479804515838623, + "step": 325 + }, + { + "epoch": 5.0, + "grad_norm": 0.6824851036071777, + "learning_rate": 9.16919191919192e-05, + "loss": 0.07680357098579407, + "step": 330 + }, + { + "epoch": 5.0, + "eval_loss": 0.0638287365436554, + "eval_mean_accuracy": 0.9559736222980618, + "eval_mean_iou": 0.9068221008671852, + "eval_runtime": 128.6336, + "eval_samples_per_second": 1.827, + "eval_steps_per_second": 0.233, + "step": 330 + }, + { + "epoch": 5.075757575757576, + "grad_norm": 0.36692366003990173, + "learning_rate": 9.156565656565656e-05, + "loss": 0.06825620532035828, + "step": 335 + }, + { + "epoch": 5.151515151515151, + "grad_norm": 0.33095893263816833, + "learning_rate": 9.143939393939395e-05, + "loss": 0.06594018936157227, + "step": 340 + }, + { + "epoch": 5.2272727272727275, + "grad_norm": 0.4086349904537201, + "learning_rate": 9.131313131313132e-05, + "loss": 0.0743526577949524, + "step": 345 + }, + { + "epoch": 5.303030303030303, + "grad_norm": 0.28984394669532776, + "learning_rate": 9.118686868686869e-05, + "loss": 0.05957569479942322, + "step": 350 + }, + { + "epoch": 5.378787878787879, + "grad_norm": 0.27274346351623535, + "learning_rate": 9.106060606060606e-05, + "loss": 0.06417046785354615, + "step": 355 + }, + { + "epoch": 5.454545454545454, + "grad_norm": 1.8646568059921265, + "learning_rate": 9.093434343434344e-05, + "loss": 0.07052257061004638, + "step": 360 + }, + { + "epoch": 5.53030303030303, + "grad_norm": 0.2765451669692993, + "learning_rate": 9.080808080808081e-05, + "loss": 0.06039916276931763, + "step": 365 + }, + { + "epoch": 5.606060606060606, + "grad_norm": 0.260680228471756, + "learning_rate": 9.068181818181819e-05, + "loss": 0.05675415992736817, + "step": 370 + }, + { + "epoch": 5.681818181818182, + "grad_norm": 0.28278031945228577, + "learning_rate": 9.055555555555556e-05, + "loss": 0.05483880639076233, + "step": 375 + }, + { + "epoch": 5.757575757575758, + "grad_norm": 0.28105831146240234, + "learning_rate": 9.042929292929293e-05, + "loss": 0.054865854978561404, + "step": 380 + }, + { + "epoch": 5.833333333333333, + "grad_norm": 0.5942025184631348, + "learning_rate": 9.030303030303031e-05, + "loss": 0.0555119514465332, + "step": 385 + }, + { + "epoch": 5.909090909090909, + "grad_norm": 0.3304992914199829, + "learning_rate": 9.017676767676768e-05, + "loss": 0.051424407958984376, + "step": 390 + }, + { + "epoch": 5.984848484848484, + "grad_norm": 0.21033477783203125, + "learning_rate": 9.005050505050505e-05, + "loss": 0.050022590160369876, + "step": 395 + }, + { + "epoch": 6.0, + "eval_loss": 0.05429566651582718, + "eval_mean_accuracy": 0.9362085034786801, + "eval_mean_iou": 0.9067576802996102, + "eval_runtime": 127.8161, + "eval_samples_per_second": 1.839, + "eval_steps_per_second": 0.235, + "step": 396 + }, + { + "epoch": 6.0606060606060606, + "grad_norm": 0.2550901770591736, + "learning_rate": 8.992424242424244e-05, + "loss": 0.05656055212020874, + "step": 400 + }, + { + "epoch": 6.136363636363637, + "grad_norm": 0.5624765753746033, + "learning_rate": 8.97979797979798e-05, + "loss": 0.05291185975074768, + "step": 405 + }, + { + "epoch": 6.212121212121212, + "grad_norm": 0.21705150604248047, + "learning_rate": 8.967171717171717e-05, + "loss": 0.0508266806602478, + "step": 410 + }, + { + "epoch": 6.287878787878788, + "grad_norm": 0.2520148754119873, + "learning_rate": 8.954545454545455e-05, + "loss": 0.050147545337677, + "step": 415 + }, + { + "epoch": 6.363636363636363, + "grad_norm": 0.2019597291946411, + "learning_rate": 8.941919191919193e-05, + "loss": 0.05105447769165039, + "step": 420 + }, + { + "epoch": 6.4393939393939394, + "grad_norm": 0.281158447265625, + "learning_rate": 8.92929292929293e-05, + "loss": 0.051430970430374146, + "step": 425 + }, + { + "epoch": 6.515151515151516, + "grad_norm": 0.24807868897914886, + "learning_rate": 8.916666666666667e-05, + "loss": 0.04538700580596924, + "step": 430 + }, + { + "epoch": 6.590909090909091, + "grad_norm": 0.2733008861541748, + "learning_rate": 8.904040404040404e-05, + "loss": 0.04853827357292175, + "step": 435 + }, + { + "epoch": 6.666666666666667, + "grad_norm": 0.302228718996048, + "learning_rate": 8.891414141414142e-05, + "loss": 0.04623064398765564, + "step": 440 + }, + { + "epoch": 6.742424242424242, + "grad_norm": 0.2073170691728592, + "learning_rate": 8.87878787878788e-05, + "loss": 0.04433267712593079, + "step": 445 + }, + { + "epoch": 6.818181818181818, + "grad_norm": 0.2997240424156189, + "learning_rate": 8.866161616161617e-05, + "loss": 0.04357313811779022, + "step": 450 + }, + { + "epoch": 6.893939393939394, + "grad_norm": 0.2226785272359848, + "learning_rate": 8.853535353535354e-05, + "loss": 0.0439825028181076, + "step": 455 + }, + { + "epoch": 6.96969696969697, + "grad_norm": 0.21638792753219604, + "learning_rate": 8.840909090909091e-05, + "loss": 0.04534157514572144, + "step": 460 + }, + { + "epoch": 7.0, + "eval_loss": 0.04328976944088936, + "eval_mean_accuracy": 0.9334131856031601, + "eval_mean_iou": 0.9108860224160366, + "eval_runtime": 124.0346, + "eval_samples_per_second": 1.895, + "eval_steps_per_second": 0.242, + "step": 462 + }, + { + "epoch": 7.045454545454546, + "grad_norm": 0.2381969839334488, + "learning_rate": 8.828282828282829e-05, + "loss": 0.04455110132694244, + "step": 465 + }, + { + "epoch": 7.121212121212121, + "grad_norm": 0.2459510862827301, + "learning_rate": 8.815656565656566e-05, + "loss": 0.04009305238723755, + "step": 470 + }, + { + "epoch": 7.196969696969697, + "grad_norm": 0.20454367995262146, + "learning_rate": 8.803030303030304e-05, + "loss": 0.0390073299407959, + "step": 475 + }, + { + "epoch": 7.2727272727272725, + "grad_norm": 0.3225804269313812, + "learning_rate": 8.790404040404041e-05, + "loss": 0.04440179169178009, + "step": 480 + }, + { + "epoch": 7.348484848484849, + "grad_norm": 0.2174624353647232, + "learning_rate": 8.777777777777778e-05, + "loss": 0.036155888438224794, + "step": 485 + }, + { + "epoch": 7.424242424242424, + "grad_norm": 0.18928103148937225, + "learning_rate": 8.765151515151515e-05, + "loss": 0.037705269455909726, + "step": 490 + }, + { + "epoch": 7.5, + "grad_norm": 0.23600833117961884, + "learning_rate": 8.752525252525253e-05, + "loss": 0.04156590700149536, + "step": 495 + }, + { + "epoch": 7.575757575757576, + "grad_norm": 0.2093035876750946, + "learning_rate": 8.73989898989899e-05, + "loss": 0.036895540356636045, + "step": 500 + }, + { + "epoch": 7.651515151515151, + "grad_norm": 0.17765530943870544, + "learning_rate": 8.727272727272727e-05, + "loss": 0.039797413349151614, + "step": 505 + }, + { + "epoch": 7.7272727272727275, + "grad_norm": 0.2168579250574112, + "learning_rate": 8.714646464646465e-05, + "loss": 0.04091753661632538, + "step": 510 + }, + { + "epoch": 7.803030303030303, + "grad_norm": 0.19193395972251892, + "learning_rate": 8.702020202020203e-05, + "loss": 0.03886389434337616, + "step": 515 + }, + { + "epoch": 7.878787878787879, + "grad_norm": 0.2441186010837555, + "learning_rate": 8.68939393939394e-05, + "loss": 0.04203253388404846, + "step": 520 + }, + { + "epoch": 7.954545454545455, + "grad_norm": 0.4751134216785431, + "learning_rate": 8.676767676767678e-05, + "loss": 0.0395833432674408, + "step": 525 + }, + { + "epoch": 8.0, + "eval_loss": 0.0367831215262413, + "eval_mean_accuracy": 0.9590330911510937, + "eval_mean_iou": 0.9082146960606328, + "eval_runtime": 130.2898, + "eval_samples_per_second": 1.804, + "eval_steps_per_second": 0.23, + "step": 528 + }, + { + "epoch": 8.030303030303031, + "grad_norm": 0.19869540631771088, + "learning_rate": 8.664141414141414e-05, + "loss": 0.03650480210781097, + "step": 530 + }, + { + "epoch": 8.106060606060606, + "grad_norm": 0.18103687465190887, + "learning_rate": 8.651515151515152e-05, + "loss": 0.03831833899021149, + "step": 535 + }, + { + "epoch": 8.181818181818182, + "grad_norm": 0.16558200120925903, + "learning_rate": 8.63888888888889e-05, + "loss": 0.03718695342540741, + "step": 540 + }, + { + "epoch": 8.257575757575758, + "grad_norm": 0.18343114852905273, + "learning_rate": 8.626262626262627e-05, + "loss": 0.033085483312606814, + "step": 545 + }, + { + "epoch": 8.333333333333334, + "grad_norm": 1.3617371320724487, + "learning_rate": 8.613636363636363e-05, + "loss": 0.0378117561340332, + "step": 550 + }, + { + "epoch": 8.409090909090908, + "grad_norm": 0.16200992465019226, + "learning_rate": 8.601010101010102e-05, + "loss": 0.03796728253364563, + "step": 555 + }, + { + "epoch": 8.484848484848484, + "grad_norm": 0.1847938746213913, + "learning_rate": 8.588383838383839e-05, + "loss": 0.036657083034515384, + "step": 560 + }, + { + "epoch": 8.56060606060606, + "grad_norm": 0.18106864392757416, + "learning_rate": 8.575757575757576e-05, + "loss": 0.037758547067642215, + "step": 565 + }, + { + "epoch": 8.636363636363637, + "grad_norm": 0.17433470487594604, + "learning_rate": 8.563131313131314e-05, + "loss": 0.033577916026115415, + "step": 570 + }, + { + "epoch": 8.712121212121213, + "grad_norm": 0.20621395111083984, + "learning_rate": 8.550505050505052e-05, + "loss": 0.033245939016342166, + "step": 575 + }, + { + "epoch": 8.787878787878787, + "grad_norm": 0.21127405762672424, + "learning_rate": 8.537878787878788e-05, + "loss": 0.038744556903839114, + "step": 580 + }, + { + "epoch": 8.863636363636363, + "grad_norm": 0.1803150773048401, + "learning_rate": 8.525252525252526e-05, + "loss": 0.03508321642875671, + "step": 585 + }, + { + "epoch": 8.93939393939394, + "grad_norm": 0.21203270554542542, + "learning_rate": 8.512626262626263e-05, + "loss": 0.03586195707321167, + "step": 590 + }, + { + "epoch": 9.0, + "eval_loss": 0.03499237075448036, + "eval_mean_accuracy": 0.9632630351861261, + "eval_mean_iou": 0.9147436291222774, + "eval_runtime": 135.0741, + "eval_samples_per_second": 1.74, + "eval_steps_per_second": 0.222, + "step": 594 + }, + { + "epoch": 9.015151515151516, + "grad_norm": 0.14731831848621368, + "learning_rate": 8.5e-05, + "loss": 0.03633248209953308, + "step": 595 + }, + { + "epoch": 9.090909090909092, + "grad_norm": 0.1524285227060318, + "learning_rate": 8.487373737373739e-05, + "loss": 0.03264537453651428, + "step": 600 + }, + { + "epoch": 9.166666666666666, + "grad_norm": 0.1411120742559433, + "learning_rate": 8.474747474747475e-05, + "loss": 0.0289734810590744, + "step": 605 + }, + { + "epoch": 9.242424242424242, + "grad_norm": 0.153798907995224, + "learning_rate": 8.462121212121212e-05, + "loss": 0.031535619497299196, + "step": 610 + }, + { + "epoch": 9.318181818181818, + "grad_norm": 0.6651473045349121, + "learning_rate": 8.44949494949495e-05, + "loss": 0.035373720526695254, + "step": 615 + }, + { + "epoch": 9.393939393939394, + "grad_norm": 0.15184816718101501, + "learning_rate": 8.436868686868688e-05, + "loss": 0.028237152099609374, + "step": 620 + }, + { + "epoch": 9.469696969696969, + "grad_norm": 0.17039573192596436, + "learning_rate": 8.424242424242424e-05, + "loss": 0.02888110876083374, + "step": 625 + }, + { + "epoch": 9.545454545454545, + "grad_norm": 0.16624340415000916, + "learning_rate": 8.411616161616162e-05, + "loss": 0.030333247780799866, + "step": 630 + }, + { + "epoch": 9.621212121212121, + "grad_norm": 0.21998383104801178, + "learning_rate": 8.3989898989899e-05, + "loss": 0.03391568660736084, + "step": 635 + }, + { + "epoch": 9.696969696969697, + "grad_norm": 0.1507062315940857, + "learning_rate": 8.386363636363637e-05, + "loss": 0.03953556418418884, + "step": 640 + }, + { + "epoch": 9.772727272727273, + "grad_norm": 0.5309189558029175, + "learning_rate": 8.373737373737373e-05, + "loss": 0.03747777044773102, + "step": 645 + }, + { + "epoch": 9.848484848484848, + "grad_norm": 0.158358633518219, + "learning_rate": 8.361111111111111e-05, + "loss": 0.032436329126358035, + "step": 650 + }, + { + "epoch": 9.924242424242424, + "grad_norm": 0.16624100506305695, + "learning_rate": 8.348484848484849e-05, + "loss": 0.03323723673820496, + "step": 655 + }, + { + "epoch": 10.0, + "grad_norm": 0.5144572854042053, + "learning_rate": 8.335858585858586e-05, + "loss": 0.03635854721069336, + "step": 660 + }, + { + "epoch": 10.0, + "eval_loss": 0.030158301815390587, + "eval_mean_accuracy": 0.9494412250720125, + "eval_mean_iou": 0.9132272169011909, + "eval_runtime": 124.0226, + "eval_samples_per_second": 1.895, + "eval_steps_per_second": 0.242, + "step": 660 + }, + { + "epoch": 10.075757575757576, + "grad_norm": 0.3217375874519348, + "learning_rate": 8.323232323232324e-05, + "loss": 0.03381457328796387, + "step": 665 + }, + { + "epoch": 10.151515151515152, + "grad_norm": 0.13216620683670044, + "learning_rate": 8.310606060606062e-05, + "loss": 0.026793745160102845, + "step": 670 + }, + { + "epoch": 10.227272727272727, + "grad_norm": 0.13275830447673798, + "learning_rate": 8.297979797979798e-05, + "loss": 0.030741795897483826, + "step": 675 + }, + { + "epoch": 10.303030303030303, + "grad_norm": 0.15839700400829315, + "learning_rate": 8.285353535353536e-05, + "loss": 0.03097023367881775, + "step": 680 + }, + { + "epoch": 10.378787878787879, + "grad_norm": 0.10817021876573563, + "learning_rate": 8.272727272727273e-05, + "loss": 0.02466929405927658, + "step": 685 + }, + { + "epoch": 10.454545454545455, + "grad_norm": 0.14087523519992828, + "learning_rate": 8.260101010101011e-05, + "loss": 0.02694886028766632, + "step": 690 + }, + { + "epoch": 10.530303030303031, + "grad_norm": 0.13011270761489868, + "learning_rate": 8.247474747474749e-05, + "loss": 0.028478020429611207, + "step": 695 + }, + { + "epoch": 10.606060606060606, + "grad_norm": 0.14002978801727295, + "learning_rate": 8.234848484848485e-05, + "loss": 0.027278423309326172, + "step": 700 + }, + { + "epoch": 10.681818181818182, + "grad_norm": 0.19965209066867828, + "learning_rate": 8.222222222222222e-05, + "loss": 0.027237522602081298, + "step": 705 + }, + { + "epoch": 10.757575757575758, + "grad_norm": 0.36491817235946655, + "learning_rate": 8.20959595959596e-05, + "loss": 0.03092048466205597, + "step": 710 + }, + { + "epoch": 10.833333333333334, + "grad_norm": 0.17862817645072937, + "learning_rate": 8.196969696969698e-05, + "loss": 0.03545044064521789, + "step": 715 + }, + { + "epoch": 10.909090909090908, + "grad_norm": 0.2442970722913742, + "learning_rate": 8.184343434343434e-05, + "loss": 0.03162646889686584, + "step": 720 + }, + { + "epoch": 10.984848484848484, + "grad_norm": 0.11630677431821823, + "learning_rate": 8.171717171717172e-05, + "loss": 0.027395763993263246, + "step": 725 + }, + { + "epoch": 11.0, + "eval_loss": 0.02825918421149254, + "eval_mean_accuracy": 0.939388232740388, + "eval_mean_iou": 0.9124427513703054, + "eval_runtime": 119.1985, + "eval_samples_per_second": 1.972, + "eval_steps_per_second": 0.252, + "step": 726 + }, + { + "epoch": 11.06060606060606, + "grad_norm": 0.2564373016357422, + "learning_rate": 8.15909090909091e-05, + "loss": 0.028311532735824586, + "step": 730 + }, + { + "epoch": 11.136363636363637, + "grad_norm": 0.19052653014659882, + "learning_rate": 8.146464646464647e-05, + "loss": 0.023883090913295747, + "step": 735 + }, + { + "epoch": 11.212121212121213, + "grad_norm": 0.1189924106001854, + "learning_rate": 8.133838383838385e-05, + "loss": 0.02626485228538513, + "step": 740 + }, + { + "epoch": 11.287878787878787, + "grad_norm": 0.29297763109207153, + "learning_rate": 8.121212121212121e-05, + "loss": 0.02630702257156372, + "step": 745 + }, + { + "epoch": 11.363636363636363, + "grad_norm": 0.2152256816625595, + "learning_rate": 8.108585858585859e-05, + "loss": 0.02656479775905609, + "step": 750 + }, + { + "epoch": 11.43939393939394, + "grad_norm": 0.11115691065788269, + "learning_rate": 8.095959595959597e-05, + "loss": 0.02337968796491623, + "step": 755 + }, + { + "epoch": 11.515151515151516, + "grad_norm": 0.11810733377933502, + "learning_rate": 8.083333333333334e-05, + "loss": 0.024256861209869383, + "step": 760 + }, + { + "epoch": 11.590909090909092, + "grad_norm": 0.15823394060134888, + "learning_rate": 8.07070707070707e-05, + "loss": 0.02603786289691925, + "step": 765 + }, + { + "epoch": 11.666666666666666, + "grad_norm": 0.09161833673715591, + "learning_rate": 8.05808080808081e-05, + "loss": 0.022551599144935607, + "step": 770 + }, + { + "epoch": 11.742424242424242, + "grad_norm": 0.18982934951782227, + "learning_rate": 8.045454545454546e-05, + "loss": 0.022460317611694335, + "step": 775 + }, + { + "epoch": 11.818181818181818, + "grad_norm": 0.12188952416181564, + "learning_rate": 8.032828282828283e-05, + "loss": 0.02466122806072235, + "step": 780 + }, + { + "epoch": 11.893939393939394, + "grad_norm": 0.1868026852607727, + "learning_rate": 8.02020202020202e-05, + "loss": 0.02946215569972992, + "step": 785 + }, + { + "epoch": 11.969696969696969, + "grad_norm": 0.14319327473640442, + "learning_rate": 8.007575757575759e-05, + "loss": 0.027072882652282713, + "step": 790 + }, + { + "epoch": 12.0, + "eval_loss": 0.025496874004602432, + "eval_mean_accuracy": 0.9336491385886578, + "eval_mean_iou": 0.9077434499940594, + "eval_runtime": 128.666, + "eval_samples_per_second": 1.826, + "eval_steps_per_second": 0.233, + "step": 792 + }, + { + "epoch": 12.045454545454545, + "grad_norm": 0.13147099316120148, + "learning_rate": 7.994949494949495e-05, + "loss": 0.0284650981426239, + "step": 795 + }, + { + "epoch": 12.121212121212121, + "grad_norm": 0.18236331641674042, + "learning_rate": 7.982323232323232e-05, + "loss": 0.027952969074249268, + "step": 800 + }, + { + "epoch": 12.196969696969697, + "grad_norm": 0.11553213000297546, + "learning_rate": 7.96969696969697e-05, + "loss": 0.025592750310897826, + "step": 805 + }, + { + "epoch": 12.272727272727273, + "grad_norm": 0.10449112206697464, + "learning_rate": 7.957070707070708e-05, + "loss": 0.022632168233394624, + "step": 810 + }, + { + "epoch": 12.348484848484848, + "grad_norm": 0.11192607879638672, + "learning_rate": 7.944444444444444e-05, + "loss": 0.022631964087486266, + "step": 815 + }, + { + "epoch": 12.424242424242424, + "grad_norm": 0.09673135727643967, + "learning_rate": 7.931818181818182e-05, + "loss": 0.023840883374214174, + "step": 820 + }, + { + "epoch": 12.5, + "grad_norm": 0.1338086873292923, + "learning_rate": 7.919191919191919e-05, + "loss": 0.02500331699848175, + "step": 825 + }, + { + "epoch": 12.575757575757576, + "grad_norm": 0.09962525963783264, + "learning_rate": 7.906565656565657e-05, + "loss": 0.024028828740119933, + "step": 830 + }, + { + "epoch": 12.651515151515152, + "grad_norm": 0.09024921804666519, + "learning_rate": 7.893939393939395e-05, + "loss": 0.0224539652466774, + "step": 835 + }, + { + "epoch": 12.727272727272727, + "grad_norm": 0.12291288375854492, + "learning_rate": 7.881313131313131e-05, + "loss": 0.019437894225120544, + "step": 840 + }, + { + "epoch": 12.803030303030303, + "grad_norm": 0.1612149327993393, + "learning_rate": 7.868686868686869e-05, + "loss": 0.024380649626255035, + "step": 845 + }, + { + "epoch": 12.878787878787879, + "grad_norm": 0.08353355526924133, + "learning_rate": 7.856060606060607e-05, + "loss": 0.021596531569957732, + "step": 850 + }, + { + "epoch": 12.954545454545455, + "grad_norm": 0.10493447631597519, + "learning_rate": 7.843434343434344e-05, + "loss": 0.01932312250137329, + "step": 855 + }, + { + "epoch": 13.0, + "eval_loss": 0.021705135703086853, + "eval_mean_accuracy": 0.9536351090178168, + "eval_mean_iou": 0.9224589709789387, + "eval_runtime": 131.3753, + "eval_samples_per_second": 1.789, + "eval_steps_per_second": 0.228, + "step": 858 + }, + { + "epoch": 13.030303030303031, + "grad_norm": 0.13715972006320953, + "learning_rate": 7.83080808080808e-05, + "loss": 0.020376771688461304, + "step": 860 + }, + { + "epoch": 13.106060606060606, + "grad_norm": 0.16817788779735565, + "learning_rate": 7.818181818181818e-05, + "loss": 0.02309820204973221, + "step": 865 + }, + { + "epoch": 13.181818181818182, + "grad_norm": 0.14915378391742706, + "learning_rate": 7.805555555555556e-05, + "loss": 0.017853561043739318, + "step": 870 + }, + { + "epoch": 13.257575757575758, + "grad_norm": 0.07708711177110672, + "learning_rate": 7.792929292929293e-05, + "loss": 0.02186441719532013, + "step": 875 + }, + { + "epoch": 13.333333333333334, + "grad_norm": 0.15334434807300568, + "learning_rate": 7.780303030303031e-05, + "loss": 0.02136687934398651, + "step": 880 + }, + { + "epoch": 13.409090909090908, + "grad_norm": 0.09887427091598511, + "learning_rate": 7.767676767676769e-05, + "loss": 0.03031507134437561, + "step": 885 + }, + { + "epoch": 13.484848484848484, + "grad_norm": 0.2321063131093979, + "learning_rate": 7.755050505050505e-05, + "loss": 0.024256505072116852, + "step": 890 + }, + { + "epoch": 13.56060606060606, + "grad_norm": 0.08225462585687637, + "learning_rate": 7.742424242424243e-05, + "loss": 0.022670431435108183, + "step": 895 + }, + { + "epoch": 13.636363636363637, + "grad_norm": 0.08809585869312286, + "learning_rate": 7.72979797979798e-05, + "loss": 0.021135908365249634, + "step": 900 + }, + { + "epoch": 13.712121212121213, + "grad_norm": 0.0759420394897461, + "learning_rate": 7.717171717171718e-05, + "loss": 0.018094108998775484, + "step": 905 + }, + { + "epoch": 13.787878787878787, + "grad_norm": 0.09289054572582245, + "learning_rate": 7.704545454545456e-05, + "loss": 0.02248305082321167, + "step": 910 + }, + { + "epoch": 13.863636363636363, + "grad_norm": 0.11058207601308823, + "learning_rate": 7.691919191919192e-05, + "loss": 0.019486013054847717, + "step": 915 + }, + { + "epoch": 13.93939393939394, + "grad_norm": 0.08340949565172195, + "learning_rate": 7.679292929292929e-05, + "loss": 0.02120814621448517, + "step": 920 + }, + { + "epoch": 14.0, + "eval_loss": 0.021065689623355865, + "eval_mean_accuracy": 0.9516235592483079, + "eval_mean_iou": 0.922198336388199, + "eval_runtime": 129.9602, + "eval_samples_per_second": 1.808, + "eval_steps_per_second": 0.231, + "step": 924 + }, + { + "epoch": 14.015151515151516, + "grad_norm": 0.09346017241477966, + "learning_rate": 7.666666666666667e-05, + "loss": 0.01847185641527176, + "step": 925 + }, + { + "epoch": 14.090909090909092, + "grad_norm": 0.09822884947061539, + "learning_rate": 7.654040404040405e-05, + "loss": 0.01933208405971527, + "step": 930 + }, + { + "epoch": 14.166666666666666, + "grad_norm": 0.07557687908411026, + "learning_rate": 7.641414141414141e-05, + "loss": 0.01722216308116913, + "step": 935 + }, + { + "epoch": 14.242424242424242, + "grad_norm": 0.08013995736837387, + "learning_rate": 7.62878787878788e-05, + "loss": 0.018065088987350465, + "step": 940 + }, + { + "epoch": 14.318181818181818, + "grad_norm": 0.0684216246008873, + "learning_rate": 7.616161616161617e-05, + "loss": 0.018077316880226135, + "step": 945 + }, + { + "epoch": 14.393939393939394, + "grad_norm": 0.088801309466362, + "learning_rate": 7.603535353535354e-05, + "loss": 0.020560190081596375, + "step": 950 + }, + { + "epoch": 14.469696969696969, + "grad_norm": 0.13157963752746582, + "learning_rate": 7.59090909090909e-05, + "loss": 0.022249290347099306, + "step": 955 + }, + { + "epoch": 14.545454545454545, + "grad_norm": 2.1536738872528076, + "learning_rate": 7.578282828282828e-05, + "loss": 0.03471018075942993, + "step": 960 + }, + { + "epoch": 14.621212121212121, + "grad_norm": 0.09772758930921555, + "learning_rate": 7.565656565656566e-05, + "loss": 0.02227720320224762, + "step": 965 + }, + { + "epoch": 14.696969696969697, + "grad_norm": 0.09131006896495819, + "learning_rate": 7.553030303030303e-05, + "loss": 0.020403587818145753, + "step": 970 + }, + { + "epoch": 14.772727272727273, + "grad_norm": 0.0750042051076889, + "learning_rate": 7.540404040404041e-05, + "loss": 0.017551617324352266, + "step": 975 + }, + { + "epoch": 14.848484848484848, + "grad_norm": 0.07039610296487808, + "learning_rate": 7.527777777777777e-05, + "loss": 0.016474211215972902, + "step": 980 + }, + { + "epoch": 14.924242424242424, + "grad_norm": 0.11718424409627914, + "learning_rate": 7.515151515151515e-05, + "loss": 0.022238066792488097, + "step": 985 + }, + { + "epoch": 15.0, + "grad_norm": 2.1115448474884033, + "learning_rate": 7.502525252525253e-05, + "loss": 0.02658311426639557, + "step": 990 + }, + { + "epoch": 15.0, + "eval_loss": 0.02002105861902237, + "eval_mean_accuracy": 0.9639992210403894, + "eval_mean_iou": 0.9170223668076133, + "eval_runtime": 128.4867, + "eval_samples_per_second": 1.829, + "eval_steps_per_second": 0.233, + "step": 990 + }, + { + "epoch": 15.075757575757576, + "grad_norm": 0.09985354542732239, + "learning_rate": 7.48989898989899e-05, + "loss": 0.0173434779047966, + "step": 995 + }, + { + "epoch": 15.151515151515152, + "grad_norm": 0.06485751271247864, + "learning_rate": 7.477272727272727e-05, + "loss": 0.015555641055107117, + "step": 1000 + }, + { + "epoch": 15.227272727272727, + "grad_norm": 0.09712902456521988, + "learning_rate": 7.464646464646466e-05, + "loss": 0.017776939272880554, + "step": 1005 + }, + { + "epoch": 15.303030303030303, + "grad_norm": 0.11155439168214798, + "learning_rate": 7.452020202020202e-05, + "loss": 0.01771646589040756, + "step": 1010 + }, + { + "epoch": 15.378787878787879, + "grad_norm": 0.08052119612693787, + "learning_rate": 7.439393939393939e-05, + "loss": 0.01627490818500519, + "step": 1015 + }, + { + "epoch": 15.454545454545455, + "grad_norm": 0.6229738593101501, + "learning_rate": 7.426767676767677e-05, + "loss": 0.020942461490631104, + "step": 1020 + }, + { + "epoch": 15.530303030303031, + "grad_norm": 0.09126853942871094, + "learning_rate": 7.414141414141415e-05, + "loss": 0.019923874735832216, + "step": 1025 + }, + { + "epoch": 15.606060606060606, + "grad_norm": 0.09157100319862366, + "learning_rate": 7.401515151515152e-05, + "loss": 0.017950919270515443, + "step": 1030 + }, + { + "epoch": 15.681818181818182, + "grad_norm": 0.110823854804039, + "learning_rate": 7.38888888888889e-05, + "loss": 0.02105557769536972, + "step": 1035 + }, + { + "epoch": 15.757575757575758, + "grad_norm": 0.09924396127462387, + "learning_rate": 7.376262626262626e-05, + "loss": 0.018968561291694643, + "step": 1040 + }, + { + "epoch": 15.833333333333334, + "grad_norm": 0.08143289387226105, + "learning_rate": 7.363636363636364e-05, + "loss": 0.022947901487350465, + "step": 1045 + }, + { + "epoch": 15.909090909090908, + "grad_norm": 0.1267194002866745, + "learning_rate": 7.351010101010102e-05, + "loss": 0.0185988187789917, + "step": 1050 + }, + { + "epoch": 15.984848484848484, + "grad_norm": 0.07231700420379639, + "learning_rate": 7.338383838383839e-05, + "loss": 0.020301318168640135, + "step": 1055 + }, + { + "epoch": 16.0, + "eval_loss": 0.019123118370771408, + "eval_mean_accuracy": 0.9410152484466635, + "eval_mean_iou": 0.9174309737442452, + "eval_runtime": 131.1467, + "eval_samples_per_second": 1.792, + "eval_steps_per_second": 0.229, + "step": 1056 + }, + { + "epoch": 16.060606060606062, + "grad_norm": 0.09236351400613785, + "learning_rate": 7.325757575757576e-05, + "loss": 0.021160760521888734, + "step": 1060 + }, + { + "epoch": 16.136363636363637, + "grad_norm": 0.42284533381462097, + "learning_rate": 7.313131313131314e-05, + "loss": 0.02362883985042572, + "step": 1065 + }, + { + "epoch": 16.21212121212121, + "grad_norm": 0.09215152263641357, + "learning_rate": 7.300505050505051e-05, + "loss": 0.021066975593566895, + "step": 1070 + }, + { + "epoch": 16.28787878787879, + "grad_norm": 0.13141851127147675, + "learning_rate": 7.287878787878788e-05, + "loss": 0.01691439151763916, + "step": 1075 + }, + { + "epoch": 16.363636363636363, + "grad_norm": 0.07483583688735962, + "learning_rate": 7.275252525252526e-05, + "loss": 0.019986030459403992, + "step": 1080 + }, + { + "epoch": 16.439393939393938, + "grad_norm": 0.07349062711000443, + "learning_rate": 7.262626262626263e-05, + "loss": 0.017853912711143494, + "step": 1085 + }, + { + "epoch": 16.515151515151516, + "grad_norm": 0.23505988717079163, + "learning_rate": 7.25e-05, + "loss": 0.020364035665988923, + "step": 1090 + }, + { + "epoch": 16.59090909090909, + "grad_norm": 0.11297628283500671, + "learning_rate": 7.237373737373738e-05, + "loss": 0.015399885177612305, + "step": 1095 + }, + { + "epoch": 16.666666666666668, + "grad_norm": 0.2779926359653473, + "learning_rate": 7.224747474747476e-05, + "loss": 0.0256108820438385, + "step": 1100 + }, + { + "epoch": 16.742424242424242, + "grad_norm": 0.1852249801158905, + "learning_rate": 7.212121212121213e-05, + "loss": 0.016493414342403413, + "step": 1105 + }, + { + "epoch": 16.818181818181817, + "grad_norm": 0.6942260265350342, + "learning_rate": 7.199494949494949e-05, + "loss": 0.01953708827495575, + "step": 1110 + }, + { + "epoch": 16.893939393939394, + "grad_norm": 0.06553991138935089, + "learning_rate": 7.186868686868687e-05, + "loss": 0.015106955170631408, + "step": 1115 + }, + { + "epoch": 16.96969696969697, + "grad_norm": 0.10458805412054062, + "learning_rate": 7.174242424242425e-05, + "loss": 0.015331870317459107, + "step": 1120 + }, + { + "epoch": 17.0, + "eval_loss": 0.017818614840507507, + "eval_mean_accuracy": 0.9448605021013948, + "eval_mean_iou": 0.9206688461230607, + "eval_runtime": 131.3994, + "eval_samples_per_second": 1.788, + "eval_steps_per_second": 0.228, + "step": 1122 + }, + { + "epoch": 17.045454545454547, + "grad_norm": 0.09815305471420288, + "learning_rate": 7.161616161616162e-05, + "loss": 0.020230047404766083, + "step": 1125 + }, + { + "epoch": 17.12121212121212, + "grad_norm": 0.06650576740503311, + "learning_rate": 7.1489898989899e-05, + "loss": 0.01909434199333191, + "step": 1130 + }, + { + "epoch": 17.196969696969695, + "grad_norm": 0.0901411771774292, + "learning_rate": 7.136363636363636e-05, + "loss": 0.01694570332765579, + "step": 1135 + }, + { + "epoch": 17.272727272727273, + "grad_norm": 0.08640895783901215, + "learning_rate": 7.123737373737374e-05, + "loss": 0.01945408284664154, + "step": 1140 + }, + { + "epoch": 17.348484848484848, + "grad_norm": 1.4024585485458374, + "learning_rate": 7.111111111111112e-05, + "loss": 0.027021926641464234, + "step": 1145 + }, + { + "epoch": 17.424242424242426, + "grad_norm": 0.07610611617565155, + "learning_rate": 7.098484848484849e-05, + "loss": 0.01787877529859543, + "step": 1150 + }, + { + "epoch": 17.5, + "grad_norm": 0.16700884699821472, + "learning_rate": 7.085858585858585e-05, + "loss": 0.020749050378799438, + "step": 1155 + }, + { + "epoch": 17.575757575757574, + "grad_norm": 0.10327634960412979, + "learning_rate": 7.073232323232324e-05, + "loss": 0.01621636897325516, + "step": 1160 + }, + { + "epoch": 17.651515151515152, + "grad_norm": 0.07323160022497177, + "learning_rate": 7.060606060606061e-05, + "loss": 0.014837837219238282, + "step": 1165 + }, + { + "epoch": 17.727272727272727, + "grad_norm": 0.07253145426511765, + "learning_rate": 7.047979797979798e-05, + "loss": 0.014891114830970765, + "step": 1170 + }, + { + "epoch": 17.803030303030305, + "grad_norm": 0.0983441174030304, + "learning_rate": 7.035353535353536e-05, + "loss": 0.016994965076446534, + "step": 1175 + }, + { + "epoch": 17.87878787878788, + "grad_norm": 0.05063416808843613, + "learning_rate": 7.022727272727274e-05, + "loss": 0.01391943097114563, + "step": 1180 + }, + { + "epoch": 17.954545454545453, + "grad_norm": 0.14329107105731964, + "learning_rate": 7.01010101010101e-05, + "loss": 0.01766567528247833, + "step": 1185 + }, + { + "epoch": 18.0, + "eval_loss": 0.016699783504009247, + "eval_mean_accuracy": 0.9589740776141348, + "eval_mean_iou": 0.9253398888821036, + "eval_runtime": 125.8614, + "eval_samples_per_second": 1.867, + "eval_steps_per_second": 0.238, + "step": 1188 + }, + { + "epoch": 18.03030303030303, + "grad_norm": 0.22402475774288177, + "learning_rate": 6.997474747474748e-05, + "loss": 0.018416079878807067, + "step": 1190 + }, + { + "epoch": 18.106060606060606, + "grad_norm": 0.22749511897563934, + "learning_rate": 6.984848484848485e-05, + "loss": 0.015402013063430786, + "step": 1195 + }, + { + "epoch": 18.181818181818183, + "grad_norm": 0.06357621401548386, + "learning_rate": 6.972222222222223e-05, + "loss": 0.01626448631286621, + "step": 1200 + }, + { + "epoch": 18.257575757575758, + "grad_norm": 0.08117947727441788, + "learning_rate": 6.95959595959596e-05, + "loss": 0.01724429875612259, + "step": 1205 + }, + { + "epoch": 18.333333333333332, + "grad_norm": 0.06303979456424713, + "learning_rate": 6.946969696969697e-05, + "loss": 0.014101198315620423, + "step": 1210 + }, + { + "epoch": 18.40909090909091, + "grad_norm": 1.6291643381118774, + "learning_rate": 6.934343434343434e-05, + "loss": 0.022020891308784485, + "step": 1215 + }, + { + "epoch": 18.484848484848484, + "grad_norm": 0.07305492460727692, + "learning_rate": 6.921717171717173e-05, + "loss": 0.017137931287288667, + "step": 1220 + }, + { + "epoch": 18.560606060606062, + "grad_norm": 0.2717479169368744, + "learning_rate": 6.90909090909091e-05, + "loss": 0.020987211167812346, + "step": 1225 + }, + { + "epoch": 18.636363636363637, + "grad_norm": 0.07229124754667282, + "learning_rate": 6.896464646464646e-05, + "loss": 0.016514962911605834, + "step": 1230 + }, + { + "epoch": 18.71212121212121, + "grad_norm": 0.08333136886358261, + "learning_rate": 6.883838383838384e-05, + "loss": 0.017013299465179443, + "step": 1235 + }, + { + "epoch": 18.78787878787879, + "grad_norm": 1.3369598388671875, + "learning_rate": 6.871212121212122e-05, + "loss": 0.02812880277633667, + "step": 1240 + }, + { + "epoch": 18.863636363636363, + "grad_norm": 0.27156689763069153, + "learning_rate": 6.858585858585859e-05, + "loss": 0.01831343173980713, + "step": 1245 + }, + { + "epoch": 18.939393939393938, + "grad_norm": 4.329402446746826, + "learning_rate": 6.845959595959597e-05, + "loss": 0.024301397800445556, + "step": 1250 + }, + { + "epoch": 19.0, + "eval_loss": 0.017827482894062996, + "eval_mean_accuracy": 0.9493296531816283, + "eval_mean_iou": 0.9195720605328624, + "eval_runtime": 117.658, + "eval_samples_per_second": 1.997, + "eval_steps_per_second": 0.255, + "step": 1254 + }, + { + "epoch": 19.015151515151516, + "grad_norm": 0.10424228012561798, + "learning_rate": 6.833333333333333e-05, + "loss": 0.016584284603595734, + "step": 1255 + }, + { + "epoch": 19.09090909090909, + "grad_norm": 0.06917144358158112, + "learning_rate": 6.820707070707071e-05, + "loss": 0.01858946233987808, + "step": 1260 + }, + { + "epoch": 19.166666666666668, + "grad_norm": 0.1799507588148117, + "learning_rate": 6.808080808080809e-05, + "loss": 0.01758815050125122, + "step": 1265 + }, + { + "epoch": 19.242424242424242, + "grad_norm": 0.049295056611299515, + "learning_rate": 6.795454545454546e-05, + "loss": 0.01489059180021286, + "step": 1270 + }, + { + "epoch": 19.318181818181817, + "grad_norm": 0.0664125382900238, + "learning_rate": 6.782828282828284e-05, + "loss": 0.017305365204811095, + "step": 1275 + }, + { + "epoch": 19.393939393939394, + "grad_norm": 0.15921832621097565, + "learning_rate": 6.77020202020202e-05, + "loss": 0.01747938394546509, + "step": 1280 + }, + { + "epoch": 19.46969696969697, + "grad_norm": 0.34342673420906067, + "learning_rate": 6.757575757575758e-05, + "loss": 0.02069648206233978, + "step": 1285 + }, + { + "epoch": 19.545454545454547, + "grad_norm": 0.07065366208553314, + "learning_rate": 6.744949494949495e-05, + "loss": 0.013132335245609283, + "step": 1290 + }, + { + "epoch": 19.62121212121212, + "grad_norm": 0.05900472402572632, + "learning_rate": 6.732323232323233e-05, + "loss": 0.014775209128856659, + "step": 1295 + }, + { + "epoch": 19.696969696969695, + "grad_norm": 0.06478390097618103, + "learning_rate": 6.71969696969697e-05, + "loss": 0.015101474523544312, + "step": 1300 + }, + { + "epoch": 19.772727272727273, + "grad_norm": 0.07208125293254852, + "learning_rate": 6.707070707070707e-05, + "loss": 0.01790134757757187, + "step": 1305 + }, + { + "epoch": 19.848484848484848, + "grad_norm": 0.1395767629146576, + "learning_rate": 6.694444444444444e-05, + "loss": 0.014292587339878083, + "step": 1310 + }, + { + "epoch": 19.924242424242426, + "grad_norm": 0.10554281622171402, + "learning_rate": 6.681818181818183e-05, + "loss": 0.020696237683296204, + "step": 1315 + }, + { + "epoch": 20.0, + "grad_norm": 0.10756992548704147, + "learning_rate": 6.66919191919192e-05, + "loss": 0.015246658027172089, + "step": 1320 + }, + { + "epoch": 20.0, + "eval_loss": 0.016175229102373123, + "eval_mean_accuracy": 0.9459515055978925, + "eval_mean_iou": 0.9201830878124588, + "eval_runtime": 94.9763, + "eval_samples_per_second": 2.474, + "eval_steps_per_second": 0.316, + "step": 1320 + }, + { + "epoch": 20.075757575757574, + "grad_norm": 0.08596938103437424, + "learning_rate": 6.656565656565656e-05, + "loss": 0.0155582457780838, + "step": 1325 + }, + { + "epoch": 20.151515151515152, + "grad_norm": 0.08639311045408249, + "learning_rate": 6.643939393939394e-05, + "loss": 0.015870945155620576, + "step": 1330 + }, + { + "epoch": 20.227272727272727, + "grad_norm": 0.1362696886062622, + "learning_rate": 6.631313131313132e-05, + "loss": 0.016588638722896575, + "step": 1335 + }, + { + "epoch": 20.303030303030305, + "grad_norm": 0.07779170572757721, + "learning_rate": 6.618686868686869e-05, + "loss": 0.013284258544445038, + "step": 1340 + }, + { + "epoch": 20.37878787878788, + "grad_norm": 0.048727214336395264, + "learning_rate": 6.606060606060607e-05, + "loss": 0.012689882516860962, + "step": 1345 + }, + { + "epoch": 20.454545454545453, + "grad_norm": 0.07485265284776688, + "learning_rate": 6.593434343434343e-05, + "loss": 0.014190675318241119, + "step": 1350 + }, + { + "epoch": 20.53030303030303, + "grad_norm": 0.07521095871925354, + "learning_rate": 6.580808080808081e-05, + "loss": 0.01317392885684967, + "step": 1355 + }, + { + "epoch": 20.606060606060606, + "grad_norm": 0.04377813637256622, + "learning_rate": 6.568181818181819e-05, + "loss": 0.01863660365343094, + "step": 1360 + }, + { + "epoch": 20.681818181818183, + "grad_norm": 0.0933665931224823, + "learning_rate": 6.555555555555556e-05, + "loss": 0.015703774988651276, + "step": 1365 + }, + { + "epoch": 20.757575757575758, + "grad_norm": 0.06146784499287605, + "learning_rate": 6.542929292929292e-05, + "loss": 0.013644726574420929, + "step": 1370 + }, + { + "epoch": 20.833333333333332, + "grad_norm": 0.055801890790462494, + "learning_rate": 6.530303030303032e-05, + "loss": 0.013401885330677033, + "step": 1375 + }, + { + "epoch": 20.90909090909091, + "grad_norm": 0.06678389012813568, + "learning_rate": 6.517676767676768e-05, + "loss": 0.014624232053756714, + "step": 1380 + }, + { + "epoch": 20.984848484848484, + "grad_norm": 0.21429677307605743, + "learning_rate": 6.505050505050505e-05, + "loss": 0.01875179260969162, + "step": 1385 + }, + { + "epoch": 21.0, + "eval_loss": 0.015344121493399143, + "eval_mean_accuracy": 0.96329469845471, + "eval_mean_iou": 0.928221529445402, + "eval_runtime": 100.7114, + "eval_samples_per_second": 2.333, + "eval_steps_per_second": 0.298, + "step": 1386 + }, + { + "epoch": 21.060606060606062, + "grad_norm": 0.1103038415312767, + "learning_rate": 6.492424242424243e-05, + "loss": 0.014627225697040558, + "step": 1390 + }, + { + "epoch": 21.136363636363637, + "grad_norm": 0.0672246590256691, + "learning_rate": 6.479797979797981e-05, + "loss": 0.014656539261341094, + "step": 1395 + }, + { + "epoch": 21.21212121212121, + "grad_norm": 0.1095500960946083, + "learning_rate": 6.467171717171717e-05, + "loss": 0.015044647455215453, + "step": 1400 + }, + { + "epoch": 21.28787878787879, + "grad_norm": 0.05674279108643532, + "learning_rate": 6.454545454545455e-05, + "loss": 0.016598919034004213, + "step": 1405 + }, + { + "epoch": 21.363636363636363, + "grad_norm": 0.08919741958379745, + "learning_rate": 6.441919191919192e-05, + "loss": 0.011910013109445571, + "step": 1410 + }, + { + "epoch": 21.439393939393938, + "grad_norm": 0.07517007738351822, + "learning_rate": 6.42929292929293e-05, + "loss": 0.014272944629192352, + "step": 1415 + }, + { + "epoch": 21.515151515151516, + "grad_norm": 0.09998401999473572, + "learning_rate": 6.416666666666668e-05, + "loss": 0.015363068878650665, + "step": 1420 + }, + { + "epoch": 21.59090909090909, + "grad_norm": 0.046401698142290115, + "learning_rate": 6.404040404040404e-05, + "loss": 0.014173193275928498, + "step": 1425 + }, + { + "epoch": 21.666666666666668, + "grad_norm": 0.08298252522945404, + "learning_rate": 6.391414141414141e-05, + "loss": 0.013823199272155761, + "step": 1430 + }, + { + "epoch": 21.742424242424242, + "grad_norm": 0.2383887767791748, + "learning_rate": 6.37878787878788e-05, + "loss": 0.014954166114330291, + "step": 1435 + }, + { + "epoch": 21.818181818181817, + "grad_norm": 0.08410036563873291, + "learning_rate": 6.366161616161617e-05, + "loss": 0.014395049214363098, + "step": 1440 + }, + { + "epoch": 21.893939393939394, + "grad_norm": 0.07239001989364624, + "learning_rate": 6.353535353535353e-05, + "loss": 0.016220720112323762, + "step": 1445 + }, + { + "epoch": 21.96969696969697, + "grad_norm": 0.03885965421795845, + "learning_rate": 6.340909090909091e-05, + "loss": 0.012613791227340698, + "step": 1450 + }, + { + "epoch": 22.0, + "eval_loss": 0.015054116025567055, + "eval_mean_accuracy": 0.9460838663239403, + "eval_mean_iou": 0.9224819473072445, + "eval_runtime": 130.6073, + "eval_samples_per_second": 1.799, + "eval_steps_per_second": 0.23, + "step": 1452 + }, + { + "epoch": 22.045454545454547, + "grad_norm": 0.07882537692785263, + "learning_rate": 6.328282828282829e-05, + "loss": 0.013986352086067199, + "step": 1455 + }, + { + "epoch": 22.12121212121212, + "grad_norm": 0.056629691272974014, + "learning_rate": 6.315656565656566e-05, + "loss": 0.012423294782638549, + "step": 1460 + }, + { + "epoch": 22.196969696969695, + "grad_norm": 0.0581502690911293, + "learning_rate": 6.303030303030302e-05, + "loss": 0.012547774612903595, + "step": 1465 + }, + { + "epoch": 22.272727272727273, + "grad_norm": 0.05007210001349449, + "learning_rate": 6.29040404040404e-05, + "loss": 0.012538580596446991, + "step": 1470 + }, + { + "epoch": 22.348484848484848, + "grad_norm": 0.06412877142429352, + "learning_rate": 6.277777777777778e-05, + "loss": 0.01328524351119995, + "step": 1475 + }, + { + "epoch": 22.424242424242426, + "grad_norm": 0.062360070645809174, + "learning_rate": 6.265151515151515e-05, + "loss": 0.013031665980815888, + "step": 1480 + }, + { + "epoch": 22.5, + "grad_norm": 0.08252230286598206, + "learning_rate": 6.252525252525253e-05, + "loss": 0.012258045375347137, + "step": 1485 + }, + { + "epoch": 22.575757575757574, + "grad_norm": 0.05508822202682495, + "learning_rate": 6.239898989898991e-05, + "loss": 0.011711065471172333, + "step": 1490 + }, + { + "epoch": 22.651515151515152, + "grad_norm": 0.1369508057832718, + "learning_rate": 6.227272727272727e-05, + "loss": 0.015299557149410248, + "step": 1495 + }, + { + "epoch": 22.727272727272727, + "grad_norm": 0.0403149351477623, + "learning_rate": 6.214646464646465e-05, + "loss": 0.013164618611335754, + "step": 1500 + }, + { + "epoch": 22.803030303030305, + "grad_norm": 0.1833992451429367, + "learning_rate": 6.202020202020202e-05, + "loss": 0.017351745069026946, + "step": 1505 + }, + { + "epoch": 22.87878787878788, + "grad_norm": 0.0749652236700058, + "learning_rate": 6.18939393939394e-05, + "loss": 0.01598891019821167, + "step": 1510 + }, + { + "epoch": 22.954545454545453, + "grad_norm": 0.04321586340665817, + "learning_rate": 6.176767676767678e-05, + "loss": 0.015352188050746918, + "step": 1515 + }, + { + "epoch": 23.0, + "eval_loss": 0.014744299463927746, + "eval_mean_accuracy": 0.9475568060044316, + "eval_mean_iou": 0.9228820618158111, + "eval_runtime": 134.2869, + "eval_samples_per_second": 1.75, + "eval_steps_per_second": 0.223, + "step": 1518 + }, + { + "epoch": 23.03030303030303, + "grad_norm": 0.05272269248962402, + "learning_rate": 6.164141414141414e-05, + "loss": 0.014622032642364502, + "step": 1520 + }, + { + "epoch": 23.106060606060606, + "grad_norm": 0.04899510741233826, + "learning_rate": 6.151515151515151e-05, + "loss": 0.017201478779315948, + "step": 1525 + }, + { + "epoch": 23.181818181818183, + "grad_norm": 0.23097753524780273, + "learning_rate": 6.13888888888889e-05, + "loss": 0.0176813006401062, + "step": 1530 + }, + { + "epoch": 23.257575757575758, + "grad_norm": 0.1082717701792717, + "learning_rate": 6.126262626262627e-05, + "loss": 0.01365693062543869, + "step": 1535 + }, + { + "epoch": 23.333333333333332, + "grad_norm": 0.04835253953933716, + "learning_rate": 6.113636363636363e-05, + "loss": 0.01303461343050003, + "step": 1540 + }, + { + "epoch": 23.40909090909091, + "grad_norm": 0.16811569035053253, + "learning_rate": 6.101010101010102e-05, + "loss": 0.01358899474143982, + "step": 1545 + }, + { + "epoch": 23.484848484848484, + "grad_norm": 0.07286785542964935, + "learning_rate": 6.0883838383838386e-05, + "loss": 0.013001954555511475, + "step": 1550 + }, + { + "epoch": 23.560606060606062, + "grad_norm": 0.03553387150168419, + "learning_rate": 6.075757575757576e-05, + "loss": 0.012401340901851654, + "step": 1555 + }, + { + "epoch": 23.636363636363637, + "grad_norm": 0.07686861604452133, + "learning_rate": 6.063131313131314e-05, + "loss": 0.011214424669742585, + "step": 1560 + }, + { + "epoch": 23.71212121212121, + "grad_norm": 0.05435880273580551, + "learning_rate": 6.050505050505051e-05, + "loss": 0.013267573714256287, + "step": 1565 + }, + { + "epoch": 23.78787878787879, + "grad_norm": 0.19973690807819366, + "learning_rate": 6.037878787878788e-05, + "loss": 0.01502605527639389, + "step": 1570 + }, + { + "epoch": 23.863636363636363, + "grad_norm": 0.05224163830280304, + "learning_rate": 6.025252525252526e-05, + "loss": 0.01584022045135498, + "step": 1575 + }, + { + "epoch": 23.939393939393938, + "grad_norm": 0.0936981588602066, + "learning_rate": 6.012626262626263e-05, + "loss": 0.015465947985649108, + "step": 1580 + }, + { + "epoch": 24.0, + "eval_loss": 0.014172257855534554, + "eval_mean_accuracy": 0.9514972485814469, + "eval_mean_iou": 0.9252738006682145, + "eval_runtime": 120.3105, + "eval_samples_per_second": 1.953, + "eval_steps_per_second": 0.249, + "step": 1584 + }, + { + "epoch": 24.015151515151516, + "grad_norm": 0.07904013246297836, + "learning_rate": 6e-05, + "loss": 0.02108345031738281, + "step": 1585 + }, + { + "epoch": 24.09090909090909, + "grad_norm": 0.08464734256267548, + "learning_rate": 5.987373737373738e-05, + "loss": 0.01407742202281952, + "step": 1590 + }, + { + "epoch": 24.166666666666668, + "grad_norm": 0.14806444942951202, + "learning_rate": 5.9747474747474754e-05, + "loss": 0.013575415313243865, + "step": 1595 + }, + { + "epoch": 24.242424242424242, + "grad_norm": 0.12283805012702942, + "learning_rate": 5.962121212121212e-05, + "loss": 0.011667483299970628, + "step": 1600 + }, + { + "epoch": 24.318181818181817, + "grad_norm": 0.0784803107380867, + "learning_rate": 5.949494949494949e-05, + "loss": 0.012746933102607726, + "step": 1605 + }, + { + "epoch": 24.393939393939394, + "grad_norm": 0.04776550456881523, + "learning_rate": 5.936868686868687e-05, + "loss": 0.011614951491355895, + "step": 1610 + }, + { + "epoch": 24.46969696969697, + "grad_norm": 0.040872711688280106, + "learning_rate": 5.9242424242424244e-05, + "loss": 0.013041725754737854, + "step": 1615 + }, + { + "epoch": 24.545454545454547, + "grad_norm": 0.03968976065516472, + "learning_rate": 5.911616161616162e-05, + "loss": 0.011115564405918122, + "step": 1620 + }, + { + "epoch": 24.62121212121212, + "grad_norm": 0.26981595158576965, + "learning_rate": 5.8989898989898996e-05, + "loss": 0.016104865074157714, + "step": 1625 + }, + { + "epoch": 24.696969696969695, + "grad_norm": 0.0788557156920433, + "learning_rate": 5.886363636363636e-05, + "loss": 0.018504132330417634, + "step": 1630 + }, + { + "epoch": 24.772727272727273, + "grad_norm": 0.05485096573829651, + "learning_rate": 5.8737373737373735e-05, + "loss": 0.018993689119815825, + "step": 1635 + }, + { + "epoch": 24.848484848484848, + "grad_norm": 0.5279337763786316, + "learning_rate": 5.8611111111111114e-05, + "loss": 0.012673009932041169, + "step": 1640 + }, + { + "epoch": 24.924242424242426, + "grad_norm": 0.12343566864728928, + "learning_rate": 5.848484848484849e-05, + "loss": 0.013657009601593018, + "step": 1645 + }, + { + "epoch": 25.0, + "grad_norm": 0.09911590814590454, + "learning_rate": 5.835858585858586e-05, + "loss": 0.015301503241062164, + "step": 1650 + }, + { + "epoch": 25.0, + "eval_loss": 0.01369437389075756, + "eval_mean_accuracy": 0.9547303232473082, + "eval_mean_iou": 0.9269078204686916, + "eval_runtime": 128.4617, + "eval_samples_per_second": 1.829, + "eval_steps_per_second": 0.234, + "step": 1650 + }, + { + "epoch": 25.075757575757574, + "grad_norm": 0.07376221567392349, + "learning_rate": 5.823232323232324e-05, + "loss": 0.013502883911132812, + "step": 1655 + }, + { + "epoch": 25.151515151515152, + "grad_norm": 0.037124961614608765, + "learning_rate": 5.810606060606061e-05, + "loss": 0.013181290030479431, + "step": 1660 + }, + { + "epoch": 25.227272727272727, + "grad_norm": 0.056431844830513, + "learning_rate": 5.797979797979798e-05, + "loss": 0.010531653463840485, + "step": 1665 + }, + { + "epoch": 25.303030303030305, + "grad_norm": 0.19079795479774475, + "learning_rate": 5.785353535353536e-05, + "loss": 0.01903575360774994, + "step": 1670 + }, + { + "epoch": 25.37878787878788, + "grad_norm": 0.06505732983350754, + "learning_rate": 5.772727272727273e-05, + "loss": 0.014971224963665009, + "step": 1675 + }, + { + "epoch": 25.454545454545453, + "grad_norm": 0.05066489800810814, + "learning_rate": 5.76010101010101e-05, + "loss": 0.009277631342411042, + "step": 1680 + }, + { + "epoch": 25.53030303030303, + "grad_norm": 0.04367240145802498, + "learning_rate": 5.747474747474748e-05, + "loss": 0.014411757886409759, + "step": 1685 + }, + { + "epoch": 25.606060606060606, + "grad_norm": 0.05328993499279022, + "learning_rate": 5.7348484848484854e-05, + "loss": 0.013387630879878997, + "step": 1690 + }, + { + "epoch": 25.681818181818183, + "grad_norm": 0.09997580200433731, + "learning_rate": 5.722222222222222e-05, + "loss": 0.013551540672779083, + "step": 1695 + }, + { + "epoch": 25.757575757575758, + "grad_norm": 0.09678027033805847, + "learning_rate": 5.70959595959596e-05, + "loss": 0.014515113830566407, + "step": 1700 + }, + { + "epoch": 25.833333333333332, + "grad_norm": 0.063142329454422, + "learning_rate": 5.696969696969697e-05, + "loss": 0.0132255420088768, + "step": 1705 + }, + { + "epoch": 25.90909090909091, + "grad_norm": 0.0717817097902298, + "learning_rate": 5.6843434343434345e-05, + "loss": 0.011613220721483231, + "step": 1710 + }, + { + "epoch": 25.984848484848484, + "grad_norm": 0.09518913924694061, + "learning_rate": 5.6717171717171724e-05, + "loss": 0.016637946665287017, + "step": 1715 + }, + { + "epoch": 26.0, + "eval_loss": 0.014714955352246761, + "eval_mean_accuracy": 0.9672183179842754, + "eval_mean_iou": 0.926534743485111, + "eval_runtime": 129.1171, + "eval_samples_per_second": 1.82, + "eval_steps_per_second": 0.232, + "step": 1716 + }, + { + "epoch": 26.060606060606062, + "grad_norm": 0.056166984140872955, + "learning_rate": 5.65909090909091e-05, + "loss": 0.013209113478660583, + "step": 1720 + }, + { + "epoch": 26.136363636363637, + "grad_norm": 0.05872166156768799, + "learning_rate": 5.646464646464646e-05, + "loss": 0.013837473094463348, + "step": 1725 + }, + { + "epoch": 26.21212121212121, + "grad_norm": 0.09089943766593933, + "learning_rate": 5.633838383838385e-05, + "loss": 0.012034112215042114, + "step": 1730 + }, + { + "epoch": 26.28787878787879, + "grad_norm": 0.17554797232151031, + "learning_rate": 5.6212121212121215e-05, + "loss": 0.013125964999198913, + "step": 1735 + }, + { + "epoch": 26.363636363636363, + "grad_norm": 0.08835634589195251, + "learning_rate": 5.608585858585859e-05, + "loss": 0.011378810554742814, + "step": 1740 + }, + { + "epoch": 26.439393939393938, + "grad_norm": 0.18092839419841766, + "learning_rate": 5.595959595959597e-05, + "loss": 0.012567883729934693, + "step": 1745 + }, + { + "epoch": 26.515151515151516, + "grad_norm": 0.2868044972419739, + "learning_rate": 5.583333333333334e-05, + "loss": 0.020671078562736513, + "step": 1750 + }, + { + "epoch": 26.59090909090909, + "grad_norm": 0.06546394526958466, + "learning_rate": 5.5707070707070706e-05, + "loss": 0.013370734453201295, + "step": 1755 + }, + { + "epoch": 26.666666666666668, + "grad_norm": 0.07877084612846375, + "learning_rate": 5.558080808080809e-05, + "loss": 0.012026229500770569, + "step": 1760 + }, + { + "epoch": 26.742424242424242, + "grad_norm": 0.039776258170604706, + "learning_rate": 5.545454545454546e-05, + "loss": 0.01264951229095459, + "step": 1765 + }, + { + "epoch": 26.818181818181817, + "grad_norm": 0.07569372653961182, + "learning_rate": 5.532828282828283e-05, + "loss": 0.012059577554464341, + "step": 1770 + }, + { + "epoch": 26.893939393939394, + "grad_norm": 0.421996146440506, + "learning_rate": 5.5202020202020196e-05, + "loss": 0.01300893872976303, + "step": 1775 + }, + { + "epoch": 26.96969696969697, + "grad_norm": 0.03665846213698387, + "learning_rate": 5.507575757575758e-05, + "loss": 0.012992760539054871, + "step": 1780 + }, + { + "epoch": 27.0, + "eval_loss": 0.014141011983156204, + "eval_mean_accuracy": 0.9426869656654299, + "eval_mean_iou": 0.9207559456923922, + "eval_runtime": 129.6443, + "eval_samples_per_second": 1.813, + "eval_steps_per_second": 0.231, + "step": 1782 + }, + { + "epoch": 27.045454545454547, + "grad_norm": 0.06108195334672928, + "learning_rate": 5.494949494949495e-05, + "loss": 0.013349318504333496, + "step": 1785 + }, + { + "epoch": 27.12121212121212, + "grad_norm": 0.1034199669957161, + "learning_rate": 5.482323232323232e-05, + "loss": 0.016627094149589537, + "step": 1790 + }, + { + "epoch": 27.196969696969695, + "grad_norm": 0.05061526596546173, + "learning_rate": 5.46969696969697e-05, + "loss": 0.013697353005409241, + "step": 1795 + }, + { + "epoch": 27.272727272727273, + "grad_norm": 0.09373223036527634, + "learning_rate": 5.457070707070707e-05, + "loss": 0.010963685810565948, + "step": 1800 + }, + { + "epoch": 27.348484848484848, + "grad_norm": 0.03771704435348511, + "learning_rate": 5.4444444444444446e-05, + "loss": 0.01018289551138878, + "step": 1805 + }, + { + "epoch": 27.424242424242426, + "grad_norm": 0.07621210813522339, + "learning_rate": 5.4318181818181825e-05, + "loss": 0.015036375820636749, + "step": 1810 + }, + { + "epoch": 27.5, + "grad_norm": 0.06018628925085068, + "learning_rate": 5.419191919191919e-05, + "loss": 0.012247322499752045, + "step": 1815 + }, + { + "epoch": 27.575757575757574, + "grad_norm": 0.07586652040481567, + "learning_rate": 5.4065656565656564e-05, + "loss": 0.01254344880580902, + "step": 1820 + }, + { + "epoch": 27.651515151515152, + "grad_norm": 0.0791565328836441, + "learning_rate": 5.393939393939394e-05, + "loss": 0.013501356542110442, + "step": 1825 + }, + { + "epoch": 27.727272727272727, + "grad_norm": 0.055143654346466064, + "learning_rate": 5.3813131313131316e-05, + "loss": 0.013556474447250366, + "step": 1830 + }, + { + "epoch": 27.803030303030305, + "grad_norm": 0.10083472728729248, + "learning_rate": 5.368686868686869e-05, + "loss": 0.012916183471679688, + "step": 1835 + }, + { + "epoch": 27.87878787878788, + "grad_norm": 0.07956752926111221, + "learning_rate": 5.356060606060607e-05, + "loss": 0.013280309736728668, + "step": 1840 + }, + { + "epoch": 27.954545454545453, + "grad_norm": 0.15146367251873016, + "learning_rate": 5.3434343434343434e-05, + "loss": 0.013282467424869538, + "step": 1845 + }, + { + "epoch": 28.0, + "eval_loss": 0.013491583988070488, + "eval_mean_accuracy": 0.9655628816751172, + "eval_mean_iou": 0.9288944108198042, + "eval_runtime": 103.8875, + "eval_samples_per_second": 2.262, + "eval_steps_per_second": 0.289, + "step": 1848 + }, + { + "epoch": 28.03030303030303, + "grad_norm": 0.07574369013309479, + "learning_rate": 5.3308080808080806e-05, + "loss": 0.010976041853427886, + "step": 1850 + }, + { + "epoch": 28.106060606060606, + "grad_norm": 0.0810118168592453, + "learning_rate": 5.3181818181818186e-05, + "loss": 0.009929781407117843, + "step": 1855 + }, + { + "epoch": 28.181818181818183, + "grad_norm": 0.05681147053837776, + "learning_rate": 5.305555555555556e-05, + "loss": 0.01240130290389061, + "step": 1860 + }, + { + "epoch": 28.257575757575758, + "grad_norm": 0.07306916266679764, + "learning_rate": 5.292929292929293e-05, + "loss": 0.012015487253665923, + "step": 1865 + }, + { + "epoch": 28.333333333333332, + "grad_norm": 0.15595616400241852, + "learning_rate": 5.280303030303031e-05, + "loss": 0.013296128809452057, + "step": 1870 + }, + { + "epoch": 28.40909090909091, + "grad_norm": 0.03395576775074005, + "learning_rate": 5.267676767676768e-05, + "loss": 0.0084619402885437, + "step": 1875 + }, + { + "epoch": 28.484848484848484, + "grad_norm": 0.07785443216562271, + "learning_rate": 5.255050505050505e-05, + "loss": 0.013635687530040741, + "step": 1880 + }, + { + "epoch": 28.560606060606062, + "grad_norm": 0.046345312148332596, + "learning_rate": 5.242424242424243e-05, + "loss": 0.012098722159862518, + "step": 1885 + }, + { + "epoch": 28.636363636363637, + "grad_norm": 0.029776595532894135, + "learning_rate": 5.22979797979798e-05, + "loss": 0.011485729366540909, + "step": 1890 + }, + { + "epoch": 28.71212121212121, + "grad_norm": 0.06558804214000702, + "learning_rate": 5.2171717171717174e-05, + "loss": 0.015296250581741333, + "step": 1895 + }, + { + "epoch": 28.78787878787879, + "grad_norm": 0.03267960250377655, + "learning_rate": 5.204545454545455e-05, + "loss": 0.012894335389137267, + "step": 1900 + }, + { + "epoch": 28.863636363636363, + "grad_norm": 0.03641599416732788, + "learning_rate": 5.1919191919191926e-05, + "loss": 0.012232333421707153, + "step": 1905 + }, + { + "epoch": 28.939393939393938, + "grad_norm": 0.06189272180199623, + "learning_rate": 5.179292929292929e-05, + "loss": 0.013554035127162934, + "step": 1910 + }, + { + "epoch": 29.0, + "eval_loss": 0.013112721964716911, + "eval_mean_accuracy": 0.9672514398960311, + "eval_mean_iou": 0.9295206182263173, + "eval_runtime": 121.3817, + "eval_samples_per_second": 1.936, + "eval_steps_per_second": 0.247, + "step": 1914 + }, + { + "epoch": 29.015151515151516, + "grad_norm": 0.07210548222064972, + "learning_rate": 5.166666666666667e-05, + "loss": 0.01132163405418396, + "step": 1915 + }, + { + "epoch": 29.09090909090909, + "grad_norm": 0.04434465616941452, + "learning_rate": 5.1540404040404044e-05, + "loss": 0.012844860553741455, + "step": 1920 + }, + { + "epoch": 29.166666666666668, + "grad_norm": 0.08480337262153625, + "learning_rate": 5.1414141414141416e-05, + "loss": 0.010472848266363143, + "step": 1925 + }, + { + "epoch": 29.242424242424242, + "grad_norm": 0.05567143112421036, + "learning_rate": 5.1287878787878796e-05, + "loss": 0.010616761445999146, + "step": 1930 + }, + { + "epoch": 29.318181818181817, + "grad_norm": 0.08190970122814178, + "learning_rate": 5.116161616161617e-05, + "loss": 0.012078990787267685, + "step": 1935 + }, + { + "epoch": 29.393939393939394, + "grad_norm": 0.06780913472175598, + "learning_rate": 5.1035353535353534e-05, + "loss": 0.015170016884803772, + "step": 1940 + }, + { + "epoch": 29.46969696969697, + "grad_norm": 0.0979032963514328, + "learning_rate": 5.090909090909091e-05, + "loss": 0.009976643323898315, + "step": 1945 + }, + { + "epoch": 29.545454545454547, + "grad_norm": 0.13055600225925446, + "learning_rate": 5.0782828282828287e-05, + "loss": 0.015534821152687072, + "step": 1950 + }, + { + "epoch": 29.62121212121212, + "grad_norm": 0.04829199239611626, + "learning_rate": 5.065656565656566e-05, + "loss": 0.011643566191196442, + "step": 1955 + }, + { + "epoch": 29.696969696969695, + "grad_norm": 0.05333337560296059, + "learning_rate": 5.0530303030303025e-05, + "loss": 0.010874416679143906, + "step": 1960 + }, + { + "epoch": 29.772727272727273, + "grad_norm": 0.08430800586938858, + "learning_rate": 5.040404040404041e-05, + "loss": 0.013982926309108735, + "step": 1965 + }, + { + "epoch": 29.848484848484848, + "grad_norm": 0.057856492698192596, + "learning_rate": 5.027777777777778e-05, + "loss": 0.01478300839662552, + "step": 1970 + }, + { + "epoch": 29.924242424242426, + "grad_norm": 0.05532738193869591, + "learning_rate": 5.015151515151515e-05, + "loss": 0.013097742199897766, + "step": 1975 + }, + { + "epoch": 30.0, + "grad_norm": 0.18274356424808502, + "learning_rate": 5.002525252525253e-05, + "loss": 0.01009620875120163, + "step": 1980 + }, + { + "epoch": 30.0, + "eval_loss": 0.013003743253648281, + "eval_mean_accuracy": 0.9606068406216899, + "eval_mean_iou": 0.9301546047278547, + "eval_runtime": 102.0507, + "eval_samples_per_second": 2.303, + "eval_steps_per_second": 0.294, + "step": 1980 + }, + { + "epoch": 30.075757575757574, + "grad_norm": 0.06038985028862953, + "learning_rate": 4.98989898989899e-05, + "loss": 0.00952836349606514, + "step": 1985 + }, + { + "epoch": 30.151515151515152, + "grad_norm": 0.10457076877355576, + "learning_rate": 4.9772727272727275e-05, + "loss": 0.012150304019451141, + "step": 1990 + }, + { + "epoch": 30.227272727272727, + "grad_norm": 0.31461235880851746, + "learning_rate": 4.964646464646465e-05, + "loss": 0.012803596258163453, + "step": 1995 + }, + { + "epoch": 30.303030303030305, + "grad_norm": 0.12801575660705566, + "learning_rate": 4.952020202020202e-05, + "loss": 0.009476866573095322, + "step": 2000 + }, + { + "epoch": 30.37878787878788, + "grad_norm": 0.0345924012362957, + "learning_rate": 4.93939393939394e-05, + "loss": 0.012969899177551269, + "step": 2005 + }, + { + "epoch": 30.454545454545453, + "grad_norm": 0.0917559266090393, + "learning_rate": 4.9267676767676765e-05, + "loss": 0.012600681185722351, + "step": 2010 + }, + { + "epoch": 30.53030303030303, + "grad_norm": 0.1583668440580368, + "learning_rate": 4.9141414141414145e-05, + "loss": 0.015196573734283448, + "step": 2015 + }, + { + "epoch": 30.606060606060606, + "grad_norm": 0.08989228308200836, + "learning_rate": 4.901515151515152e-05, + "loss": 0.01427859365940094, + "step": 2020 + }, + { + "epoch": 30.681818181818183, + "grad_norm": 0.20922039449214935, + "learning_rate": 4.888888888888889e-05, + "loss": 0.012277430295944214, + "step": 2025 + }, + { + "epoch": 30.757575757575758, + "grad_norm": 0.04618125408887863, + "learning_rate": 4.876262626262626e-05, + "loss": 0.015073756873607635, + "step": 2030 + }, + { + "epoch": 30.833333333333332, + "grad_norm": 0.09186513721942902, + "learning_rate": 4.863636363636364e-05, + "loss": 0.009782224893569946, + "step": 2035 + }, + { + "epoch": 30.90909090909091, + "grad_norm": 0.09393519163131714, + "learning_rate": 4.851010101010101e-05, + "loss": 0.01244906559586525, + "step": 2040 + }, + { + "epoch": 30.984848484848484, + "grad_norm": 0.04698757082223892, + "learning_rate": 4.838383838383839e-05, + "loss": 0.011347094923257828, + "step": 2045 + }, + { + "epoch": 31.0, + "eval_loss": 0.012866747565567493, + "eval_mean_accuracy": 0.9574178509480215, + "eval_mean_iou": 0.929484823554314, + "eval_runtime": 121.0946, + "eval_samples_per_second": 1.941, + "eval_steps_per_second": 0.248, + "step": 2046 + }, + { + "epoch": 31.060606060606062, + "grad_norm": 0.0460100881755352, + "learning_rate": 4.825757575757576e-05, + "loss": 0.009506873041391372, + "step": 2050 + }, + { + "epoch": 31.136363636363637, + "grad_norm": 0.08907371014356613, + "learning_rate": 4.813131313131313e-05, + "loss": 0.012121839821338654, + "step": 2055 + }, + { + "epoch": 31.21212121212121, + "grad_norm": 0.053708516061306, + "learning_rate": 4.8005050505050505e-05, + "loss": 0.01105894073843956, + "step": 2060 + }, + { + "epoch": 31.28787878787879, + "grad_norm": 0.029567470774054527, + "learning_rate": 4.787878787878788e-05, + "loss": 0.008609072118997575, + "step": 2065 + }, + { + "epoch": 31.363636363636363, + "grad_norm": 0.08797413855791092, + "learning_rate": 4.775252525252526e-05, + "loss": 0.011014562845230103, + "step": 2070 + }, + { + "epoch": 31.439393939393938, + "grad_norm": 0.08459515124559402, + "learning_rate": 4.762626262626263e-05, + "loss": 0.011200347542762756, + "step": 2075 + }, + { + "epoch": 31.515151515151516, + "grad_norm": 0.053671907633543015, + "learning_rate": 4.75e-05, + "loss": 0.010198388248682022, + "step": 2080 + }, + { + "epoch": 31.59090909090909, + "grad_norm": 0.0929805114865303, + "learning_rate": 4.7373737373737375e-05, + "loss": 0.013522854447364807, + "step": 2085 + }, + { + "epoch": 31.666666666666668, + "grad_norm": 0.10578818619251251, + "learning_rate": 4.7247474747474755e-05, + "loss": 0.013305488228797912, + "step": 2090 + }, + { + "epoch": 31.742424242424242, + "grad_norm": 0.05142006650567055, + "learning_rate": 4.712121212121212e-05, + "loss": 0.009187790006399155, + "step": 2095 + }, + { + "epoch": 31.818181818181817, + "grad_norm": 0.08935363590717316, + "learning_rate": 4.69949494949495e-05, + "loss": 0.014767760038375854, + "step": 2100 + }, + { + "epoch": 31.893939393939394, + "grad_norm": 0.1454441249370575, + "learning_rate": 4.686868686868687e-05, + "loss": 0.014438904821872711, + "step": 2105 + }, + { + "epoch": 31.96969696969697, + "grad_norm": 0.06975588947534561, + "learning_rate": 4.6742424242424245e-05, + "loss": 0.012650656700134277, + "step": 2110 + }, + { + "epoch": 32.0, + "eval_loss": 0.012378966435790062, + "eval_mean_accuracy": 0.9605900631726554, + "eval_mean_iou": 0.9310517643124889, + "eval_runtime": 133.4744, + "eval_samples_per_second": 1.761, + "eval_steps_per_second": 0.225, + "step": 2112 + }, + { + "epoch": 32.04545454545455, + "grad_norm": 0.04895917698740959, + "learning_rate": 4.661616161616162e-05, + "loss": 0.013399934768676758, + "step": 2115 + }, + { + "epoch": 32.121212121212125, + "grad_norm": 0.03478637710213661, + "learning_rate": 4.6489898989899e-05, + "loss": 0.010184010118246078, + "step": 2120 + }, + { + "epoch": 32.196969696969695, + "grad_norm": 0.07323820143938065, + "learning_rate": 4.636363636363636e-05, + "loss": 0.013803583383560181, + "step": 2125 + }, + { + "epoch": 32.27272727272727, + "grad_norm": 0.03756466880440712, + "learning_rate": 4.623737373737374e-05, + "loss": 0.010909663140773773, + "step": 2130 + }, + { + "epoch": 32.34848484848485, + "grad_norm": 0.09230446070432663, + "learning_rate": 4.6111111111111115e-05, + "loss": 0.012808184325695037, + "step": 2135 + }, + { + "epoch": 32.42424242424242, + "grad_norm": 0.06738702207803726, + "learning_rate": 4.598484848484849e-05, + "loss": 0.010556499660015106, + "step": 2140 + }, + { + "epoch": 32.5, + "grad_norm": 0.05251049995422363, + "learning_rate": 4.585858585858586e-05, + "loss": 0.013995076715946197, + "step": 2145 + }, + { + "epoch": 32.57575757575758, + "grad_norm": 0.028716392815113068, + "learning_rate": 4.573232323232323e-05, + "loss": 0.014915336668491364, + "step": 2150 + }, + { + "epoch": 32.65151515151515, + "grad_norm": 0.111383818089962, + "learning_rate": 4.5606060606060606e-05, + "loss": 0.010672840476036071, + "step": 2155 + }, + { + "epoch": 32.72727272727273, + "grad_norm": 0.06055324897170067, + "learning_rate": 4.5479797979797985e-05, + "loss": 0.009975516051054002, + "step": 2160 + }, + { + "epoch": 32.803030303030305, + "grad_norm": 0.10540654510259628, + "learning_rate": 4.535353535353535e-05, + "loss": 0.010461199283599853, + "step": 2165 + }, + { + "epoch": 32.878787878787875, + "grad_norm": 0.03463799133896828, + "learning_rate": 4.522727272727273e-05, + "loss": 0.013615190982818604, + "step": 2170 + }, + { + "epoch": 32.95454545454545, + "grad_norm": 0.03951095789670944, + "learning_rate": 4.51010101010101e-05, + "loss": 0.012134815007448197, + "step": 2175 + }, + { + "epoch": 33.0, + "eval_loss": 0.012309151701629162, + "eval_mean_accuracy": 0.9624573979079548, + "eval_mean_iou": 0.9320484887522253, + "eval_runtime": 128.0787, + "eval_samples_per_second": 1.835, + "eval_steps_per_second": 0.234, + "step": 2178 + }, + { + "epoch": 33.03030303030303, + "grad_norm": 0.0988498404622078, + "learning_rate": 4.4974747474747476e-05, + "loss": 0.012665390968322754, + "step": 2180 + }, + { + "epoch": 33.10606060606061, + "grad_norm": 0.03966812044382095, + "learning_rate": 4.484848484848485e-05, + "loss": 0.011929359287023544, + "step": 2185 + }, + { + "epoch": 33.18181818181818, + "grad_norm": 0.08401142805814743, + "learning_rate": 4.472222222222223e-05, + "loss": 0.014081376791000366, + "step": 2190 + }, + { + "epoch": 33.25757575757576, + "grad_norm": 0.056711092591285706, + "learning_rate": 4.4595959595959594e-05, + "loss": 0.01047600582242012, + "step": 2195 + }, + { + "epoch": 33.333333333333336, + "grad_norm": 0.07973317056894302, + "learning_rate": 4.4469696969696973e-05, + "loss": 0.010295311361551285, + "step": 2200 + }, + { + "epoch": 33.40909090909091, + "grad_norm": 0.057509079575538635, + "learning_rate": 4.4343434343434346e-05, + "loss": 0.011236566305160522, + "step": 2205 + }, + { + "epoch": 33.484848484848484, + "grad_norm": 0.02464018575847149, + "learning_rate": 4.421717171717172e-05, + "loss": 0.011486275494098664, + "step": 2210 + }, + { + "epoch": 33.56060606060606, + "grad_norm": 0.055086731910705566, + "learning_rate": 4.409090909090909e-05, + "loss": 0.010625626891851425, + "step": 2215 + }, + { + "epoch": 33.63636363636363, + "grad_norm": 0.037378594279289246, + "learning_rate": 4.396464646464647e-05, + "loss": 0.011299673467874527, + "step": 2220 + }, + { + "epoch": 33.71212121212121, + "grad_norm": 0.029233945533633232, + "learning_rate": 4.383838383838384e-05, + "loss": 0.009685883671045304, + "step": 2225 + }, + { + "epoch": 33.78787878787879, + "grad_norm": 0.08113767206668854, + "learning_rate": 4.3712121212121216e-05, + "loss": 0.017820751667022704, + "step": 2230 + }, + { + "epoch": 33.86363636363637, + "grad_norm": 0.043832819908857346, + "learning_rate": 4.358585858585859e-05, + "loss": 0.010486038029193878, + "step": 2235 + }, + { + "epoch": 33.93939393939394, + "grad_norm": 0.08607415854930878, + "learning_rate": 4.345959595959596e-05, + "loss": 0.009716688096523285, + "step": 2240 + }, + { + "epoch": 34.0, + "eval_loss": 0.012541228905320168, + "eval_mean_accuracy": 0.9557628985762149, + "eval_mean_iou": 0.9294742780204088, + "eval_runtime": 114.6337, + "eval_samples_per_second": 2.05, + "eval_steps_per_second": 0.262, + "step": 2244 + }, + { + "epoch": 34.015151515151516, + "grad_norm": 0.06817738711833954, + "learning_rate": 4.3333333333333334e-05, + "loss": 0.011883329600095749, + "step": 2245 + }, + { + "epoch": 34.09090909090909, + "grad_norm": 0.027514943853020668, + "learning_rate": 4.320707070707071e-05, + "loss": 0.010265685617923737, + "step": 2250 + }, + { + "epoch": 34.166666666666664, + "grad_norm": 0.05477522313594818, + "learning_rate": 4.308080808080808e-05, + "loss": 0.015430842339992524, + "step": 2255 + }, + { + "epoch": 34.24242424242424, + "grad_norm": 0.1429450362920761, + "learning_rate": 4.295454545454546e-05, + "loss": 0.012162236869335175, + "step": 2260 + }, + { + "epoch": 34.31818181818182, + "grad_norm": 0.05187097191810608, + "learning_rate": 4.282828282828283e-05, + "loss": 0.01110200136899948, + "step": 2265 + }, + { + "epoch": 34.39393939393939, + "grad_norm": 0.058833882212638855, + "learning_rate": 4.2702020202020204e-05, + "loss": 0.010439897328615189, + "step": 2270 + }, + { + "epoch": 34.46969696969697, + "grad_norm": 0.0466909185051918, + "learning_rate": 4.257575757575758e-05, + "loss": 0.010986138880252839, + "step": 2275 + }, + { + "epoch": 34.54545454545455, + "grad_norm": 0.04007340967655182, + "learning_rate": 4.244949494949495e-05, + "loss": 0.008282290399074554, + "step": 2280 + }, + { + "epoch": 34.621212121212125, + "grad_norm": 0.04508442059159279, + "learning_rate": 4.232323232323233e-05, + "loss": 0.013152590394020081, + "step": 2285 + }, + { + "epoch": 34.696969696969695, + "grad_norm": 0.10995684564113617, + "learning_rate": 4.21969696969697e-05, + "loss": 0.01127975881099701, + "step": 2290 + }, + { + "epoch": 34.77272727272727, + "grad_norm": 0.049980636686086655, + "learning_rate": 4.2070707070707074e-05, + "loss": 0.010247409343719482, + "step": 2295 + }, + { + "epoch": 34.84848484848485, + "grad_norm": 0.08392784744501114, + "learning_rate": 4.194444444444445e-05, + "loss": 0.009239023178815841, + "step": 2300 + }, + { + "epoch": 34.92424242424242, + "grad_norm": 0.06979428231716156, + "learning_rate": 4.181818181818182e-05, + "loss": 0.012678584456443787, + "step": 2305 + }, + { + "epoch": 35.0, + "grad_norm": 0.11974897235631943, + "learning_rate": 4.169191919191919e-05, + "loss": 0.011441192030906678, + "step": 2310 + }, + { + "epoch": 35.0, + "eval_loss": 0.012302711606025696, + "eval_mean_accuracy": 0.9575431248924429, + "eval_mean_iou": 0.929818341629525, + "eval_runtime": 128.0025, + "eval_samples_per_second": 1.836, + "eval_steps_per_second": 0.234, + "step": 2310 + }, + { + "epoch": 35.07575757575758, + "grad_norm": 0.044510453939437866, + "learning_rate": 4.156565656565657e-05, + "loss": 0.011607617139816284, + "step": 2315 + }, + { + "epoch": 35.15151515151515, + "grad_norm": 0.028081513941287994, + "learning_rate": 4.143939393939394e-05, + "loss": 0.009932840615510941, + "step": 2320 + }, + { + "epoch": 35.22727272727273, + "grad_norm": 0.1112414002418518, + "learning_rate": 4.131313131313132e-05, + "loss": 0.01200428456068039, + "step": 2325 + }, + { + "epoch": 35.303030303030305, + "grad_norm": 0.06273142993450165, + "learning_rate": 4.118686868686869e-05, + "loss": 0.009124813973903656, + "step": 2330 + }, + { + "epoch": 35.378787878787875, + "grad_norm": 0.18656454980373383, + "learning_rate": 4.106060606060606e-05, + "loss": 0.01326664537191391, + "step": 2335 + }, + { + "epoch": 35.45454545454545, + "grad_norm": 0.28831732273101807, + "learning_rate": 4.0934343434343435e-05, + "loss": 0.011876031756401062, + "step": 2340 + }, + { + "epoch": 35.53030303030303, + "grad_norm": 0.06218568608164787, + "learning_rate": 4.0808080808080814e-05, + "loss": 0.010482250154018402, + "step": 2345 + }, + { + "epoch": 35.60606060606061, + "grad_norm": 0.15333756804466248, + "learning_rate": 4.068181818181818e-05, + "loss": 0.013770133256912231, + "step": 2350 + }, + { + "epoch": 35.68181818181818, + "grad_norm": 0.06387151032686234, + "learning_rate": 4.055555555555556e-05, + "loss": 0.0125322163105011, + "step": 2355 + }, + { + "epoch": 35.75757575757576, + "grad_norm": 0.08826161175966263, + "learning_rate": 4.042929292929293e-05, + "loss": 0.00941377878189087, + "step": 2360 + }, + { + "epoch": 35.833333333333336, + "grad_norm": 0.06639755517244339, + "learning_rate": 4.0303030303030305e-05, + "loss": 0.011935046315193177, + "step": 2365 + }, + { + "epoch": 35.90909090909091, + "grad_norm": 0.08986376225948334, + "learning_rate": 4.017676767676768e-05, + "loss": 0.012639522552490234, + "step": 2370 + }, + { + "epoch": 35.984848484848484, + "grad_norm": 0.06664793938398361, + "learning_rate": 4.005050505050506e-05, + "loss": 0.009842528402805329, + "step": 2375 + }, + { + "epoch": 36.0, + "eval_loss": 0.013065427541732788, + "eval_mean_accuracy": 0.9690303641731419, + "eval_mean_iou": 0.9267217772759947, + "eval_runtime": 122.2497, + "eval_samples_per_second": 1.922, + "eval_steps_per_second": 0.245, + "step": 2376 + }, + { + "epoch": 36.06060606060606, + "grad_norm": 0.08203178644180298, + "learning_rate": 3.992424242424242e-05, + "loss": 0.010381530970335007, + "step": 2380 + }, + { + "epoch": 36.13636363636363, + "grad_norm": 0.02810048684477806, + "learning_rate": 3.97979797979798e-05, + "loss": 0.009733760356903076, + "step": 2385 + }, + { + "epoch": 36.21212121212121, + "grad_norm": 0.02368428185582161, + "learning_rate": 3.967171717171717e-05, + "loss": 0.013787275552749634, + "step": 2390 + }, + { + "epoch": 36.28787878787879, + "grad_norm": 0.039396047592163086, + "learning_rate": 3.954545454545455e-05, + "loss": 0.007747463881969452, + "step": 2395 + }, + { + "epoch": 36.36363636363637, + "grad_norm": 0.04371464252471924, + "learning_rate": 3.941919191919192e-05, + "loss": 0.007868523895740508, + "step": 2400 + }, + { + "epoch": 36.43939393939394, + "grad_norm": 0.14136487245559692, + "learning_rate": 3.929292929292929e-05, + "loss": 0.014180171489715575, + "step": 2405 + }, + { + "epoch": 36.515151515151516, + "grad_norm": 0.07642136514186859, + "learning_rate": 3.9166666666666665e-05, + "loss": 0.010061027109622955, + "step": 2410 + }, + { + "epoch": 36.59090909090909, + "grad_norm": 0.06523944437503815, + "learning_rate": 3.9040404040404045e-05, + "loss": 0.01457439810037613, + "step": 2415 + }, + { + "epoch": 36.666666666666664, + "grad_norm": 0.04182974249124527, + "learning_rate": 3.891414141414141e-05, + "loss": 0.010128732770681381, + "step": 2420 + }, + { + "epoch": 36.74242424242424, + "grad_norm": 0.0671948567032814, + "learning_rate": 3.878787878787879e-05, + "loss": 0.008621278405189513, + "step": 2425 + }, + { + "epoch": 36.81818181818182, + "grad_norm": 0.041709158569574356, + "learning_rate": 3.866161616161616e-05, + "loss": 0.012394620478153229, + "step": 2430 + }, + { + "epoch": 36.89393939393939, + "grad_norm": 0.09257160872220993, + "learning_rate": 3.8535353535353536e-05, + "loss": 0.014160445332527161, + "step": 2435 + }, + { + "epoch": 36.96969696969697, + "grad_norm": 0.06153567135334015, + "learning_rate": 3.840909090909091e-05, + "loss": 0.010199051350355148, + "step": 2440 + }, + { + "epoch": 37.0, + "eval_loss": 0.01247350126504898, + "eval_mean_accuracy": 0.947838304748147, + "eval_mean_iou": 0.9263655631039313, + "eval_runtime": 113.1721, + "eval_samples_per_second": 2.076, + "eval_steps_per_second": 0.265, + "step": 2442 + }, + { + "epoch": 37.04545454545455, + "grad_norm": 0.04190421476960182, + "learning_rate": 3.828282828282829e-05, + "loss": 0.014294871687889099, + "step": 2445 + }, + { + "epoch": 37.121212121212125, + "grad_norm": 0.030225971713662148, + "learning_rate": 3.815656565656566e-05, + "loss": 0.008857598155736923, + "step": 2450 + }, + { + "epoch": 37.196969696969695, + "grad_norm": 0.0416252426803112, + "learning_rate": 3.803030303030303e-05, + "loss": 0.011946979165077209, + "step": 2455 + }, + { + "epoch": 37.27272727272727, + "grad_norm": 0.07532456517219543, + "learning_rate": 3.7904040404040406e-05, + "loss": 0.012517808377742768, + "step": 2460 + }, + { + "epoch": 37.34848484848485, + "grad_norm": 0.030337825417518616, + "learning_rate": 3.777777777777778e-05, + "loss": 0.010050948709249496, + "step": 2465 + }, + { + "epoch": 37.42424242424242, + "grad_norm": 0.04655507579445839, + "learning_rate": 3.765151515151516e-05, + "loss": 0.013394375145435334, + "step": 2470 + }, + { + "epoch": 37.5, + "grad_norm": 0.04270685464143753, + "learning_rate": 3.7525252525252524e-05, + "loss": 0.011980725079774856, + "step": 2475 + }, + { + "epoch": 37.57575757575758, + "grad_norm": 0.08057626336812973, + "learning_rate": 3.73989898989899e-05, + "loss": 0.011076629161834717, + "step": 2480 + }, + { + "epoch": 37.65151515151515, + "grad_norm": 0.10764868557453156, + "learning_rate": 3.7272727272727276e-05, + "loss": 0.01195506528019905, + "step": 2485 + }, + { + "epoch": 37.72727272727273, + "grad_norm": 0.17322705686092377, + "learning_rate": 3.714646464646465e-05, + "loss": 0.01209639310836792, + "step": 2490 + }, + { + "epoch": 37.803030303030305, + "grad_norm": 0.03481109440326691, + "learning_rate": 3.702020202020202e-05, + "loss": 0.009965900331735611, + "step": 2495 + }, + { + "epoch": 37.878787878787875, + "grad_norm": 0.04805314540863037, + "learning_rate": 3.68939393939394e-05, + "loss": 0.010469646006822587, + "step": 2500 + }, + { + "epoch": 37.95454545454545, + "grad_norm": 0.04709004983305931, + "learning_rate": 3.6767676767676766e-05, + "loss": 0.010794250667095185, + "step": 2505 + }, + { + "epoch": 38.0, + "eval_loss": 0.012067321687936783, + "eval_mean_accuracy": 0.9579479342238059, + "eval_mean_iou": 0.9303814147695261, + "eval_runtime": 123.8862, + "eval_samples_per_second": 1.897, + "eval_steps_per_second": 0.242, + "step": 2508 + }, + { + "epoch": 38.03030303030303, + "grad_norm": 0.053920697420835495, + "learning_rate": 3.6641414141414146e-05, + "loss": 0.015359872579574585, + "step": 2510 + }, + { + "epoch": 38.10606060606061, + "grad_norm": 0.06264694780111313, + "learning_rate": 3.651515151515152e-05, + "loss": 0.008003075420856477, + "step": 2515 + }, + { + "epoch": 38.18181818181818, + "grad_norm": 0.06981729716062546, + "learning_rate": 3.638888888888889e-05, + "loss": 0.009499958157539368, + "step": 2520 + }, + { + "epoch": 38.25757575757576, + "grad_norm": 0.039335332810878754, + "learning_rate": 3.6262626262626264e-05, + "loss": 0.00912391096353531, + "step": 2525 + }, + { + "epoch": 38.333333333333336, + "grad_norm": 0.02505698800086975, + "learning_rate": 3.613636363636364e-05, + "loss": 0.009153645485639572, + "step": 2530 + }, + { + "epoch": 38.40909090909091, + "grad_norm": 0.04599090293049812, + "learning_rate": 3.601010101010101e-05, + "loss": 0.011685807257890701, + "step": 2535 + }, + { + "epoch": 38.484848484848484, + "grad_norm": 0.06485294550657272, + "learning_rate": 3.588383838383839e-05, + "loss": 0.010705886036157608, + "step": 2540 + }, + { + "epoch": 38.56060606060606, + "grad_norm": 0.04006669670343399, + "learning_rate": 3.575757575757576e-05, + "loss": 0.009650326520204543, + "step": 2545 + }, + { + "epoch": 38.63636363636363, + "grad_norm": 0.07800548523664474, + "learning_rate": 3.5631313131313134e-05, + "loss": 0.010487107932567597, + "step": 2550 + }, + { + "epoch": 38.71212121212121, + "grad_norm": 0.06234106048941612, + "learning_rate": 3.5505050505050506e-05, + "loss": 0.012009174376726151, + "step": 2555 + }, + { + "epoch": 38.78787878787879, + "grad_norm": 0.08380723744630814, + "learning_rate": 3.537878787878788e-05, + "loss": 0.013227254152297974, + "step": 2560 + }, + { + "epoch": 38.86363636363637, + "grad_norm": 0.07339825481176376, + "learning_rate": 3.525252525252525e-05, + "loss": 0.011329187452793122, + "step": 2565 + }, + { + "epoch": 38.93939393939394, + "grad_norm": 0.062090445309877396, + "learning_rate": 3.512626262626263e-05, + "loss": 0.01237587109208107, + "step": 2570 + }, + { + "epoch": 39.0, + "eval_loss": 0.011892168782651424, + "eval_mean_accuracy": 0.9633421295457115, + "eval_mean_iou": 0.9317269900389232, + "eval_runtime": 121.2184, + "eval_samples_per_second": 1.939, + "eval_steps_per_second": 0.247, + "step": 2574 + }, + { + "epoch": 39.015151515151516, + "grad_norm": 0.042648062109947205, + "learning_rate": 3.5e-05, + "loss": 0.013401171565055848, + "step": 2575 + }, + { + "epoch": 39.09090909090909, + "grad_norm": 0.07739891111850739, + "learning_rate": 3.4873737373737376e-05, + "loss": 0.00837702676653862, + "step": 2580 + }, + { + "epoch": 39.166666666666664, + "grad_norm": 0.08827576786279678, + "learning_rate": 3.474747474747475e-05, + "loss": 0.016149674355983735, + "step": 2585 + }, + { + "epoch": 39.24242424242424, + "grad_norm": 0.034467313438653946, + "learning_rate": 3.462121212121212e-05, + "loss": 0.01063224971294403, + "step": 2590 + }, + { + "epoch": 39.31818181818182, + "grad_norm": 0.03776657581329346, + "learning_rate": 3.4494949494949494e-05, + "loss": 0.011631693691015244, + "step": 2595 + }, + { + "epoch": 39.39393939393939, + "grad_norm": 0.09946456551551819, + "learning_rate": 3.4368686868686874e-05, + "loss": 0.011299125850200653, + "step": 2600 + }, + { + "epoch": 39.46969696969697, + "grad_norm": 0.084682397544384, + "learning_rate": 3.424242424242424e-05, + "loss": 0.010811488330364227, + "step": 2605 + }, + { + "epoch": 39.54545454545455, + "grad_norm": 0.04472370445728302, + "learning_rate": 3.411616161616162e-05, + "loss": 0.011801199615001678, + "step": 2610 + }, + { + "epoch": 39.621212121212125, + "grad_norm": 0.04462462663650513, + "learning_rate": 3.398989898989899e-05, + "loss": 0.009020185470581055, + "step": 2615 + }, + { + "epoch": 39.696969696969695, + "grad_norm": 0.05104508623480797, + "learning_rate": 3.3863636363636364e-05, + "loss": 0.009547712653875351, + "step": 2620 + }, + { + "epoch": 39.77272727272727, + "grad_norm": 0.0403415709733963, + "learning_rate": 3.373737373737374e-05, + "loss": 0.012834307551383973, + "step": 2625 + }, + { + "epoch": 39.84848484848485, + "grad_norm": 0.041731830686330795, + "learning_rate": 3.3611111111111116e-05, + "loss": 0.010653684288263321, + "step": 2630 + }, + { + "epoch": 39.92424242424242, + "grad_norm": 0.023394132032990456, + "learning_rate": 3.348484848484848e-05, + "loss": 0.009225589036941529, + "step": 2635 + }, + { + "epoch": 40.0, + "grad_norm": 0.12495539337396622, + "learning_rate": 3.335858585858586e-05, + "loss": 0.010562360286712646, + "step": 2640 + }, + { + "epoch": 40.0, + "eval_loss": 0.011702131479978561, + "eval_mean_accuracy": 0.9584469273349585, + "eval_mean_iou": 0.9320973671888121, + "eval_runtime": 118.4202, + "eval_samples_per_second": 1.984, + "eval_steps_per_second": 0.253, + "step": 2640 + }, + { + "epoch": 40.07575757575758, + "grad_norm": 0.025423705577850342, + "learning_rate": 3.3232323232323234e-05, + "loss": 0.009257949143648147, + "step": 2645 + }, + { + "epoch": 40.15151515151515, + "grad_norm": 0.019826961681246758, + "learning_rate": 3.310606060606061e-05, + "loss": 0.008152774721384048, + "step": 2650 + }, + { + "epoch": 40.22727272727273, + "grad_norm": 0.02955153025686741, + "learning_rate": 3.297979797979798e-05, + "loss": 0.00982634350657463, + "step": 2655 + }, + { + "epoch": 40.303030303030305, + "grad_norm": 0.21577662229537964, + "learning_rate": 3.285353535353535e-05, + "loss": 0.009934905916452408, + "step": 2660 + }, + { + "epoch": 40.378787878787875, + "grad_norm": 0.07803802192211151, + "learning_rate": 3.272727272727273e-05, + "loss": 0.0093360535800457, + "step": 2665 + }, + { + "epoch": 40.45454545454545, + "grad_norm": 0.026731858029961586, + "learning_rate": 3.2601010101010104e-05, + "loss": 0.009444894641637802, + "step": 2670 + }, + { + "epoch": 40.53030303030303, + "grad_norm": 0.13624735176563263, + "learning_rate": 3.247474747474748e-05, + "loss": 0.01355384886264801, + "step": 2675 + }, + { + "epoch": 40.60606060606061, + "grad_norm": 0.04551715776324272, + "learning_rate": 3.234848484848485e-05, + "loss": 0.010317550599575042, + "step": 2680 + }, + { + "epoch": 40.68181818181818, + "grad_norm": 0.06463024020195007, + "learning_rate": 3.222222222222223e-05, + "loss": 0.011432089656591416, + "step": 2685 + }, + { + "epoch": 40.75757575757576, + "grad_norm": 0.08428828418254852, + "learning_rate": 3.2095959595959595e-05, + "loss": 0.010098323971033097, + "step": 2690 + }, + { + "epoch": 40.833333333333336, + "grad_norm": 0.03458369895815849, + "learning_rate": 3.1969696969696974e-05, + "loss": 0.01031261757016182, + "step": 2695 + }, + { + "epoch": 40.90909090909091, + "grad_norm": 0.09597554802894592, + "learning_rate": 3.184343434343435e-05, + "loss": 0.014230598509311677, + "step": 2700 + }, + { + "epoch": 40.984848484848484, + "grad_norm": 0.29024210572242737, + "learning_rate": 3.171717171717172e-05, + "loss": 0.015265010297298431, + "step": 2705 + }, + { + "epoch": 41.0, + "eval_loss": 0.011500607244670391, + "eval_mean_accuracy": 0.9583417608921113, + "eval_mean_iou": 0.9318924607365622, + "eval_runtime": 116.7032, + "eval_samples_per_second": 2.014, + "eval_steps_per_second": 0.257, + "step": 2706 + }, + { + "epoch": 41.06060606060606, + "grad_norm": 0.03993074595928192, + "learning_rate": 3.159090909090909e-05, + "loss": 0.009505432844161988, + "step": 2710 + }, + { + "epoch": 41.13636363636363, + "grad_norm": 0.09030765295028687, + "learning_rate": 3.146464646464647e-05, + "loss": 0.011441759020090102, + "step": 2715 + }, + { + "epoch": 41.21212121212121, + "grad_norm": 0.04697788134217262, + "learning_rate": 3.133838383838384e-05, + "loss": 0.010460856556892394, + "step": 2720 + }, + { + "epoch": 41.28787878787879, + "grad_norm": 0.03971889615058899, + "learning_rate": 3.121212121212122e-05, + "loss": 0.012089692801237107, + "step": 2725 + }, + { + "epoch": 41.36363636363637, + "grad_norm": 0.07178938388824463, + "learning_rate": 3.108585858585858e-05, + "loss": 0.012376116216182708, + "step": 2730 + }, + { + "epoch": 41.43939393939394, + "grad_norm": 0.037269022315740585, + "learning_rate": 3.095959595959596e-05, + "loss": 0.00851663276553154, + "step": 2735 + }, + { + "epoch": 41.515151515151516, + "grad_norm": 0.06559725105762482, + "learning_rate": 3.0833333333333335e-05, + "loss": 0.007378444075584412, + "step": 2740 + }, + { + "epoch": 41.59090909090909, + "grad_norm": 0.08047222346067429, + "learning_rate": 3.070707070707071e-05, + "loss": 0.01434701532125473, + "step": 2745 + }, + { + "epoch": 41.666666666666664, + "grad_norm": 0.03132939338684082, + "learning_rate": 3.058080808080808e-05, + "loss": 0.00925743579864502, + "step": 2750 + }, + { + "epoch": 41.74242424242424, + "grad_norm": 0.061435453593730927, + "learning_rate": 3.0454545454545456e-05, + "loss": 0.010742439329624176, + "step": 2755 + }, + { + "epoch": 41.81818181818182, + "grad_norm": 0.07348079234361649, + "learning_rate": 3.032828282828283e-05, + "loss": 0.012243803590536118, + "step": 2760 + }, + { + "epoch": 41.89393939393939, + "grad_norm": 0.029909593984484673, + "learning_rate": 3.0202020202020205e-05, + "loss": 0.010278792679309845, + "step": 2765 + }, + { + "epoch": 41.96969696969697, + "grad_norm": 0.04879944026470184, + "learning_rate": 3.0075757575757578e-05, + "loss": 0.010090172290802002, + "step": 2770 + }, + { + "epoch": 42.0, + "eval_loss": 0.011566005647182465, + "eval_mean_accuracy": 0.9584326847000554, + "eval_mean_iou": 0.9318156802074272, + "eval_runtime": 133.0983, + "eval_samples_per_second": 1.766, + "eval_steps_per_second": 0.225, + "step": 2772 + }, + { + "epoch": 42.04545454545455, + "grad_norm": 0.1064978688955307, + "learning_rate": 2.994949494949495e-05, + "loss": 0.0097226582467556, + "step": 2775 + }, + { + "epoch": 42.121212121212125, + "grad_norm": 0.04498125985264778, + "learning_rate": 2.9823232323232327e-05, + "loss": 0.011855735629796981, + "step": 2780 + }, + { + "epoch": 42.196969696969695, + "grad_norm": 0.04883813112974167, + "learning_rate": 2.96969696969697e-05, + "loss": 0.009904016554355622, + "step": 2785 + }, + { + "epoch": 42.27272727272727, + "grad_norm": 0.046945348381996155, + "learning_rate": 2.9570707070707072e-05, + "loss": 0.010881377756595612, + "step": 2790 + }, + { + "epoch": 42.34848484848485, + "grad_norm": 0.02215217612683773, + "learning_rate": 2.9444444444444448e-05, + "loss": 0.007913407683372498, + "step": 2795 + }, + { + "epoch": 42.42424242424242, + "grad_norm": 0.043503038585186005, + "learning_rate": 2.9318181818181817e-05, + "loss": 0.009736873209476471, + "step": 2800 + }, + { + "epoch": 42.5, + "grad_norm": 0.10303665697574615, + "learning_rate": 2.9191919191919193e-05, + "loss": 0.008227808773517609, + "step": 2805 + }, + { + "epoch": 42.57575757575758, + "grad_norm": 0.05184203386306763, + "learning_rate": 2.906565656565657e-05, + "loss": 0.01101607084274292, + "step": 2810 + }, + { + "epoch": 42.65151515151515, + "grad_norm": 0.04836731776595116, + "learning_rate": 2.893939393939394e-05, + "loss": 0.009844834357500077, + "step": 2815 + }, + { + "epoch": 42.72727272727273, + "grad_norm": 0.0433465912938118, + "learning_rate": 2.8813131313131315e-05, + "loss": 0.01228542923927307, + "step": 2820 + }, + { + "epoch": 42.803030303030305, + "grad_norm": 0.07033395022153854, + "learning_rate": 2.868686868686869e-05, + "loss": 0.010997830331325531, + "step": 2825 + }, + { + "epoch": 42.878787878787875, + "grad_norm": 0.05709235742688179, + "learning_rate": 2.856060606060606e-05, + "loss": 0.010565277189016342, + "step": 2830 + }, + { + "epoch": 42.95454545454545, + "grad_norm": 0.042601678520441055, + "learning_rate": 2.8434343434343436e-05, + "loss": 0.01025083363056183, + "step": 2835 + }, + { + "epoch": 43.0, + "eval_loss": 0.011511821299791336, + "eval_mean_accuracy": 0.9567285056289297, + "eval_mean_iou": 0.9317909272783457, + "eval_runtime": 131.5825, + "eval_samples_per_second": 1.786, + "eval_steps_per_second": 0.228, + "step": 2838 + }, + { + "epoch": 43.03030303030303, + "grad_norm": 0.07863672077655792, + "learning_rate": 2.8308080808080812e-05, + "loss": 0.009466660022735596, + "step": 2840 + }, + { + "epoch": 43.10606060606061, + "grad_norm": 0.0454140268266201, + "learning_rate": 2.818181818181818e-05, + "loss": 0.008517540246248245, + "step": 2845 + }, + { + "epoch": 43.18181818181818, + "grad_norm": 0.06355108320713043, + "learning_rate": 2.8055555555555557e-05, + "loss": 0.010655120760202409, + "step": 2850 + }, + { + "epoch": 43.25757575757576, + "grad_norm": 0.046616166830062866, + "learning_rate": 2.7929292929292933e-05, + "loss": 0.009025464951992034, + "step": 2855 + }, + { + "epoch": 43.333333333333336, + "grad_norm": 0.0362597331404686, + "learning_rate": 2.7803030303030303e-05, + "loss": 0.009227286279201507, + "step": 2860 + }, + { + "epoch": 43.40909090909091, + "grad_norm": 0.03718133270740509, + "learning_rate": 2.767676767676768e-05, + "loss": 0.009892939776182174, + "step": 2865 + }, + { + "epoch": 43.484848484848484, + "grad_norm": 0.02011614851653576, + "learning_rate": 2.7550505050505055e-05, + "loss": 0.01040583997964859, + "step": 2870 + }, + { + "epoch": 43.56060606060606, + "grad_norm": 0.04579155892133713, + "learning_rate": 2.7424242424242424e-05, + "loss": 0.009527173638343812, + "step": 2875 + }, + { + "epoch": 43.63636363636363, + "grad_norm": 0.08099280297756195, + "learning_rate": 2.72979797979798e-05, + "loss": 0.009907497465610504, + "step": 2880 + }, + { + "epoch": 43.71212121212121, + "grad_norm": 0.07010460644960403, + "learning_rate": 2.717171717171717e-05, + "loss": 0.01049983873963356, + "step": 2885 + }, + { + "epoch": 43.78787878787879, + "grad_norm": 0.03897179290652275, + "learning_rate": 2.7045454545454545e-05, + "loss": 0.0107826367020607, + "step": 2890 + }, + { + "epoch": 43.86363636363637, + "grad_norm": 0.05496282875537872, + "learning_rate": 2.691919191919192e-05, + "loss": 0.01101618930697441, + "step": 2895 + }, + { + "epoch": 43.93939393939394, + "grad_norm": 0.04230291023850441, + "learning_rate": 2.679292929292929e-05, + "loss": 0.012428686767816544, + "step": 2900 + }, + { + "epoch": 44.0, + "eval_loss": 0.01140758115798235, + "eval_mean_accuracy": 0.9594083785105904, + "eval_mean_iou": 0.9325577580409594, + "eval_runtime": 131.5354, + "eval_samples_per_second": 1.787, + "eval_steps_per_second": 0.228, + "step": 2904 + }, + { + "epoch": 44.015151515151516, + "grad_norm": 0.05273181572556496, + "learning_rate": 2.6666666666666667e-05, + "loss": 0.009919434785842896, + "step": 2905 + }, + { + "epoch": 44.09090909090909, + "grad_norm": 0.03720299154520035, + "learning_rate": 2.6540404040404043e-05, + "loss": 0.00709642544388771, + "step": 2910 + }, + { + "epoch": 44.166666666666664, + "grad_norm": 0.06484781205654144, + "learning_rate": 2.6414141414141412e-05, + "loss": 0.010354585945606232, + "step": 2915 + }, + { + "epoch": 44.24242424242424, + "grad_norm": 0.046142786741256714, + "learning_rate": 2.6287878787878788e-05, + "loss": 0.008282921463251113, + "step": 2920 + }, + { + "epoch": 44.31818181818182, + "grad_norm": 0.05513034760951996, + "learning_rate": 2.6161616161616164e-05, + "loss": 0.009675325453281402, + "step": 2925 + }, + { + "epoch": 44.39393939393939, + "grad_norm": 0.05165668576955795, + "learning_rate": 2.6035353535353537e-05, + "loss": 0.009244360774755479, + "step": 2930 + }, + { + "epoch": 44.46969696969697, + "grad_norm": 0.055930089205503464, + "learning_rate": 2.590909090909091e-05, + "loss": 0.011424873769283295, + "step": 2935 + }, + { + "epoch": 44.54545454545455, + "grad_norm": 0.04240183159708977, + "learning_rate": 2.5782828282828285e-05, + "loss": 0.012450555711984635, + "step": 2940 + }, + { + "epoch": 44.621212121212125, + "grad_norm": 0.08810874819755554, + "learning_rate": 2.5656565656565658e-05, + "loss": 0.014258894324302673, + "step": 2945 + }, + { + "epoch": 44.696969696969695, + "grad_norm": 0.03436823934316635, + "learning_rate": 2.553030303030303e-05, + "loss": 0.00923032984137535, + "step": 2950 + }, + { + "epoch": 44.77272727272727, + "grad_norm": 0.05433015152812004, + "learning_rate": 2.5404040404040407e-05, + "loss": 0.012869897484779357, + "step": 2955 + }, + { + "epoch": 44.84848484848485, + "grad_norm": 0.03229755163192749, + "learning_rate": 2.527777777777778e-05, + "loss": 0.00944179892539978, + "step": 2960 + }, + { + "epoch": 44.92424242424242, + "grad_norm": 0.036414794623851776, + "learning_rate": 2.5151515151515155e-05, + "loss": 0.010217881202697754, + "step": 2965 + }, + { + "epoch": 45.0, + "grad_norm": 0.07419764995574951, + "learning_rate": 2.5025252525252525e-05, + "loss": 0.008034738153219223, + "step": 2970 + }, + { + "epoch": 45.0, + "eval_loss": 0.011631883680820465, + "eval_mean_accuracy": 0.9626006972211876, + "eval_mean_iou": 0.9331003113400276, + "eval_runtime": 126.0017, + "eval_samples_per_second": 1.865, + "eval_steps_per_second": 0.238, + "step": 2970 + }, + { + "epoch": 45.07575757575758, + "grad_norm": 0.033514603972435, + "learning_rate": 2.48989898989899e-05, + "loss": 0.010970038175582886, + "step": 2975 + }, + { + "epoch": 45.15151515151515, + "grad_norm": 0.02625657618045807, + "learning_rate": 2.4772727272727277e-05, + "loss": 0.007355897873640061, + "step": 2980 + }, + { + "epoch": 45.22727272727273, + "grad_norm": 0.04583245888352394, + "learning_rate": 2.464646464646465e-05, + "loss": 0.011273422837257385, + "step": 2985 + }, + { + "epoch": 45.303030303030305, + "grad_norm": 0.07183682173490524, + "learning_rate": 2.4520202020202022e-05, + "loss": 0.011306219547986985, + "step": 2990 + }, + { + "epoch": 45.378787878787875, + "grad_norm": 0.07157629728317261, + "learning_rate": 2.4393939393939395e-05, + "loss": 0.008732597529888152, + "step": 2995 + }, + { + "epoch": 45.45454545454545, + "grad_norm": 0.039751023054122925, + "learning_rate": 2.426767676767677e-05, + "loss": 0.011157110333442688, + "step": 3000 + }, + { + "epoch": 45.53030303030303, + "grad_norm": 0.044345300644636154, + "learning_rate": 2.4141414141414143e-05, + "loss": 0.01005566343665123, + "step": 3005 + }, + { + "epoch": 45.60606060606061, + "grad_norm": 0.03709288686513901, + "learning_rate": 2.4015151515151516e-05, + "loss": 0.007837090641260147, + "step": 3010 + }, + { + "epoch": 45.68181818181818, + "grad_norm": 0.019442206248641014, + "learning_rate": 2.3888888888888892e-05, + "loss": 0.009214700758457183, + "step": 3015 + }, + { + "epoch": 45.75757575757576, + "grad_norm": 0.06979381293058395, + "learning_rate": 2.3762626262626265e-05, + "loss": 0.01170349344611168, + "step": 3020 + }, + { + "epoch": 45.833333333333336, + "grad_norm": 0.04758612811565399, + "learning_rate": 2.3636363636363637e-05, + "loss": 0.007557286322116852, + "step": 3025 + }, + { + "epoch": 45.90909090909091, + "grad_norm": 0.049681905657052994, + "learning_rate": 2.351010101010101e-05, + "loss": 0.011722264438867569, + "step": 3030 + }, + { + "epoch": 45.984848484848484, + "grad_norm": 0.046274639666080475, + "learning_rate": 2.3383838383838386e-05, + "loss": 0.012813003361225128, + "step": 3035 + }, + { + "epoch": 46.0, + "eval_loss": 0.011177596636116505, + "eval_mean_accuracy": 0.9611352128776586, + "eval_mean_iou": 0.933434953203968, + "eval_runtime": 131.5367, + "eval_samples_per_second": 1.787, + "eval_steps_per_second": 0.228, + "step": 3036 + }, + { + "epoch": 46.06060606060606, + "grad_norm": 0.06310581415891647, + "learning_rate": 2.325757575757576e-05, + "loss": 0.013872361183166504, + "step": 3040 + }, + { + "epoch": 46.13636363636363, + "grad_norm": 0.09346885234117508, + "learning_rate": 2.313131313131313e-05, + "loss": 0.00968407616019249, + "step": 3045 + }, + { + "epoch": 46.21212121212121, + "grad_norm": 0.04300539195537567, + "learning_rate": 2.3005050505050507e-05, + "loss": 0.010871496796607972, + "step": 3050 + }, + { + "epoch": 46.28787878787879, + "grad_norm": 0.03472427278757095, + "learning_rate": 2.287878787878788e-05, + "loss": 0.009670841693878173, + "step": 3055 + }, + { + "epoch": 46.36363636363637, + "grad_norm": 0.025575704872608185, + "learning_rate": 2.2752525252525253e-05, + "loss": 0.009251946955919266, + "step": 3060 + }, + { + "epoch": 46.43939393939394, + "grad_norm": 0.03574152663350105, + "learning_rate": 2.262626262626263e-05, + "loss": 0.00931553691625595, + "step": 3065 + }, + { + "epoch": 46.515151515151516, + "grad_norm": 0.037589121609926224, + "learning_rate": 2.25e-05, + "loss": 0.00977773144841194, + "step": 3070 + }, + { + "epoch": 46.59090909090909, + "grad_norm": 0.0656251311302185, + "learning_rate": 2.2373737373737374e-05, + "loss": 0.010009729862213134, + "step": 3075 + }, + { + "epoch": 46.666666666666664, + "grad_norm": 0.032728422433137894, + "learning_rate": 2.2247474747474747e-05, + "loss": 0.010632945597171784, + "step": 3080 + }, + { + "epoch": 46.74242424242424, + "grad_norm": 0.05859195813536644, + "learning_rate": 2.2121212121212123e-05, + "loss": 0.007086991518735886, + "step": 3085 + }, + { + "epoch": 46.81818181818182, + "grad_norm": 0.03191753104329109, + "learning_rate": 2.1994949494949495e-05, + "loss": 0.008117115497589112, + "step": 3090 + }, + { + "epoch": 46.89393939393939, + "grad_norm": 0.08097993582487106, + "learning_rate": 2.1868686868686868e-05, + "loss": 0.010112101584672928, + "step": 3095 + }, + { + "epoch": 46.96969696969697, + "grad_norm": 0.06209854781627655, + "learning_rate": 2.1742424242424244e-05, + "loss": 0.013057228922843934, + "step": 3100 + }, + { + "epoch": 47.0, + "eval_loss": 0.011305268853902817, + "eval_mean_accuracy": 0.9640215164036744, + "eval_mean_iou": 0.9334811346975173, + "eval_runtime": 128.4156, + "eval_samples_per_second": 1.83, + "eval_steps_per_second": 0.234, + "step": 3102 + }, + { + "epoch": 47.04545454545455, + "grad_norm": 0.10861490666866302, + "learning_rate": 2.1616161616161617e-05, + "loss": 0.01261637955904007, + "step": 3105 + }, + { + "epoch": 47.121212121212125, + "grad_norm": 0.09244271367788315, + "learning_rate": 2.148989898989899e-05, + "loss": 0.009592726081609725, + "step": 3110 + }, + { + "epoch": 47.196969696969695, + "grad_norm": 0.04153317213058472, + "learning_rate": 2.1363636363636362e-05, + "loss": 0.009246806055307389, + "step": 3115 + }, + { + "epoch": 47.27272727272727, + "grad_norm": 0.04215581715106964, + "learning_rate": 2.1237373737373738e-05, + "loss": 0.009788258373737336, + "step": 3120 + }, + { + "epoch": 47.34848484848485, + "grad_norm": 0.07929627597332001, + "learning_rate": 2.111111111111111e-05, + "loss": 0.008763598650693894, + "step": 3125 + }, + { + "epoch": 47.42424242424242, + "grad_norm": 0.025632839649915695, + "learning_rate": 2.0984848484848483e-05, + "loss": 0.0119766466319561, + "step": 3130 + }, + { + "epoch": 47.5, + "grad_norm": 0.1177750676870346, + "learning_rate": 2.085858585858586e-05, + "loss": 0.011509133875370026, + "step": 3135 + }, + { + "epoch": 47.57575757575758, + "grad_norm": 0.029964802786707878, + "learning_rate": 2.0732323232323232e-05, + "loss": 0.007907600700855255, + "step": 3140 + }, + { + "epoch": 47.65151515151515, + "grad_norm": 0.02477479726076126, + "learning_rate": 2.0606060606060608e-05, + "loss": 0.010433636605739594, + "step": 3145 + }, + { + "epoch": 47.72727272727273, + "grad_norm": 0.05213652923703194, + "learning_rate": 2.047979797979798e-05, + "loss": 0.009080395102500916, + "step": 3150 + }, + { + "epoch": 47.803030303030305, + "grad_norm": 0.05658382922410965, + "learning_rate": 2.0353535353535353e-05, + "loss": 0.011339858174324036, + "step": 3155 + }, + { + "epoch": 47.878787878787875, + "grad_norm": 0.026541512459516525, + "learning_rate": 2.022727272727273e-05, + "loss": 0.008360118418931962, + "step": 3160 + }, + { + "epoch": 47.95454545454545, + "grad_norm": 0.06262459605932236, + "learning_rate": 2.0101010101010102e-05, + "loss": 0.011185683310031891, + "step": 3165 + }, + { + "epoch": 48.0, + "eval_loss": 0.01118179876357317, + "eval_mean_accuracy": 0.9647069830443908, + "eval_mean_iou": 0.9341387007791688, + "eval_runtime": 122.5535, + "eval_samples_per_second": 1.918, + "eval_steps_per_second": 0.245, + "step": 3168 + }, + { + "epoch": 48.03030303030303, + "grad_norm": 0.07240039855241776, + "learning_rate": 1.9974747474747478e-05, + "loss": 0.01001751869916916, + "step": 3170 + }, + { + "epoch": 48.10606060606061, + "grad_norm": 0.021767230704426765, + "learning_rate": 1.984848484848485e-05, + "loss": 0.009527213126420974, + "step": 3175 + }, + { + "epoch": 48.18181818181818, + "grad_norm": 0.058266088366508484, + "learning_rate": 1.9722222222222224e-05, + "loss": 0.008091110736131668, + "step": 3180 + }, + { + "epoch": 48.25757575757576, + "grad_norm": 0.10201539844274521, + "learning_rate": 1.95959595959596e-05, + "loss": 0.01108373999595642, + "step": 3185 + }, + { + "epoch": 48.333333333333336, + "grad_norm": 0.11515853554010391, + "learning_rate": 1.9469696969696972e-05, + "loss": 0.009322765469551086, + "step": 3190 + }, + { + "epoch": 48.40909090909091, + "grad_norm": 0.0666147843003273, + "learning_rate": 1.9343434343434345e-05, + "loss": 0.01010623425245285, + "step": 3195 + }, + { + "epoch": 48.484848484848484, + "grad_norm": 0.05063946545124054, + "learning_rate": 1.9217171717171718e-05, + "loss": 0.007946109026670456, + "step": 3200 + }, + { + "epoch": 48.56060606060606, + "grad_norm": 0.018964502960443497, + "learning_rate": 1.9090909090909094e-05, + "loss": 0.009537026286125183, + "step": 3205 + }, + { + "epoch": 48.63636363636363, + "grad_norm": 0.048037681728601456, + "learning_rate": 1.8964646464646466e-05, + "loss": 0.013727232813835144, + "step": 3210 + }, + { + "epoch": 48.71212121212121, + "grad_norm": 0.03602692857384682, + "learning_rate": 1.883838383838384e-05, + "loss": 0.012323687970638274, + "step": 3215 + }, + { + "epoch": 48.78787878787879, + "grad_norm": 0.037855323404073715, + "learning_rate": 1.8712121212121215e-05, + "loss": 0.009915858507156372, + "step": 3220 + }, + { + "epoch": 48.86363636363637, + "grad_norm": 0.0638260692358017, + "learning_rate": 1.8585858585858588e-05, + "loss": 0.008882324397563934, + "step": 3225 + }, + { + "epoch": 48.93939393939394, + "grad_norm": 0.02506132610142231, + "learning_rate": 1.845959595959596e-05, + "loss": 0.007825832068920135, + "step": 3230 + }, + { + "epoch": 49.0, + "eval_loss": 0.011304161511361599, + "eval_mean_accuracy": 0.9649350277709923, + "eval_mean_iou": 0.9342392577832371, + "eval_runtime": 96.2506, + "eval_samples_per_second": 2.442, + "eval_steps_per_second": 0.312, + "step": 3234 + }, + { + "epoch": 49.015151515151516, + "grad_norm": 0.07730422914028168, + "learning_rate": 1.8333333333333333e-05, + "loss": 0.010349252820014953, + "step": 3235 + }, + { + "epoch": 49.09090909090909, + "grad_norm": 0.046799108386039734, + "learning_rate": 1.820707070707071e-05, + "loss": 0.009587443619966506, + "step": 3240 + }, + { + "epoch": 49.166666666666664, + "grad_norm": 0.05787617340683937, + "learning_rate": 1.808080808080808e-05, + "loss": 0.007644562423229218, + "step": 3245 + }, + { + "epoch": 49.24242424242424, + "grad_norm": 0.034447263926267624, + "learning_rate": 1.7954545454545454e-05, + "loss": 0.010553071647882462, + "step": 3250 + }, + { + "epoch": 49.31818181818182, + "grad_norm": 0.03742067143321037, + "learning_rate": 1.782828282828283e-05, + "loss": 0.01145610138773918, + "step": 3255 + }, + { + "epoch": 49.39393939393939, + "grad_norm": 0.10363037884235382, + "learning_rate": 1.7702020202020203e-05, + "loss": 0.009171058237552644, + "step": 3260 + }, + { + "epoch": 49.46969696969697, + "grad_norm": 0.03897767513990402, + "learning_rate": 1.7575757575757576e-05, + "loss": 0.01240767389535904, + "step": 3265 + }, + { + "epoch": 49.54545454545455, + "grad_norm": 0.04528479278087616, + "learning_rate": 1.744949494949495e-05, + "loss": 0.007641658931970596, + "step": 3270 + }, + { + "epoch": 49.621212121212125, + "grad_norm": 0.03156213089823723, + "learning_rate": 1.7323232323232324e-05, + "loss": 0.010232439637184143, + "step": 3275 + }, + { + "epoch": 49.696969696969695, + "grad_norm": 0.04798737168312073, + "learning_rate": 1.7196969696969697e-05, + "loss": 0.00997835174202919, + "step": 3280 + }, + { + "epoch": 49.77272727272727, + "grad_norm": 0.07831096649169922, + "learning_rate": 1.707070707070707e-05, + "loss": 0.008293437957763671, + "step": 3285 + }, + { + "epoch": 49.84848484848485, + "grad_norm": 0.06381958723068237, + "learning_rate": 1.6944444444444446e-05, + "loss": 0.009885338693857193, + "step": 3290 + }, + { + "epoch": 49.92424242424242, + "grad_norm": 0.05880928784608841, + "learning_rate": 1.6818181818181818e-05, + "loss": 0.009186813980340958, + "step": 3295 + }, + { + "epoch": 50.0, + "grad_norm": 0.15343432128429413, + "learning_rate": 1.669191919191919e-05, + "loss": 0.010943562537431718, + "step": 3300 + }, + { + "epoch": 50.0, + "eval_loss": 0.011101212352514267, + "eval_mean_accuracy": 0.9609297183190703, + "eval_mean_iou": 0.9340968316965657, + "eval_runtime": 101.2008, + "eval_samples_per_second": 2.322, + "eval_steps_per_second": 0.296, + "step": 3300 + }, + { + "epoch": 50.07575757575758, + "grad_norm": 0.07198474556207657, + "learning_rate": 1.6565656565656567e-05, + "loss": 0.008009252697229385, + "step": 3305 + }, + { + "epoch": 50.15151515151515, + "grad_norm": 0.04599815607070923, + "learning_rate": 1.643939393939394e-05, + "loss": 0.009303316473960876, + "step": 3310 + }, + { + "epoch": 50.22727272727273, + "grad_norm": 0.03742392361164093, + "learning_rate": 1.6313131313131312e-05, + "loss": 0.008956626802682877, + "step": 3315 + }, + { + "epoch": 50.303030303030305, + "grad_norm": 0.027768045663833618, + "learning_rate": 1.6186868686868685e-05, + "loss": 0.009809549152851104, + "step": 3320 + }, + { + "epoch": 50.378787878787875, + "grad_norm": 0.07138533145189285, + "learning_rate": 1.606060606060606e-05, + "loss": 0.009467899054288863, + "step": 3325 + }, + { + "epoch": 50.45454545454545, + "grad_norm": 0.034720975905656815, + "learning_rate": 1.5934343434343434e-05, + "loss": 0.00765787735581398, + "step": 3330 + }, + { + "epoch": 50.53030303030303, + "grad_norm": 0.0255584754049778, + "learning_rate": 1.5808080808080806e-05, + "loss": 0.007916782051324844, + "step": 3335 + }, + { + "epoch": 50.60606060606061, + "grad_norm": 0.020745839923620224, + "learning_rate": 1.5681818181818182e-05, + "loss": 0.014176034927368164, + "step": 3340 + }, + { + "epoch": 50.68181818181818, + "grad_norm": 0.020964210852980614, + "learning_rate": 1.5555555555555555e-05, + "loss": 0.008006905764341354, + "step": 3345 + }, + { + "epoch": 50.75757575757576, + "grad_norm": 0.04937268793582916, + "learning_rate": 1.542929292929293e-05, + "loss": 0.010936376452445985, + "step": 3350 + }, + { + "epoch": 50.833333333333336, + "grad_norm": 0.04155882075428963, + "learning_rate": 1.5303030303030304e-05, + "loss": 0.01320691704750061, + "step": 3355 + }, + { + "epoch": 50.90909090909091, + "grad_norm": 0.056736163794994354, + "learning_rate": 1.5176767676767678e-05, + "loss": 0.010511702299118042, + "step": 3360 + }, + { + "epoch": 50.984848484848484, + "grad_norm": 0.06294524669647217, + "learning_rate": 1.505050505050505e-05, + "loss": 0.010828429460525512, + "step": 3365 + }, + { + "epoch": 51.0, + "eval_loss": 0.011181595735251904, + "eval_mean_accuracy": 0.9583220038159196, + "eval_mean_iou": 0.9330225501252771, + "eval_runtime": 109.5069, + "eval_samples_per_second": 2.146, + "eval_steps_per_second": 0.274, + "step": 3366 + }, + { + "epoch": 51.06060606060606, + "grad_norm": 0.04714588448405266, + "learning_rate": 1.4924242424242423e-05, + "loss": 0.012847204506397248, + "step": 3370 + }, + { + "epoch": 51.13636363636363, + "grad_norm": 0.05521787703037262, + "learning_rate": 1.47979797979798e-05, + "loss": 0.011039341986179351, + "step": 3375 + }, + { + "epoch": 51.21212121212121, + "grad_norm": 0.026737432926893234, + "learning_rate": 1.4671717171717172e-05, + "loss": 0.008258116245269776, + "step": 3380 + }, + { + "epoch": 51.28787878787879, + "grad_norm": 0.03438520058989525, + "learning_rate": 1.4545454545454545e-05, + "loss": 0.007780172675848007, + "step": 3385 + }, + { + "epoch": 51.36363636363637, + "grad_norm": 0.04365864768624306, + "learning_rate": 1.441919191919192e-05, + "loss": 0.01288648098707199, + "step": 3390 + }, + { + "epoch": 51.43939393939394, + "grad_norm": 0.022486280649900436, + "learning_rate": 1.4292929292929293e-05, + "loss": 0.007616708427667618, + "step": 3395 + }, + { + "epoch": 51.515151515151516, + "grad_norm": 0.03314690664410591, + "learning_rate": 1.4166666666666668e-05, + "loss": 0.00829450711607933, + "step": 3400 + }, + { + "epoch": 51.59090909090909, + "grad_norm": 0.05905763804912567, + "learning_rate": 1.404040404040404e-05, + "loss": 0.011117340624332428, + "step": 3405 + }, + { + "epoch": 51.666666666666664, + "grad_norm": 0.06490912288427353, + "learning_rate": 1.3914141414141416e-05, + "loss": 0.010782060027122498, + "step": 3410 + }, + { + "epoch": 51.74242424242424, + "grad_norm": 0.045841652899980545, + "learning_rate": 1.3787878787878789e-05, + "loss": 0.012234783172607422, + "step": 3415 + }, + { + "epoch": 51.81818181818182, + "grad_norm": 0.03178030997514725, + "learning_rate": 1.3661616161616162e-05, + "loss": 0.007770595699548721, + "step": 3420 + }, + { + "epoch": 51.89393939393939, + "grad_norm": 0.03490876033902168, + "learning_rate": 1.3535353535353538e-05, + "loss": 0.008505021035671235, + "step": 3425 + }, + { + "epoch": 51.96969696969697, + "grad_norm": 0.030197303742170334, + "learning_rate": 1.340909090909091e-05, + "loss": 0.008493304252624512, + "step": 3430 + }, + { + "epoch": 52.0, + "eval_loss": 0.011221250519156456, + "eval_mean_accuracy": 0.9615658678694942, + "eval_mean_iou": 0.9336675897619844, + "eval_runtime": 127.6478, + "eval_samples_per_second": 1.841, + "eval_steps_per_second": 0.235, + "step": 3432 + }, + { + "epoch": 52.04545454545455, + "grad_norm": 0.031460296362638474, + "learning_rate": 1.3282828282828283e-05, + "loss": 0.007830271124839782, + "step": 3435 + }, + { + "epoch": 52.121212121212125, + "grad_norm": 0.06011766940355301, + "learning_rate": 1.3156565656565656e-05, + "loss": 0.011578293144702911, + "step": 3440 + }, + { + "epoch": 52.196969696969695, + "grad_norm": 0.027353808283805847, + "learning_rate": 1.3030303030303032e-05, + "loss": 0.011657704412937165, + "step": 3445 + }, + { + "epoch": 52.27272727272727, + "grad_norm": 0.06623313575983047, + "learning_rate": 1.2904040404040404e-05, + "loss": 0.00961935669183731, + "step": 3450 + }, + { + "epoch": 52.34848484848485, + "grad_norm": 0.03869352489709854, + "learning_rate": 1.2777777777777777e-05, + "loss": 0.009706974774599076, + "step": 3455 + }, + { + "epoch": 52.42424242424242, + "grad_norm": 0.03232342004776001, + "learning_rate": 1.2651515151515153e-05, + "loss": 0.0071423880755901335, + "step": 3460 + }, + { + "epoch": 52.5, + "grad_norm": 0.04407987743616104, + "learning_rate": 1.2525252525252526e-05, + "loss": 0.010362230986356736, + "step": 3465 + }, + { + "epoch": 52.57575757575758, + "grad_norm": 0.027869466692209244, + "learning_rate": 1.2398989898989898e-05, + "loss": 0.010962443798780442, + "step": 3470 + }, + { + "epoch": 52.65151515151515, + "grad_norm": 0.040494274348020554, + "learning_rate": 1.2272727272727273e-05, + "loss": 0.009437073022127151, + "step": 3475 + }, + { + "epoch": 52.72727272727273, + "grad_norm": 0.020000159740447998, + "learning_rate": 1.2146464646464647e-05, + "loss": 0.00779109001159668, + "step": 3480 + }, + { + "epoch": 52.803030303030305, + "grad_norm": 0.024865848943591118, + "learning_rate": 1.202020202020202e-05, + "loss": 0.010508288443088532, + "step": 3485 + }, + { + "epoch": 52.878787878787875, + "grad_norm": 0.03874952346086502, + "learning_rate": 1.1893939393939394e-05, + "loss": 0.011132716387510299, + "step": 3490 + }, + { + "epoch": 52.95454545454545, + "grad_norm": 0.06774380803108215, + "learning_rate": 1.1767676767676768e-05, + "loss": 0.008581660687923431, + "step": 3495 + }, + { + "epoch": 53.0, + "eval_loss": 0.011078035458922386, + "eval_mean_accuracy": 0.9639549892994278, + "eval_mean_iou": 0.9348918179919883, + "eval_runtime": 128.7842, + "eval_samples_per_second": 1.825, + "eval_steps_per_second": 0.233, + "step": 3498 + }, + { + "epoch": 53.03030303030303, + "grad_norm": 0.03777788579463959, + "learning_rate": 1.1641414141414143e-05, + "loss": 0.007856415212154388, + "step": 3500 + }, + { + "epoch": 53.10606060606061, + "grad_norm": 0.0253172405064106, + "learning_rate": 1.1515151515151517e-05, + "loss": 0.007626376301050186, + "step": 3505 + }, + { + "epoch": 53.18181818181818, + "grad_norm": 0.11541735380887985, + "learning_rate": 1.138888888888889e-05, + "loss": 0.011122234165668488, + "step": 3510 + }, + { + "epoch": 53.25757575757576, + "grad_norm": 0.019282640889286995, + "learning_rate": 1.1262626262626264e-05, + "loss": 0.007058703154325485, + "step": 3515 + }, + { + "epoch": 53.333333333333336, + "grad_norm": 0.04660727456212044, + "learning_rate": 1.1136363636363637e-05, + "loss": 0.01152946799993515, + "step": 3520 + }, + { + "epoch": 53.40909090909091, + "grad_norm": 0.02544434368610382, + "learning_rate": 1.1010101010101011e-05, + "loss": 0.008742015063762664, + "step": 3525 + }, + { + "epoch": 53.484848484848484, + "grad_norm": 0.04845629632472992, + "learning_rate": 1.0883838383838384e-05, + "loss": 0.009559187293052673, + "step": 3530 + }, + { + "epoch": 53.56060606060606, + "grad_norm": 0.10683389753103256, + "learning_rate": 1.0757575757575758e-05, + "loss": 0.012566661834716797, + "step": 3535 + }, + { + "epoch": 53.63636363636363, + "grad_norm": 0.0668366402387619, + "learning_rate": 1.0631313131313132e-05, + "loss": 0.009748588502407073, + "step": 3540 + }, + { + "epoch": 53.71212121212121, + "grad_norm": 0.049143511801958084, + "learning_rate": 1.0505050505050505e-05, + "loss": 0.010839685797691345, + "step": 3545 + }, + { + "epoch": 53.78787878787879, + "grad_norm": 0.051120057702064514, + "learning_rate": 1.037878787878788e-05, + "loss": 0.010533088445663452, + "step": 3550 + }, + { + "epoch": 53.86363636363637, + "grad_norm": 0.07547329366207123, + "learning_rate": 1.0252525252525252e-05, + "loss": 0.009898938238620758, + "step": 3555 + }, + { + "epoch": 53.93939393939394, + "grad_norm": 0.03128774091601372, + "learning_rate": 1.0126262626262626e-05, + "loss": 0.006742162257432937, + "step": 3560 + }, + { + "epoch": 54.0, + "eval_loss": 0.011005544103682041, + "eval_mean_accuracy": 0.9580727530364511, + "eval_mean_iou": 0.9336404658202794, + "eval_runtime": 129.2404, + "eval_samples_per_second": 1.818, + "eval_steps_per_second": 0.232, + "step": 3564 + }, + { + "epoch": 54.015151515151516, + "grad_norm": 0.03134159371256828, + "learning_rate": 1e-05, + "loss": 0.009006843715906144, + "step": 3565 + }, + { + "epoch": 54.09090909090909, + "grad_norm": 0.04034997522830963, + "learning_rate": 9.873737373737373e-06, + "loss": 0.007617304474115372, + "step": 3570 + }, + { + "epoch": 54.166666666666664, + "grad_norm": 0.038605932146310806, + "learning_rate": 9.747474747474748e-06, + "loss": 0.008944347500801086, + "step": 3575 + }, + { + "epoch": 54.24242424242424, + "grad_norm": 0.024546312168240547, + "learning_rate": 9.62121212121212e-06, + "loss": 0.01022036448121071, + "step": 3580 + }, + { + "epoch": 54.31818181818182, + "grad_norm": 0.037967827171087265, + "learning_rate": 9.494949494949495e-06, + "loss": 0.009848921746015548, + "step": 3585 + }, + { + "epoch": 54.39393939393939, + "grad_norm": 0.034657567739486694, + "learning_rate": 9.36868686868687e-06, + "loss": 0.008719929307699204, + "step": 3590 + }, + { + "epoch": 54.46969696969697, + "grad_norm": 0.05650673434138298, + "learning_rate": 9.242424242424244e-06, + "loss": 0.01114274114370346, + "step": 3595 + }, + { + "epoch": 54.54545454545455, + "grad_norm": 0.07947558909654617, + "learning_rate": 9.116161616161616e-06, + "loss": 0.011273042112588883, + "step": 3600 + }, + { + "epoch": 54.621212121212125, + "grad_norm": 0.07150841504335403, + "learning_rate": 8.98989898989899e-06, + "loss": 0.009610799700021743, + "step": 3605 + }, + { + "epoch": 54.696969696969695, + "grad_norm": 0.02577025629580021, + "learning_rate": 8.863636363636365e-06, + "loss": 0.011200150102376938, + "step": 3610 + }, + { + "epoch": 54.77272727272727, + "grad_norm": 0.03333451598882675, + "learning_rate": 8.737373737373738e-06, + "loss": 0.00876312255859375, + "step": 3615 + }, + { + "epoch": 54.84848484848485, + "grad_norm": 0.062200237065553665, + "learning_rate": 8.611111111111112e-06, + "loss": 0.00956970900297165, + "step": 3620 + }, + { + "epoch": 54.92424242424242, + "grad_norm": 0.020428841933608055, + "learning_rate": 8.484848484848486e-06, + "loss": 0.0077986598014831545, + "step": 3625 + }, + { + "epoch": 55.0, + "grad_norm": 0.022901061922311783, + "learning_rate": 8.358585858585859e-06, + "loss": 0.008456150442361832, + "step": 3630 + }, + { + "epoch": 55.0, + "eval_loss": 0.011020266450941563, + "eval_mean_accuracy": 0.964376678506385, + "eval_mean_iou": 0.9352202952510814, + "eval_runtime": 128.0314, + "eval_samples_per_second": 1.835, + "eval_steps_per_second": 0.234, + "step": 3630 + }, + { + "epoch": 55.07575757575758, + "grad_norm": 0.030423011630773544, + "learning_rate": 8.232323232323233e-06, + "loss": 0.010200446099042892, + "step": 3635 + }, + { + "epoch": 55.15151515151515, + "grad_norm": 0.03606997802853584, + "learning_rate": 8.106060606060606e-06, + "loss": 0.011270485073328017, + "step": 3640 + }, + { + "epoch": 55.22727272727273, + "grad_norm": 0.04661742225289345, + "learning_rate": 7.97979797979798e-06, + "loss": 0.009173568338155746, + "step": 3645 + }, + { + "epoch": 55.303030303030305, + "grad_norm": 0.04383421316742897, + "learning_rate": 7.853535353535355e-06, + "loss": 0.007146291434764862, + "step": 3650 + }, + { + "epoch": 55.378787878787875, + "grad_norm": 0.026614926755428314, + "learning_rate": 7.727272727272727e-06, + "loss": 0.009720848500728607, + "step": 3655 + }, + { + "epoch": 55.45454545454545, + "grad_norm": 0.08967319130897522, + "learning_rate": 7.6010101010101016e-06, + "loss": 0.009636265784502029, + "step": 3660 + }, + { + "epoch": 55.53030303030303, + "grad_norm": 0.0401042178273201, + "learning_rate": 7.474747474747475e-06, + "loss": 0.009760026633739472, + "step": 3665 + }, + { + "epoch": 55.60606060606061, + "grad_norm": 0.045736007392406464, + "learning_rate": 7.3484848484848486e-06, + "loss": 0.010854852199554444, + "step": 3670 + }, + { + "epoch": 55.68181818181818, + "grad_norm": 0.033871814608573914, + "learning_rate": 7.222222222222222e-06, + "loss": 0.01032974049448967, + "step": 3675 + }, + { + "epoch": 55.75757575757576, + "grad_norm": 0.04295116290450096, + "learning_rate": 7.095959595959596e-06, + "loss": 0.008035247027873994, + "step": 3680 + }, + { + "epoch": 55.833333333333336, + "grad_norm": 0.045053619891405106, + "learning_rate": 6.969696969696971e-06, + "loss": 0.008929308503866196, + "step": 3685 + }, + { + "epoch": 55.90909090909091, + "grad_norm": 0.04975677281618118, + "learning_rate": 6.843434343434343e-06, + "loss": 0.006927791982889175, + "step": 3690 + }, + { + "epoch": 55.984848484848484, + "grad_norm": 0.059825796633958817, + "learning_rate": 6.717171717171718e-06, + "loss": 0.009465140849351883, + "step": 3695 + }, + { + "epoch": 56.0, + "eval_loss": 0.010990266688168049, + "eval_mean_accuracy": 0.9629591678427577, + "eval_mean_iou": 0.9352932605927924, + "eval_runtime": 129.1893, + "eval_samples_per_second": 1.819, + "eval_steps_per_second": 0.232, + "step": 3696 + }, + { + "epoch": 56.06060606060606, + "grad_norm": 0.05501623824238777, + "learning_rate": 6.59090909090909e-06, + "loss": 0.010021034628152847, + "step": 3700 + }, + { + "epoch": 56.13636363636363, + "grad_norm": 0.038092657923698425, + "learning_rate": 6.464646464646465e-06, + "loss": 0.008492992073297501, + "step": 3705 + }, + { + "epoch": 56.21212121212121, + "grad_norm": 0.04801051691174507, + "learning_rate": 6.338383838383839e-06, + "loss": 0.007577387243509292, + "step": 3710 + }, + { + "epoch": 56.28787878787879, + "grad_norm": 0.04741519317030907, + "learning_rate": 6.212121212121212e-06, + "loss": 0.010240931808948518, + "step": 3715 + }, + { + "epoch": 56.36363636363637, + "grad_norm": 0.03398001194000244, + "learning_rate": 6.085858585858586e-06, + "loss": 0.010102768242359162, + "step": 3720 + }, + { + "epoch": 56.43939393939394, + "grad_norm": 0.031030435115098953, + "learning_rate": 5.9595959595959605e-06, + "loss": 0.008902082592248917, + "step": 3725 + }, + { + "epoch": 56.515151515151516, + "grad_norm": 0.036280982196331024, + "learning_rate": 5.833333333333334e-06, + "loss": 0.00760902538895607, + "step": 3730 + }, + { + "epoch": 56.59090909090909, + "grad_norm": 0.057811904698610306, + "learning_rate": 5.7070707070707075e-06, + "loss": 0.010345153510570526, + "step": 3735 + }, + { + "epoch": 56.666666666666664, + "grad_norm": 0.03845837712287903, + "learning_rate": 5.580808080808081e-06, + "loss": 0.010506625473499297, + "step": 3740 + }, + { + "epoch": 56.74242424242424, + "grad_norm": 0.10140617191791534, + "learning_rate": 5.4545454545454545e-06, + "loss": 0.012108326703310014, + "step": 3745 + }, + { + "epoch": 56.81818181818182, + "grad_norm": 0.048700787127017975, + "learning_rate": 5.328282828282829e-06, + "loss": 0.009876859933137893, + "step": 3750 + }, + { + "epoch": 56.89393939393939, + "grad_norm": 0.02889069728553295, + "learning_rate": 5.202020202020202e-06, + "loss": 0.007143153995275498, + "step": 3755 + }, + { + "epoch": 56.96969696969697, + "grad_norm": 0.056097764521837234, + "learning_rate": 5.075757575757576e-06, + "loss": 0.009521093964576722, + "step": 3760 + }, + { + "epoch": 57.0, + "eval_loss": 0.010925506241619587, + "eval_mean_accuracy": 0.9631511953362232, + "eval_mean_iou": 0.9353266318454168, + "eval_runtime": 124.125, + "eval_samples_per_second": 1.893, + "eval_steps_per_second": 0.242, + "step": 3762 + }, + { + "epoch": 57.04545454545455, + "grad_norm": 0.11279812455177307, + "learning_rate": 4.949494949494949e-06, + "loss": 0.009715686738491058, + "step": 3765 + }, + { + "epoch": 57.121212121212125, + "grad_norm": 0.03260188549757004, + "learning_rate": 4.823232323232324e-06, + "loss": 0.007641904056072235, + "step": 3770 + }, + { + "epoch": 57.196969696969695, + "grad_norm": 0.02904953621327877, + "learning_rate": 4.696969696969697e-06, + "loss": 0.009131749719381332, + "step": 3775 + }, + { + "epoch": 57.27272727272727, + "grad_norm": 0.03240164741873741, + "learning_rate": 4.5707070707070715e-06, + "loss": 0.008857764303684235, + "step": 3780 + }, + { + "epoch": 57.34848484848485, + "grad_norm": 0.04988917335867882, + "learning_rate": 4.444444444444445e-06, + "loss": 0.01045645698904991, + "step": 3785 + }, + { + "epoch": 57.42424242424242, + "grad_norm": 0.0701695904135704, + "learning_rate": 4.3181818181818185e-06, + "loss": 0.008719290792942046, + "step": 3790 + }, + { + "epoch": 57.5, + "grad_norm": 0.050978612154722214, + "learning_rate": 4.191919191919192e-06, + "loss": 0.009997940063476563, + "step": 3795 + }, + { + "epoch": 57.57575757575758, + "grad_norm": 0.043475229293107986, + "learning_rate": 4.0656565656565655e-06, + "loss": 0.009253481775522232, + "step": 3800 + }, + { + "epoch": 57.65151515151515, + "grad_norm": 0.029040975496172905, + "learning_rate": 3.939393939393939e-06, + "loss": 0.007665455341339111, + "step": 3805 + }, + { + "epoch": 57.72727272727273, + "grad_norm": 0.06640396267175674, + "learning_rate": 3.8131313131313138e-06, + "loss": 0.010579461604356766, + "step": 3810 + }, + { + "epoch": 57.803030303030305, + "grad_norm": 0.08239873498678207, + "learning_rate": 3.6868686868686873e-06, + "loss": 0.010458643734455108, + "step": 3815 + }, + { + "epoch": 57.878787878787875, + "grad_norm": 0.10676904767751694, + "learning_rate": 3.5606060606060608e-06, + "loss": 0.0094834104180336, + "step": 3820 + }, + { + "epoch": 57.95454545454545, + "grad_norm": 0.03905234485864639, + "learning_rate": 3.4343434343434343e-06, + "loss": 0.008727557957172394, + "step": 3825 + }, + { + "epoch": 58.0, + "eval_loss": 0.01097872480750084, + "eval_mean_accuracy": 0.9627232464561621, + "eval_mean_iou": 0.9353214655584683, + "eval_runtime": 131.6127, + "eval_samples_per_second": 1.786, + "eval_steps_per_second": 0.228, + "step": 3828 + }, + { + "epoch": 58.03030303030303, + "grad_norm": 0.054592762142419815, + "learning_rate": 3.308080808080808e-06, + "loss": 0.009655465185642243, + "step": 3830 + }, + { + "epoch": 58.10606060606061, + "grad_norm": 0.03344636410474777, + "learning_rate": 3.1818181818181817e-06, + "loss": 0.010180885344743729, + "step": 3835 + }, + { + "epoch": 58.18181818181818, + "grad_norm": 0.028060782700777054, + "learning_rate": 3.0555555555555556e-06, + "loss": 0.007437943667173386, + "step": 3840 + }, + { + "epoch": 58.25757575757576, + "grad_norm": 0.04080144688487053, + "learning_rate": 2.9292929292929295e-06, + "loss": 0.009837774187326431, + "step": 3845 + }, + { + "epoch": 58.333333333333336, + "grad_norm": 0.02282833121716976, + "learning_rate": 2.803030303030303e-06, + "loss": 0.011929982900619506, + "step": 3850 + }, + { + "epoch": 58.40909090909091, + "grad_norm": 0.042346447706222534, + "learning_rate": 2.676767676767677e-06, + "loss": 0.00977603644132614, + "step": 3855 + }, + { + "epoch": 58.484848484848484, + "grad_norm": 0.03931480273604393, + "learning_rate": 2.550505050505051e-06, + "loss": 0.010096165537834167, + "step": 3860 + }, + { + "epoch": 58.56060606060606, + "grad_norm": 0.030665693804621696, + "learning_rate": 2.4242424242424244e-06, + "loss": 0.008990837633609772, + "step": 3865 + }, + { + "epoch": 58.63636363636363, + "grad_norm": 0.03012900799512863, + "learning_rate": 2.297979797979798e-06, + "loss": 0.009017810970544816, + "step": 3870 + }, + { + "epoch": 58.71212121212121, + "grad_norm": 0.05317816883325577, + "learning_rate": 2.171717171717172e-06, + "loss": 0.009120066463947297, + "step": 3875 + }, + { + "epoch": 58.78787878787879, + "grad_norm": 0.032719820737838745, + "learning_rate": 2.0454545454545457e-06, + "loss": 0.007630207389593124, + "step": 3880 + }, + { + "epoch": 58.86363636363637, + "grad_norm": 0.039780352264642715, + "learning_rate": 1.9191919191919192e-06, + "loss": 0.009303425997495651, + "step": 3885 + }, + { + "epoch": 58.93939393939394, + "grad_norm": 0.04738495871424675, + "learning_rate": 1.7929292929292932e-06, + "loss": 0.008831740915775299, + "step": 3890 + }, + { + "epoch": 59.0, + "eval_loss": 0.010911540128290653, + "eval_mean_accuracy": 0.9615655601299188, + "eval_mean_iou": 0.934988494410513, + "eval_runtime": 132.2712, + "eval_samples_per_second": 1.777, + "eval_steps_per_second": 0.227, + "step": 3894 + }, + { + "epoch": 59.015151515151516, + "grad_norm": 0.03257240727543831, + "learning_rate": 1.6666666666666667e-06, + "loss": 0.010058857500553131, + "step": 3895 + }, + { + "epoch": 59.09090909090909, + "grad_norm": 0.04967406019568443, + "learning_rate": 1.5404040404040406e-06, + "loss": 0.009966336935758591, + "step": 3900 + }, + { + "epoch": 59.166666666666664, + "grad_norm": 0.046529125422239304, + "learning_rate": 1.4141414141414143e-06, + "loss": 0.008606496453285217, + "step": 3905 + }, + { + "epoch": 59.24242424242424, + "grad_norm": 0.03924192860722542, + "learning_rate": 1.287878787878788e-06, + "loss": 0.00987698957324028, + "step": 3910 + }, + { + "epoch": 59.31818181818182, + "grad_norm": 0.057813610881567, + "learning_rate": 1.1616161616161617e-06, + "loss": 0.009356192499399184, + "step": 3915 + }, + { + "epoch": 59.39393939393939, + "grad_norm": 0.029922932386398315, + "learning_rate": 1.0353535353535354e-06, + "loss": 0.009119167923927307, + "step": 3920 + }, + { + "epoch": 59.46969696969697, + "grad_norm": 0.01778416335582733, + "learning_rate": 9.09090909090909e-07, + "loss": 0.008041075617074966, + "step": 3925 + }, + { + "epoch": 59.54545454545455, + "grad_norm": 0.031064176931977272, + "learning_rate": 7.828282828282829e-07, + "loss": 0.00947614312171936, + "step": 3930 + }, + { + "epoch": 59.621212121212125, + "grad_norm": 0.09324733912944794, + "learning_rate": 6.565656565656566e-07, + "loss": 0.008029075711965561, + "step": 3935 + }, + { + "epoch": 59.696969696969695, + "grad_norm": 0.025645431131124496, + "learning_rate": 5.303030303030304e-07, + "loss": 0.009906331449747086, + "step": 3940 + }, + { + "epoch": 59.77272727272727, + "grad_norm": 0.02265552617609501, + "learning_rate": 4.0404040404040405e-07, + "loss": 0.008281528204679488, + "step": 3945 + }, + { + "epoch": 59.84848484848485, + "grad_norm": 0.030991435050964355, + "learning_rate": 2.777777777777778e-07, + "loss": 0.01081312596797943, + "step": 3950 + }, + { + "epoch": 59.92424242424242, + "grad_norm": 0.023471498861908913, + "learning_rate": 1.5151515151515152e-07, + "loss": 0.009151553362607956, + "step": 3955 + }, + { + "epoch": 60.0, + "grad_norm": 0.020656142383813858, + "learning_rate": 2.5252525252525253e-08, + "loss": 0.009477240592241287, + "step": 3960 + }, + { + "epoch": 60.0, + "eval_loss": 0.010845237411558628, + "eval_mean_accuracy": 0.9630000382383805, + "eval_mean_iou": 0.9352648071406789, + "eval_runtime": 129.4443, + "eval_samples_per_second": 1.815, + "eval_steps_per_second": 0.232, + "step": 3960 + } + ], + "logging_steps": 5, + "max_steps": 3960, + "num_input_tokens_seen": 0, + "num_train_epochs": 60, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 0.0, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +}