{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.2690152121697358, "eval_steps": 500, "global_step": 100, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.012810248198558846, "grad_norm": 0.011678964830935001, "learning_rate": 0.0, "loss": 0.0, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.234375, "rewards/strict_format_reward_func/std": 0.2514837086200714, "step": 1, "time_profile/curr_logprobs_and_update": 0.2617394689877983, "time_profile/old_logps": 33.7158753760159, "time_profile/ref_logps": 33.76804309990257, "time_profile/reward": 0.0047389911487698555, "time_profile/text_rollout": 474.3757369443774, "train/gen_step": 64.0, "und/advantages_abs_mean": 0.2001953125, "und/advantages_max": 2.125, "und/advantages_mean": 0.0, "und/advantages_min": -0.4375, "und/advantages_std": 0.2579461336135864, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.751953125, "und/completion/min_length": 250.0, "und/frac_reward_zero_std": 0.046875, "und/kl": 0.0, "und/loss": 0.0, "und/prompt/max_length": 162.0, "und/prompt/mean_length": 107.0, "und/prompt/min_length": 52.0, "und/reward": 0.193359375, "und/reward_std": 0.28102907538414 }, { "epoch": 0.025620496397117692, "grad_norm": 0.00965250376611948, "learning_rate": 1e-05, "loss": 0.0, "step": 2, "time_profile/curr_logprobs_and_update": 0.26145937727415003, "train/gen_step": 128.0, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/kl": 0.0, "und/loss": 0.0 }, { "epoch": 0.03843074459567654, "grad_norm": 0.04503251612186432, "learning_rate": 1e-05, "loss": 0.0, "rewards/correctness_reward_func/mean": 0.4375, "rewards/correctness_reward_func/std": 0.8333333730697632, "rewards/strict_format_reward_func/mean": 0.15625, "rewards/strict_format_reward_func/std": 0.233588308095932, "step": 3, "time_profile/curr_logprobs_and_update": 0.2616513767570723, "time_profile/old_logps": 33.65838770195842, "time_profile/ref_logps": 33.75480777770281, "time_profile/reward": 0.004094251431524754, "time_profile/text_rollout": 504.14777049794793, "train/gen_step": 192.0, "und/advantages_abs_mean": 0.443603515625, "und/advantages_max": 2.1875, "und/advantages_mean": 0.0, "und/advantages_min": -1.6875, "und/advantages_std": 0.7080574631690979, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.79296875, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.421875, "und/kl": 0.00019532828423507453, "und/loss": 7.946068224740088e-06, "und/prompt/max_length": 247.0, "und/prompt/mean_length": 201.5, "und/prompt/min_length": 156.0, "und/reward": 0.5078125, "und/reward_std": 0.9087191820144653 }, { "epoch": 0.051240992794235385, "grad_norm": 0.04360778629779816, "learning_rate": 1e-05, "loss": 0.0001, "step": 4, "time_profile/curr_logprobs_and_update": 0.261664173391182, "train/gen_step": 256.0, "und/clip_ratio/high_mean": 0.0003149772164761089, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0003149772164761089, "und/kl": 0.00025711835689889995, "und/loss": 0.000590196544485444 }, { "epoch": 0.06405124099279423, "grad_norm": 0.005336050409823656, "learning_rate": 1e-05, "loss": 0.0001, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0625, "rewards/strict_format_reward_func/std": 0.1666666716337204, "step": 5, "time_profile/curr_logprobs_and_update": 0.2615119785768911, "time_profile/old_logps": 33.64620256703347, "time_profile/ref_logps": 33.72650748398155, "time_profile/reward": 0.004272850230336189, "time_profile/text_rollout": 491.0981372119859, "train/gen_step": 320.0, "und/advantages_abs_mean": 0.095703125, "und/advantages_max": 2.1875, "und/advantages_mean": 0.0, "und/advantages_min": -0.375, "und/advantages_std": 0.21501831710338593, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.9296875, "und/completion/min_length": 252.0, "und/frac_reward_zero_std": 0.5, "und/kl": 0.00318289036204078, "und/loss": 0.00012254955510471177, "und/prompt/max_length": 247.0, "und/prompt/mean_length": 152.5, "und/prompt/min_length": 58.0, "und/reward": 0.06640625, "und/reward_std": 0.23300214111804962 }, { "epoch": 0.07686148919135308, "grad_norm": 0.006930716801434755, "learning_rate": 1e-05, "loss": 0.0003, "step": 6, "time_profile/curr_logprobs_and_update": 0.26136891632631887, "train/gen_step": 384.0, "und/clip_ratio/high_mean": 0.0004957187084073666, "und/clip_ratio/low_mean": 0.00018175119475927204, "und/clip_ratio/region_mean": 0.0006774698958906811, "und/kl": 0.0036970916280552046, "und/loss": 5.486323918546532e-05 }, { "epoch": 0.08967173738991192, "grad_norm": 0.012152941897511482, "learning_rate": 1e-05, "loss": 0.0001, "rewards/correctness_reward_func/mean": 0.03125, "rewards/correctness_reward_func/std": 0.25, "rewards/strict_format_reward_func/mean": 0.15625, "rewards/strict_format_reward_func/std": 0.233588308095932, "step": 7, "time_profile/curr_logprobs_and_update": 0.26169058527739253, "time_profile/old_logps": 33.673401176929474, "time_profile/ref_logps": 33.76787843555212, "time_profile/reward": 0.00381280854344368, "time_profile/text_rollout": 473.39493833016604, "train/gen_step": 448.0, "und/advantages_abs_mean": 0.197509765625, "und/advantages_max": 2.0625, "und/advantages_mean": 0.0, "und/advantages_min": -0.5625, "und/advantages_std": 0.2547253966331482, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.755859375, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.0625, "und/kl": 0.0018817147374647902, "und/loss": 4.349416121840477e-05, "und/prompt/max_length": 162.0, "und/prompt/mean_length": 107.0, "und/prompt/min_length": 52.0, "und/reward": 0.1689453125, "und/reward_std": 0.27497100830078125 }, { "epoch": 0.10248198558847077, "grad_norm": 0.011709543876349926, "learning_rate": 1e-05, "loss": 0.0001, "step": 8, "time_profile/curr_logprobs_and_update": 0.2615692205145024, "train/gen_step": 512.0, "und/clip_ratio/high_mean": 0.00020409237549756654, "und/clip_ratio/low_mean": 0.00016332855375367217, "und/clip_ratio/region_mean": 0.0003674209328892175, "und/kl": 0.0024647556260788406, "und/loss": -4.9581576604396105e-05 }, { "epoch": 0.11529223378702963, "grad_norm": 0.05086541920900345, "learning_rate": 1e-05, "loss": 0.0001, "rewards/correctness_reward_func/mean": 0.6875, "rewards/correctness_reward_func/std": 0.9574271440505981, "rewards/strict_format_reward_func/mean": 0.2734375, "rewards/strict_format_reward_func/std": 0.250866562128067, "step": 9, "time_profile/curr_logprobs_and_update": 0.2616978839650983, "time_profile/old_logps": 33.66389245353639, "time_profile/ref_logps": 33.785877684131265, "time_profile/reward": 0.004252792336046696, "time_profile/text_rollout": 488.8415644085035, "train/gen_step": 576.0, "und/advantages_abs_mean": 0.7744140625, "und/advantages_max": 2.0625, "und/advantages_mean": 0.0, "und/advantages_min": -2.0625, "und/advantages_std": 0.9534159302711487, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.6171875, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.0, "und/kl": 0.0022935827773835626, "und/loss": 0.00010531611042097211, "und/prompt/max_length": 156.0, "und/prompt/mean_length": 156.0, "und/prompt/min_length": 156.0, "und/reward": 1.041015625, "und/reward_std": 1.1144300699234009 }, { "epoch": 0.12810248198558846, "grad_norm": 0.07067691534757614, "learning_rate": 1e-05, "loss": 0.0002, "step": 10, "time_profile/curr_logprobs_and_update": 0.26148990585352294, "train/gen_step": 640.0, "und/clip_ratio/high_mean": 0.000207764285732992, "und/clip_ratio/low_mean": 0.00026153815633733757, "und/clip_ratio/region_mean": 0.00046930244570830837, "und/kl": 0.0025965598733819206, "und/loss": -0.00032083288533613086 }, { "epoch": 0.14091273018414732, "grad_norm": 0.011798444204032421, "learning_rate": 1e-05, "loss": 0.0001, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.3359375, "rewards/strict_format_reward_func/std": 0.2366211861371994, "step": 11, "time_profile/curr_logprobs_and_update": 0.2613880275603151, "time_profile/old_logps": 33.668298315256834, "time_profile/ref_logps": 33.73314023669809, "time_profile/reward": 0.0044129379093647, "time_profile/text_rollout": 454.3680015280843, "train/gen_step": 704.0, "und/advantages_abs_mean": 0.1396484375, "und/advantages_max": 1.75, "und/advantages_mean": 0.0, "und/advantages_min": -0.4375, "und/advantages_std": 0.20029333233833313, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.6640625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.28125, "und/kl": 0.0030043681945244316, "und/loss": 4.8866542471159846e-05, "und/prompt/max_length": 49.0, "und/prompt/mean_length": 49.0, "und/prompt/min_length": 49.0, "und/reward": 0.34765625, "und/reward_std": 0.25070229172706604 }, { "epoch": 0.15372297838270615, "grad_norm": 0.00986644346266985, "learning_rate": 1e-05, "loss": 0.0002, "step": 12, "time_profile/curr_logprobs_and_update": 0.2613714834296843, "train/gen_step": 768.0, "und/clip_ratio/high_mean": 0.0003061741153942421, "und/clip_ratio/low_mean": 0.0004524241703620646, "und/clip_ratio/region_mean": 0.0007585982857563067, "und/kl": 0.0042751526357278635, "und/loss": 0.0005581114168933254 }, { "epoch": 0.16653322658126501, "grad_norm": 0.044079430401325226, "learning_rate": 1e-05, "loss": 0.0002, "rewards/correctness_reward_func/mean": 0.78125, "rewards/correctness_reward_func/std": 0.983494758605957, "rewards/strict_format_reward_func/mean": 0.375, "rewards/strict_format_reward_func/std": 0.2182178944349289, "step": 13, "time_profile/curr_logprobs_and_update": 0.2613269358407706, "time_profile/old_logps": 33.61356293410063, "time_profile/ref_logps": 33.69876565504819, "time_profile/reward": 0.004282657988369465, "time_profile/text_rollout": 455.1681863386184, "train/gen_step": 832.0, "und/advantages_abs_mean": 0.33984375, "und/advantages_max": 1.0625, "und/advantages_mean": 0.0, "und/advantages_min": -2.1875, "und/advantages_std": 0.5323442816734314, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.69140625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.171875, "und/kl": 0.00535378480435611, "und/loss": 0.00024319114163517952, "und/prompt/max_length": 55.0, "und/prompt/mean_length": 51.0, "und/prompt/min_length": 47.0, "und/reward": 1.21875, "und/reward_std": 1.1256657838821411 }, { "epoch": 0.17934347477982385, "grad_norm": 0.04768079146742821, "learning_rate": 1e-05, "loss": 0.0001, "step": 14, "time_profile/curr_logprobs_and_update": 0.2612948026799131, "train/gen_step": 896.0, "und/clip_ratio/high_mean": 0.0003154302030452527, "und/clip_ratio/low_mean": 0.0004365383902040776, "und/clip_ratio/region_mean": 0.0007519685932493303, "und/kl": 0.005003000936994795, "und/loss": -6.164377555251122e-05 }, { "epoch": 0.1921537229783827, "grad_norm": 0.013837647624313831, "learning_rate": 1e-05, "loss": 0.0001, "rewards/correctness_reward_func/mean": 0.03125, "rewards/correctness_reward_func/std": 0.25, "rewards/strict_format_reward_func/mean": 0.3203125, "rewards/strict_format_reward_func/std": 0.24180518090724945, "step": 15, "time_profile/curr_logprobs_and_update": 0.261544900276931, "time_profile/old_logps": 33.64624940324575, "time_profile/ref_logps": 33.727370373904705, "time_profile/reward": 0.004188773222267628, "time_profile/text_rollout": 454.6840885374695, "train/gen_step": 960.0, "und/advantages_abs_mean": 0.149658203125, "und/advantages_max": 1.9375, "und/advantages_mean": 0.0, "und/advantages_min": -0.5625, "und/advantages_std": 0.2218771129846573, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.619140625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.203125, "und/kl": 0.002458287915942492, "und/loss": 0.00015533823614077846, "und/prompt/max_length": 51.0, "und/prompt/mean_length": 49.5, "und/prompt/min_length": 48.0, "und/reward": 0.3330078125, "und/reward_std": 0.2671593129634857 }, { "epoch": 0.20496397117694154, "grad_norm": 0.012836061418056488, "learning_rate": 1e-05, "loss": 0.0001, "step": 16, "time_profile/curr_logprobs_and_update": 0.2614677896053763, "train/gen_step": 1024.0, "und/clip_ratio/high_mean": 0.0010820379393408075, "und/clip_ratio/low_mean": 5.425347262644209e-05, "und/clip_ratio/region_mean": 0.0011362914119672496, "und/kl": 0.0026947305022986257, "und/loss": 1.8947941015312608e-05 }, { "epoch": 0.2177742193755004, "grad_norm": 0.03666432574391365, "learning_rate": 1e-05, "loss": 0.0002, "rewards/correctness_reward_func/mean": 0.125, "rewards/correctness_reward_func/std": 0.48795005679130554, "rewards/strict_format_reward_func/mean": 0.484375, "rewards/strict_format_reward_func/std": 0.08768405020236969, "step": 17, "time_profile/curr_logprobs_and_update": 0.26129884374677204, "time_profile/old_logps": 33.58372610062361, "time_profile/ref_logps": 33.71312084700912, "time_profile/reward": 0.004377733916044235, "time_profile/text_rollout": 453.4618791854009, "train/gen_step": 1088.0, "und/advantages_abs_mean": 0.24462890625, "und/advantages_max": 1.875, "und/advantages_mean": 0.0, "und/advantages_min": -0.9375, "und/advantages_std": 0.4919235110282898, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.58984375, "und/completion/min_length": 216.0, "und/frac_reward_zero_std": 0.453125, "und/kl": 0.003914871423148725, "und/loss": 0.00013992213200708647, "und/prompt/max_length": 45.0, "und/prompt/mean_length": 44.0, "und/prompt/min_length": 43.0, "und/reward": 0.6201171875, "und/reward_std": 0.5268946886062622 }, { "epoch": 0.23058446757405926, "grad_norm": 0.04468945786356926, "learning_rate": 1e-05, "loss": 0.0003, "step": 18, "time_profile/curr_logprobs_and_update": 0.26125960882927757, "train/gen_step": 1152.0, "und/clip_ratio/high_mean": 0.00032832663418957964, "und/clip_ratio/low_mean": 0.0001570866115798708, "und/clip_ratio/region_mean": 0.00048541324576945044, "und/kl": 0.004693901622886187, "und/loss": 0.0001238020138245588 }, { "epoch": 0.2433947157726181, "grad_norm": 0.04278712719678879, "learning_rate": 1e-05, "loss": 0.0002, "rewards/correctness_reward_func/mean": 0.34375, "rewards/correctness_reward_func/std": 0.7605084180831909, "rewards/strict_format_reward_func/mean": 0.4140625, "rewards/strict_format_reward_func/std": 0.19012710452079773, "step": 19, "time_profile/curr_logprobs_and_update": 0.2615380200877553, "time_profile/old_logps": 33.66465362161398, "time_profile/ref_logps": 33.743118970654905, "time_profile/reward": 0.00426104012876749, "time_profile/text_rollout": 454.9674591496587, "train/gen_step": 1216.0, "und/advantages_abs_mean": 0.5419921875, "und/advantages_max": 1.875, "und/advantages_mean": 0.0, "und/advantages_min": -1.6875, "und/advantages_std": 0.6924689412117004, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.603515625, "und/completion/min_length": 247.0, "und/frac_reward_zero_std": 0.0, "und/kl": 0.005255785068584373, "und/loss": 0.000339222839102149, "und/prompt/max_length": 53.0, "und/prompt/mean_length": 52.0, "und/prompt/min_length": 51.0, "und/reward": 0.7529296875, "und/reward_std": 0.8764092326164246 }, { "epoch": 0.2562049639711769, "grad_norm": 0.05156658962368965, "learning_rate": 1e-05, "loss": 0.0005, "step": 20, "time_profile/curr_logprobs_and_update": 0.2612772498978302, "train/gen_step": 1280.0, "und/clip_ratio/high_mean": 0.000682802070514299, "und/clip_ratio/low_mean": 0.0004532934799499344, "und/clip_ratio/region_mean": 0.0011360955541022122, "und/kl": 0.007088688389558229, "und/loss": -0.000385153922252357 }, { "epoch": 0.2690152121697358, "grad_norm": 0.010599356144666672, "learning_rate": 1e-05, "loss": 0.0004, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.03125, "rewards/strict_format_reward_func/std": 0.12198751419782639, "step": 21, "time_profile/curr_logprobs_and_update": 0.26165466259408277, "time_profile/old_logps": 33.66928117442876, "time_profile/ref_logps": 33.74430500715971, "time_profile/reward": 0.005226008594036102, "time_profile/text_rollout": 505.46357171423733, "train/gen_step": 1344.0, "und/advantages_abs_mean": 0.0986328125, "und/advantages_max": 2.1875, "und/advantages_mean": 0.0, "und/advantages_min": -0.375, "und/advantages_std": 0.23039571940898895, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.90625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.546875, "und/kl": 0.009255951059458312, "und/loss": 0.0003061078954686991, "und/prompt/max_length": 247.0, "und/prompt/mean_length": 204.5, "und/prompt/min_length": 162.0, "und/reward": 0.068359375, "und/reward_std": 0.2506641745567322 }, { "epoch": 0.28182546036829464, "grad_norm": 0.006633961573243141, "learning_rate": 1e-05, "loss": 0.0005, "step": 22, "time_profile/curr_logprobs_and_update": 0.2616323969559744, "train/gen_step": 1408.0, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.00017277644656132907, "und/clip_ratio/region_mean": 0.00017277644656132907, "und/kl": 0.01153348765183182, "und/loss": 0.0003008711780587703 }, { "epoch": 0.2946357085668535, "grad_norm": 0.01536457147449255, "learning_rate": 1e-05, "loss": 0.0005, "rewards/correctness_reward_func/mean": 0.90625, "rewards/correctness_reward_func/std": 1.003466248512268, "rewards/strict_format_reward_func/mean": 0.34375, "rewards/strict_format_reward_func/std": 0.233588308095932, "step": 23, "time_profile/curr_logprobs_and_update": 0.26142213236016687, "time_profile/old_logps": 33.64063487108797, "time_profile/ref_logps": 33.71632799040526, "time_profile/reward": 0.004507049918174744, "time_profile/text_rollout": 456.5940729761496, "train/gen_step": 1472.0, "und/advantages_abs_mean": 0.3193359375, "und/advantages_max": 1.0, "und/advantages_mean": 0.0, "und/advantages_min": -2.1875, "und/advantages_std": 0.508188784122467, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.421875, "und/completion/min_length": 126.0, "und/frac_reward_zero_std": 0.109375, "und/kl": 0.011904407194379019, "und/loss": 0.000405974641296325, "und/prompt/max_length": 68.0, "und/prompt/mean_length": 57.0, "und/prompt/min_length": 46.0, "und/reward": 1.1806640625, "und/reward_std": 1.1607900857925415 }, { "epoch": 0.3074459567654123, "grad_norm": 0.017077524214982986, "learning_rate": 1e-05, "loss": 0.0007, "step": 24, "time_profile/curr_logprobs_and_update": 0.26139352870814037, "train/gen_step": 1536.0, "und/clip_ratio/high_mean": 0.0003636985675257165, "und/clip_ratio/low_mean": 0.000808012075140141, "und/clip_ratio/region_mean": 0.0011717106426658574, "und/kl": 0.0162865712918574, "und/loss": 0.0004572992978353341 }, { "epoch": 0.32025620496397117, "grad_norm": 0.011341015808284283, "learning_rate": 1e-05, "loss": 0.0009, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.15625, "rewards/strict_format_reward_func/std": 0.233588308095932, "step": 25, "time_profile/curr_logprobs_and_update": 0.2614934611192439, "time_profile/old_logps": 33.66488476470113, "time_profile/ref_logps": 33.74662786722183, "time_profile/reward": 0.0043584490194916725, "time_profile/text_rollout": 473.53795502893627, "train/gen_step": 1600.0, "und/advantages_abs_mean": 0.18212890625, "und/advantages_max": 2.1875, "und/advantages_mean": 0.0, "und/advantages_min": -0.3125, "und/advantages_std": 0.2327723652124405, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.763671875, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.078125, "und/kl": 0.02267802683309128, "und/loss": 0.001361295189667544, "und/prompt/max_length": 162.0, "und/prompt/mean_length": 109.5, "und/prompt/min_length": 57.0, "und/reward": 0.14453125, "und/reward_std": 0.2475108504295349 }, { "epoch": 0.33306645316253003, "grad_norm": 0.014138264581561089, "learning_rate": 1e-05, "loss": 0.0014, "step": 26, "time_profile/curr_logprobs_and_update": 0.2614813649561256, "train/gen_step": 1664.0, "und/clip_ratio/high_mean": 9.385726662003435e-05, "und/clip_ratio/low_mean": 0.0006062783650122583, "und/clip_ratio/region_mean": 0.0007001356316322926, "und/kl": 0.03298538580929744, "und/loss": 0.0003666753910565035 }, { "epoch": 0.3458767013610889, "grad_norm": 0.036602944135665894, "learning_rate": 1e-05, "loss": 0.0017, "rewards/correctness_reward_func/mean": 0.15625, "rewards/correctness_reward_func/std": 0.5409794449806213, "rewards/strict_format_reward_func/mean": 0.296875, "rewards/strict_format_reward_func/std": 0.24750742316246033, "step": 27, "time_profile/curr_logprobs_and_update": 0.26141884991375264, "time_profile/old_logps": 33.63414192106575, "time_profile/ref_logps": 33.71741700544953, "time_profile/reward": 0.004512332379817963, "time_profile/text_rollout": 454.9066513767466, "train/gen_step": 1728.0, "und/advantages_abs_mean": 0.361572265625, "und/advantages_max": 2.0, "und/advantages_mean": 0.0, "und/advantages_min": -1.25, "und/advantages_std": 0.5393062829971313, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.65234375, "und/completion/min_length": 250.0, "und/frac_reward_zero_std": 0.015625, "und/kl": 0.04265276483056368, "und/loss": 0.0006861621513962746, "und/prompt/max_length": 59.0, "und/prompt/mean_length": 52.0, "und/prompt/min_length": 45.0, "und/reward": 0.44921875, "und/reward_std": 0.6556687951087952 }, { "epoch": 0.3586869495596477, "grad_norm": 0.05098080635070801, "learning_rate": 1e-05, "loss": 0.0015, "step": 28, "time_profile/curr_logprobs_and_update": 0.261381899996195, "train/gen_step": 1792.0, "und/clip_ratio/high_mean": 0.0009093214357562829, "und/clip_ratio/low_mean": 0.0001023065487970598, "und/clip_ratio/region_mean": 0.0010116279845533427, "und/kl": 0.037396987856482156, "und/loss": 0.0021928895730525255 }, { "epoch": 0.37149719775820655, "grad_norm": 0.009760422632098198, "learning_rate": 1e-05, "loss": 0.0023, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.2734375, "rewards/strict_format_reward_func/std": 0.250866562128067, "step": 29, "time_profile/curr_logprobs_and_update": 0.26138593546056654, "time_profile/old_logps": 33.63254508841783, "time_profile/ref_logps": 33.774293441325426, "time_profile/reward": 0.004322300665080547, "time_profile/text_rollout": 455.36977090034634, "train/gen_step": 1856.0, "und/advantages_abs_mean": 0.2197265625, "und/advantages_max": 0.4375, "und/advantages_mean": 0.0, "und/advantages_min": -0.4375, "und/advantages_std": 0.23460420966148376, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.712890625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.015625, "und/kl": 0.056478848066035425, "und/loss": 0.006321181455859914, "und/prompt/max_length": 54.0, "und/prompt/mean_length": 53.5, "und/prompt/min_length": 53.0, "und/reward": 0.236328125, "und/reward_std": 0.24987001717090607 }, { "epoch": 0.3843074459567654, "grad_norm": 0.010228869505226612, "learning_rate": 1e-05, "loss": 0.0015, "step": 30, "time_profile/curr_logprobs_and_update": 0.2613552274706308, "train/gen_step": 1920.0, "und/clip_ratio/high_mean": 0.0011887354085047264, "und/clip_ratio/low_mean": 4.245923992129974e-05, "und/clip_ratio/region_mean": 0.001231194648426026, "und/kl": 0.023746514114463935, "und/loss": 0.0005230403039604425 }, { "epoch": 0.3971176941553243, "grad_norm": 0.050902411341667175, "learning_rate": 1e-05, "loss": 0.0008, "rewards/correctness_reward_func/mean": 0.5, "rewards/correctness_reward_func/std": 0.8728715777397156, "rewards/strict_format_reward_func/mean": 0.34375, "rewards/strict_format_reward_func/std": 0.233588308095932, "step": 31, "time_profile/curr_logprobs_and_update": 0.26159091785666533, "time_profile/old_logps": 33.64319808781147, "time_profile/ref_logps": 33.71481241378933, "time_profile/reward": 0.004479923285543919, "time_profile/text_rollout": 473.2088502328843, "train/gen_step": 1984.0, "und/advantages_abs_mean": 0.535400390625, "und/advantages_max": 1.5625, "und/advantages_mean": 0.0, "und/advantages_min": -1.75, "und/advantages_std": 0.6867490410804749, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.015625, "und/kl": 0.018857156013837084, "und/loss": 0.0008609343785792589, "und/prompt/max_length": 162.0, "und/prompt/mean_length": 108.0, "und/prompt/min_length": 54.0, "und/reward": 0.83203125, "und/reward_std": 1.0073648691177368 }, { "epoch": 0.4099279423538831, "grad_norm": 0.06242002546787262, "learning_rate": 1e-05, "loss": 0.0015, "step": 32, "time_profile/curr_logprobs_and_update": 0.26177428937808145, "train/gen_step": 2048.0, "und/clip_ratio/high_mean": 0.0007951511470309924, "und/clip_ratio/low_mean": 0.00020771568233612925, "und/clip_ratio/region_mean": 0.0010028668257291429, "und/kl": 0.03375891799441888, "und/loss": 0.000749626662582159 }, { "epoch": 0.42273819055244194, "grad_norm": 0.04212071746587753, "learning_rate": 1e-05, "loss": 0.0006, "rewards/correctness_reward_func/mean": 0.5, "rewards/correctness_reward_func/std": 0.8728715777397156, "rewards/strict_format_reward_func/mean": 0.3125, "rewards/strict_format_reward_func/std": 0.24397502839565277, "step": 33, "time_profile/curr_logprobs_and_update": 0.26152711105532944, "time_profile/old_logps": 33.65837723109871, "time_profile/ref_logps": 33.75508733931929, "time_profile/reward": 0.0049184514209628105, "time_profile/text_rollout": 472.5054632341489, "train/gen_step": 2112.0, "und/advantages_abs_mean": 0.53515625, "und/advantages_max": 1.75, "und/advantages_mean": 0.0, "und/advantages_min": -1.75, "und/advantages_std": 0.6916294097900391, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.671875, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.015625, "und/kl": 0.015675050650315825, "und/loss": 0.0005543280858546495, "und/prompt/max_length": 162.0, "und/prompt/mean_length": 104.5, "und/prompt/min_length": 47.0, "und/reward": 0.84375, "und/reward_std": 1.0150530338287354 }, { "epoch": 0.4355484387510008, "grad_norm": 0.0346144400537014, "learning_rate": 1e-05, "loss": 0.0012, "step": 34, "time_profile/curr_logprobs_and_update": 0.2614700558333425, "train/gen_step": 2176.0, "und/clip_ratio/high_mean": 0.0002005133756028954, "und/clip_ratio/low_mean": 0.00024482345179421827, "und/clip_ratio/region_mean": 0.00044533682739711367, "und/kl": 0.025506131180009106, "und/loss": 0.001075362495612353 }, { "epoch": 0.44835868694955966, "grad_norm": 0.04864125698804855, "learning_rate": 1e-05, "loss": 0.0008, "rewards/correctness_reward_func/mean": 0.1875, "rewards/correctness_reward_func/std": 0.5875696539878845, "rewards/strict_format_reward_func/mean": 0.25, "rewards/strict_format_reward_func/std": 0.2519763112068176, "step": 35, "time_profile/curr_logprobs_and_update": 0.2617200283857528, "time_profile/old_logps": 33.63409499172121, "time_profile/ref_logps": 33.74282674957067, "time_profile/reward": 0.004408497363328934, "time_profile/text_rollout": 487.6969060283154, "train/gen_step": 2240.0, "und/advantages_abs_mean": 0.337890625, "und/advantages_max": 1.875, "und/advantages_mean": 0.0, "und/advantages_min": -1.4375, "und/advantages_std": 0.585414707660675, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.32421875, "und/completion/min_length": 175.0, "und/frac_reward_zero_std": 0.46875, "und/kl": 0.019317339225381147, "und/loss": 0.0008824675582559394, "und/prompt/max_length": 247.0, "und/prompt/mean_length": 145.0, "und/prompt/min_length": 43.0, "und/reward": 0.478515625, "und/reward_std": 0.7729067802429199 }, { "epoch": 0.4611689351481185, "grad_norm": 0.043183691799640656, "learning_rate": 1e-05, "loss": 0.0009, "step": 36, "time_profile/curr_logprobs_and_update": 0.26155611700960435, "train/gen_step": 2304.0, "und/clip_ratio/high_mean": 0.00021402155107352883, "und/clip_ratio/low_mean": 0.00021690024368581362, "und/clip_ratio/region_mean": 0.00043092179839732125, "und/kl": 0.024045594029303174, "und/loss": 0.0011579781168506997 }, { "epoch": 0.4739791833466773, "grad_norm": 0.010638823732733727, "learning_rate": 1e-05, "loss": 0.0009, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.109375, "rewards/strict_format_reward_func/std": 0.2083333432674408, "step": 37, "time_profile/curr_logprobs_and_update": 0.2616240307397675, "time_profile/old_logps": 33.59795498009771, "time_profile/ref_logps": 33.70516960695386, "time_profile/reward": 0.004617785103619099, "time_profile/text_rollout": 487.4186108149588, "train/gen_step": 2368.0, "und/advantages_abs_mean": 0.136962890625, "und/advantages_max": 2.1875, "und/advantages_mean": 0.0, "und/advantages_min": -0.625, "und/advantages_std": 0.25544461607933044, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.880859375, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.4375, "und/kl": 0.022708220010827063, "und/loss": 0.0009557972493325906, "und/prompt/max_length": 247.0, "und/prompt/mean_length": 151.0, "und/prompt/min_length": 55.0, "und/reward": 0.1318359375, "und/reward_std": 0.2962620258331299 }, { "epoch": 0.4867894315452362, "grad_norm": 0.010083939880132675, "learning_rate": 1e-05, "loss": 0.001, "step": 38, "time_profile/curr_logprobs_and_update": 0.2614540692156879, "train/gen_step": 2432.0, "und/clip_ratio/high_mean": 0.000718017738108756, "und/clip_ratio/low_mean": 0.00015644555605831556, "und/clip_ratio/region_mean": 0.0008744632941670716, "und/kl": 0.024760687578236684, "und/loss": 0.001179238304757746 }, { "epoch": 0.49959967974379504, "grad_norm": 0.03500046208500862, "learning_rate": 1e-05, "loss": 0.0015, "rewards/correctness_reward_func/mean": 0.71875, "rewards/correctness_reward_func/std": 0.9672207236289978, "rewards/strict_format_reward_func/mean": 0.234375, "rewards/strict_format_reward_func/std": 0.2514837086200714, "step": 39, "time_profile/curr_logprobs_and_update": 0.2616706858971156, "time_profile/old_logps": 33.69548157788813, "time_profile/ref_logps": 33.755788206122816, "time_profile/reward": 0.004642653279006481, "time_profile/text_rollout": 503.46805787831545, "train/gen_step": 2496.0, "und/advantages_abs_mean": 0.399658203125, "und/advantages_max": 1.375, "und/advantages_mean": 0.0, "und/advantages_min": -1.9375, "und/advantages_std": 0.6456708908081055, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 253.01953125, "und/completion/min_length": 21.0, "und/frac_reward_zero_std": 0.515625, "und/kl": 0.03765736366040073, "und/loss": 0.000907381055640144, "und/prompt/max_length": 313.0, "und/prompt/mean_length": 179.5, "und/prompt/min_length": 46.0, "und/reward": 0.9072265625, "und/reward_std": 1.1414990425109863 }, { "epoch": 0.5124099279423538, "grad_norm": 0.047211844474077225, "learning_rate": 1e-05, "loss": 0.0014, "step": 40, "time_profile/curr_logprobs_and_update": 0.2616095836274326, "train/gen_step": 2560.0, "und/clip_ratio/high_mean": 0.00029819617702742107, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.00029819617702742107, "und/kl": 0.03165427554631606, "und/loss": 0.0009896907330357863 }, { "epoch": 0.5252201761409128, "grad_norm": 0.012100051157176495, "learning_rate": 1e-05, "loss": 0.0132, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.171875, "rewards/strict_format_reward_func/std": 0.23935678601264954, "step": 41, "time_profile/curr_logprobs_and_update": 0.26168578446959145, "time_profile/old_logps": 33.68038745317608, "time_profile/ref_logps": 33.777247093617916, "time_profile/reward": 0.004466196522116661, "time_profile/text_rollout": 491.4084973083809, "train/gen_step": 2624.0, "und/advantages_abs_mean": 0.169189453125, "und/advantages_max": 0.4375, "und/advantages_mean": 0.0, "und/advantages_min": -0.375, "und/advantages_std": 0.20586435496807098, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.802734375, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.109375, "und/kl": 0.33045044860773487, "und/loss": 0.013275288394652307, "und/prompt/max_length": 162.0, "und/prompt/mean_length": 162.0, "und/prompt/min_length": 162.0, "und/reward": 0.1376953125, "und/reward_std": 0.22357389330863953 }, { "epoch": 0.5380304243394716, "grad_norm": 0.009308804757893085, "learning_rate": 1e-05, "loss": 0.0254, "step": 42, "time_profile/curr_logprobs_and_update": 0.26165859241154976, "train/gen_step": 2688.0, "und/clip_ratio/high_mean": 0.00019616381541709416, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.00019616381541709416, "und/kl": 0.6357102408874198, "und/loss": 0.028003237792290747 }, { "epoch": 0.5508406725380304, "grad_norm": 0.055917106568813324, "learning_rate": 1e-05, "loss": 0.0037, "rewards/correctness_reward_func/mean": 0.5625, "rewards/correctness_reward_func/std": 0.9063270092010498, "rewards/strict_format_reward_func/mean": 0.3125, "rewards/strict_format_reward_func/std": 0.24397502839565277, "step": 43, "time_profile/curr_logprobs_and_update": 0.26144762606418226, "time_profile/old_logps": 33.624562768265605, "time_profile/ref_logps": 33.73785715736449, "time_profile/reward": 0.004686887376010418, "time_profile/text_rollout": 454.951701448299, "train/gen_step": 2752.0, "und/advantages_abs_mean": 0.566650390625, "und/advantages_max": 1.5, "und/advantages_mean": 0.0, "und/advantages_min": -1.9375, "und/advantages_std": 0.7055483460426331, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.75390625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.0, "und/kl": 0.09146726290055085, "und/loss": 0.0016110616270452738, "und/prompt/max_length": 57.0, "und/prompt/mean_length": 50.5, "und/prompt/min_length": 44.0, "und/reward": 0.859375, "und/reward_std": 0.9939887523651123 }, { "epoch": 0.5636509207365893, "grad_norm": 0.0613945871591568, "learning_rate": 1e-05, "loss": 0.0074, "step": 44, "time_profile/curr_logprobs_and_update": 0.26146509549289476, "train/gen_step": 2816.0, "und/clip_ratio/high_mean": 0.00215041778574232, "und/clip_ratio/low_mean": 0.00034797564512700774, "und/clip_ratio/region_mean": 0.002498393430869328, "und/kl": 0.0361222850260674, "und/loss": 0.005510351620614529 }, { "epoch": 0.5764611689351481, "grad_norm": 0.009947679936885834, "learning_rate": 1e-05, "loss": 0.0012, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.28125, "rewards/strict_format_reward_func/std": 0.25, "step": 45, "time_profile/curr_logprobs_and_update": 0.26160548186453525, "time_profile/old_logps": 33.671746659092605, "time_profile/ref_logps": 33.75626815017313, "time_profile/reward": 0.3108100714161992, "time_profile/text_rollout": 472.72174780536443, "train/gen_step": 2880.0, "und/advantages_abs_mean": 0.1962890625, "und/advantages_max": 0.4375, "und/advantages_mean": 0.0, "und/advantages_min": -0.4375, "und/advantages_std": 0.2217392474412918, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.67578125, "und/completion/min_length": 250.0, "und/frac_reward_zero_std": 0.046875, "und/kl": 0.02999453744996572, "und/loss": 0.0007559580495524187, "und/prompt/max_length": 162.0, "und/prompt/mean_length": 104.5, "und/prompt/min_length": 47.0, "und/reward": 0.255859375, "und/reward_std": 0.25017574429512024 }, { "epoch": 0.589271417133707, "grad_norm": 0.012918645516037941, "learning_rate": 1e-05, "loss": 0.0016, "step": 46, "time_profile/curr_logprobs_and_update": 0.2616680209466722, "train/gen_step": 2944.0, "und/clip_ratio/high_mean": 0.0007937824193504639, "und/clip_ratio/low_mean": 4.763719334732741e-05, "und/clip_ratio/region_mean": 0.0008414196126977913, "und/kl": 0.03639418514649151, "und/loss": 0.001037593943193471 }, { "epoch": 0.6020816653322658, "grad_norm": 0.005492714699357748, "learning_rate": 1e-05, "loss": 0.0545, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.140625, "rewards/strict_format_reward_func/std": 0.22658175230026245, "step": 47, "time_profile/curr_logprobs_and_update": 0.2618762481288286, "time_profile/old_logps": 33.7198549490422, "time_profile/ref_logps": 33.79357417766005, "time_profile/reward": 0.004348478280007839, "time_profile/text_rollout": 491.14609040971845, "train/gen_step": 3008.0, "und/advantages_abs_mean": 0.188232421875, "und/advantages_max": 1.9375, "und/advantages_mean": 0.0, "und/advantages_min": -0.5625, "und/advantages_std": 0.24418380856513977, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.794921875, "und/completion/min_length": 252.0, "und/frac_reward_zero_std": 0.046875, "und/kl": 1.3629103746789042, "und/loss": 0.0005332264117896557, "und/prompt/max_length": 162.0, "und/prompt/mean_length": 162.0, "und/prompt/min_length": 162.0, "und/reward": 0.1552734375, "und/reward_std": 0.26322463154792786 }, { "epoch": 0.6148919135308246, "grad_norm": 0.007346948143094778, "learning_rate": 1e-05, "loss": 0.0133, "step": 48, "time_profile/curr_logprobs_and_update": 0.26197397308715153, "train/gen_step": 3072.0, "und/clip_ratio/high_mean": 4.882812572759576e-05, "und/clip_ratio/low_mean": 7.812499825377017e-05, "und/clip_ratio/region_mean": 0.00012695312398136593, "und/kl": 0.3335349300832604, "und/loss": 0.000871355994604528 }, { "epoch": 0.6277021617293835, "grad_norm": 0.041968513280153275, "learning_rate": 1e-05, "loss": 0.0017, "rewards/correctness_reward_func/mean": 0.71875, "rewards/correctness_reward_func/std": 0.9672207236289978, "rewards/strict_format_reward_func/mean": 0.3359375, "rewards/strict_format_reward_func/std": 0.2366211861371994, "step": 49, "time_profile/curr_logprobs_and_update": 0.26143342345312703, "time_profile/old_logps": 33.66150098387152, "time_profile/ref_logps": 33.79388515371829, "time_profile/reward": 0.0042456211522221565, "time_profile/text_rollout": 455.7687773555517, "train/gen_step": 3136.0, "und/advantages_abs_mean": 0.4990234375, "und/advantages_max": 1.5625, "und/advantages_mean": 0.0, "und/advantages_min": -2.1875, "und/advantages_std": 0.6655394434928894, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.501953125, "und/completion/min_length": 159.0, "und/frac_reward_zero_std": 0.03125, "und/kl": 0.043663310549163725, "und/loss": 0.001001961762085557, "und/prompt/max_length": 57.0, "und/prompt/mean_length": 52.5, "und/prompt/min_length": 48.0, "und/reward": 0.9833984375, "und/reward_std": 1.0895419120788574 }, { "epoch": 0.6405124099279423, "grad_norm": 0.03877226635813713, "learning_rate": 1e-05, "loss": 0.0014, "step": 50, "time_profile/curr_logprobs_and_update": 0.26151250462862663, "train/gen_step": 3200.0, "und/clip_ratio/high_mean": 0.0006906339149281848, "und/clip_ratio/low_mean": 0.001115720056986902, "und/clip_ratio/region_mean": 0.0018063539719150867, "und/kl": 0.03958344936108915, "und/loss": 0.0018563272897154093 }, { "epoch": 0.6533226581265013, "grad_norm": 0.040287211537361145, "learning_rate": 1e-05, "loss": 0.0016, "rewards/correctness_reward_func/mean": 0.8125, "rewards/correctness_reward_func/std": 0.9900296926498413, "rewards/strict_format_reward_func/mean": 0.359375, "rewards/strict_format_reward_func/std": 0.22658175230026245, "step": 51, "time_profile/curr_logprobs_and_update": 0.2614360564039089, "time_profile/old_logps": 33.677048296667635, "time_profile/ref_logps": 33.734667072072625, "time_profile/reward": 0.00466107577085495, "time_profile/text_rollout": 454.67956179752946, "train/gen_step": 3264.0, "und/advantages_abs_mean": 0.498779296875, "und/advantages_max": 1.8125, "und/advantages_mean": 0.0, "und/advantages_min": -2.1875, "und/advantages_std": 0.669204831123352, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.6875, "und/completion/min_length": 251.0, "und/frac_reward_zero_std": 0.015625, "und/kl": 0.03883673060772708, "und/loss": 0.001454857557376954, "und/prompt/max_length": 55.0, "und/prompt/mean_length": 52.0, "und/prompt/min_length": 49.0, "und/reward": 0.9443359375, "und/reward_std": 1.075581669807434 }, { "epoch": 0.6661329063250601, "grad_norm": 0.038634829223155975, "learning_rate": 1e-05, "loss": 0.0012, "step": 52, "time_profile/curr_logprobs_and_update": 0.2613996758009307, "train/gen_step": 3328.0, "und/clip_ratio/high_mean": 0.0003874025678669568, "und/clip_ratio/low_mean": 0.00019797183631453663, "und/clip_ratio/region_mean": 0.0005853744041814934, "und/kl": 0.03153260219551157, "und/loss": 0.0016808376690278237 }, { "epoch": 0.6789431545236189, "grad_norm": 0.05004860833287239, "learning_rate": 1e-05, "loss": 0.0109, "rewards/correctness_reward_func/mean": 0.15625, "rewards/correctness_reward_func/std": 0.5409794449806213, "rewards/strict_format_reward_func/mean": 0.1953125, "rewards/strict_format_reward_func/std": 0.24587368965148926, "step": 53, "time_profile/curr_logprobs_and_update": 0.26163243876362685, "time_profile/old_logps": 33.65926602482796, "time_profile/ref_logps": 33.75773391965777, "time_profile/reward": 0.004499710164964199, "time_profile/text_rollout": 516.6223611980677, "train/gen_step": 3392.0, "und/advantages_abs_mean": 0.2509765625, "und/advantages_max": 2.0625, "und/advantages_mean": 0.0, "und/advantages_min": -1.125, "und/advantages_std": 0.5241244435310364, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.7734375, "und/completion/min_length": 254.0, "und/frac_reward_zero_std": 0.5, "und/kl": 0.2721175526821753, "und/loss": 0.054259198752447446, "und/prompt/max_length": 313.0, "und/prompt/mean_length": 234.5, "und/prompt/min_length": 156.0, "und/reward": 0.322265625, "und/reward_std": 0.6323413252830505 }, { "epoch": 0.6917534027221778, "grad_norm": 0.035311710089445114, "learning_rate": 1e-05, "loss": 0.0031, "step": 54, "time_profile/curr_logprobs_and_update": 0.26147235184907913, "train/gen_step": 3456.0, "und/clip_ratio/high_mean": 0.00019807044009212404, "und/clip_ratio/low_mean": 9.089218292501755e-05, "und/clip_ratio/region_mean": 0.0002889626230171416, "und/kl": 0.0721228092443198, "und/loss": 0.002958318164701268 }, { "epoch": 0.7045636509207366, "grad_norm": 0.04633024334907532, "learning_rate": 1e-05, "loss": 0.0015, "rewards/correctness_reward_func/mean": 0.90625, "rewards/correctness_reward_func/std": 1.003466248512268, "rewards/strict_format_reward_func/mean": 0.484375, "rewards/strict_format_reward_func/std": 0.08768405020236969, "step": 55, "time_profile/curr_logprobs_and_update": 0.2614114769385196, "time_profile/old_logps": 33.599201523698866, "time_profile/ref_logps": 33.73441112972796, "time_profile/reward": 0.004696068353950977, "time_profile/text_rollout": 453.6825319575146, "train/gen_step": 3520.0, "und/advantages_abs_mean": 0.341064453125, "und/advantages_max": 1.875, "und/advantages_mean": 0.0, "und/advantages_min": -2.1875, "und/advantages_std": 0.6085155010223389, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.72265625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.40625, "und/kl": 0.03839961695484817, "und/loss": 0.0013346272653507185, "und/prompt/max_length": 47.0, "und/prompt/mean_length": 45.5, "und/prompt/min_length": 44.0, "und/reward": 1.3662109375, "und/reward_std": 1.0347063541412354 }, { "epoch": 0.7173738991192954, "grad_norm": 0.047499582171440125, "learning_rate": 1e-05, "loss": 0.0034, "step": 56, "time_profile/curr_logprobs_and_update": 0.2612421384692425, "train/gen_step": 3584.0, "und/clip_ratio/high_mean": 0.0003667974451673217, "und/clip_ratio/low_mean": 0.00013469014447764494, "und/clip_ratio/region_mean": 0.0005014875860069878, "und/kl": 0.08546131956973113, "und/loss": 0.00697715762282769 }, { "epoch": 0.7301841473178543, "grad_norm": 0.014783147722482681, "learning_rate": 1e-05, "loss": 0.004, "rewards/correctness_reward_func/mean": 0.09375, "rewards/correctness_reward_func/std": 0.42608407139778137, "rewards/strict_format_reward_func/mean": 0.296875, "rewards/strict_format_reward_func/std": 0.24750742316246033, "step": 57, "time_profile/curr_logprobs_and_update": 0.2614298419211991, "time_profile/old_logps": 33.64691087603569, "time_profile/ref_logps": 33.739184183999896, "time_profile/reward": 0.004393537528812885, "time_profile/text_rollout": 454.8285795925185, "train/gen_step": 3648.0, "und/advantages_abs_mean": 0.20654296875, "und/advantages_max": 1.8125, "und/advantages_mean": 0.0, "und/advantages_min": -0.9375, "und/advantages_std": 0.3522537350654602, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.716796875, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.21875, "und/kl": 0.10082034709921572, "und/loss": 0.0015968736425975294, "und/prompt/max_length": 56.0, "und/prompt/mean_length": 51.0, "und/prompt/min_length": 46.0, "und/reward": 0.3818359375, "und/reward_std": 0.42684176564216614 }, { "epoch": 0.7429943955164131, "grad_norm": 0.014119211584329605, "learning_rate": 1e-05, "loss": 0.0117, "step": 58, "time_profile/curr_logprobs_and_update": 0.26134903807542287, "train/gen_step": 3712.0, "und/clip_ratio/high_mean": 4.882812572759576e-05, "und/clip_ratio/low_mean": 0.00023804241573088802, "und/clip_ratio/region_mean": 0.0002868705414584838, "und/kl": 0.2931558708951343, "und/loss": 0.001128198406345149 }, { "epoch": 0.755804643714972, "grad_norm": 0.03397494554519653, "learning_rate": 1e-05, "loss": 0.0028, "rewards/correctness_reward_func/mean": 0.15625, "rewards/correctness_reward_func/std": 0.5409794449806213, "rewards/strict_format_reward_func/mean": 0.484375, "rewards/strict_format_reward_func/std": 0.08768405020236969, "step": 59, "time_profile/curr_logprobs_and_update": 0.2614648619637592, "time_profile/old_logps": 33.627349744550884, "time_profile/ref_logps": 33.70189340412617, "time_profile/reward": 0.00450173020362854, "time_profile/text_rollout": 454.01288072485477, "train/gen_step": 3776.0, "und/advantages_abs_mean": 0.373046875, "und/advantages_max": 1.8125, "und/advantages_mean": 0.0, "und/advantages_min": -1.6875, "und/advantages_std": 0.6158081293106079, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.609375, "und/completion/min_length": 251.0, "und/frac_reward_zero_std": 0.453125, "und/kl": 0.07084466355445329, "und/loss": 0.002370536963894665, "und/prompt/max_length": 47.0, "und/prompt/mean_length": 46.0, "und/prompt/min_length": 45.0, "und/reward": 0.7958984375, "und/reward_std": 0.739467978477478 }, { "epoch": 0.7686148919135308, "grad_norm": 0.046682216227054596, "learning_rate": 1e-05, "loss": 0.0041, "step": 60, "time_profile/curr_logprobs_and_update": 0.26129540299007203, "train/gen_step": 3840.0, "und/clip_ratio/high_mean": 0.000178795224201167, "und/clip_ratio/low_mean": 0.0001412087440257892, "und/clip_ratio/region_mean": 0.0003200039645889774, "und/kl": 0.10260857833782211, "und/loss": 0.0027817148465487662 }, { "epoch": 0.7814251401120896, "grad_norm": 0.014530722051858902, "learning_rate": 1e-05, "loss": 0.0184, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.1953125, "rewards/strict_format_reward_func/std": 0.24587368965148926, "step": 61, "time_profile/curr_logprobs_and_update": 0.2614762183802668, "time_profile/old_logps": 33.65403878968209, "time_profile/ref_logps": 33.731373984366655, "time_profile/reward": 0.004584863781929016, "time_profile/text_rollout": 472.9927922273055, "train/gen_step": 3904.0, "und/advantages_abs_mean": 0.203125, "und/advantages_max": 1.5, "und/advantages_mean": 0.0, "und/advantages_min": -0.5, "und/advantages_std": 0.23512499034404755, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.74609375, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.015625, "und/kl": 0.4595181476797734, "und/loss": 0.0014087316812947392, "und/prompt/max_length": 162.0, "und/prompt/mean_length": 106.0, "und/prompt/min_length": 50.0, "und/reward": 0.20703125, "und/reward_std": 0.25815349817276 }, { "epoch": 0.7942353883106485, "grad_norm": 0.014059546403586864, "learning_rate": 1e-05, "loss": 0.0132, "step": 62, "time_profile/curr_logprobs_and_update": 0.2614812095707748, "train/gen_step": 3968.0, "und/clip_ratio/high_mean": 0.0003522909028106369, "und/clip_ratio/low_mean": 0.00013392856999416836, "und/clip_ratio/region_mean": 0.00048621947280480526, "und/kl": 0.32975731499755057, "und/loss": 0.0033024766016751528 }, { "epoch": 0.8070456365092074, "grad_norm": 0.013789220713078976, "learning_rate": 1e-05, "loss": 0.0013, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.2890625, "rewards/strict_format_reward_func/std": 0.24888142943382263, "step": 63, "time_profile/curr_logprobs_and_update": 0.26154088985640556, "time_profile/old_logps": 33.65780791267753, "time_profile/ref_logps": 33.74037241283804, "time_profile/reward": 0.003031497821211815, "time_profile/text_rollout": 454.97051773872226, "train/gen_step": 4032.0, "und/advantages_abs_mean": 0.210693359375, "und/advantages_max": 0.375, "und/advantages_mean": 0.0, "und/advantages_min": -0.4375, "und/advantages_std": 0.22973118722438812, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.74609375, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.046875, "und/kl": 0.033402759159798734, "und/loss": 0.0011463420676136593, "und/prompt/max_length": 52.0, "und/prompt/mean_length": 50.5, "und/prompt/min_length": 49.0, "und/reward": 0.2783203125, "und/reward_std": 0.24863366782665253 }, { "epoch": 0.8198558847077662, "grad_norm": 0.015820274129509926, "learning_rate": 1e-05, "loss": 0.0016, "step": 64, "time_profile/curr_logprobs_and_update": 0.2614853223931277, "train/gen_step": 4096.0, "und/clip_ratio/high_mean": 0.0008117913530441001, "und/clip_ratio/low_mean": 0.00011440205707913265, "und/clip_ratio/region_mean": 0.0009261934137612116, "und/kl": 0.03982456024095882, "und/loss": 0.0010868344670598162 }, { "epoch": 0.8326661329063251, "grad_norm": 0.041560448706150055, "learning_rate": 1e-05, "loss": 0.0025, "rewards/correctness_reward_func/mean": 0.28125, "rewards/correctness_reward_func/std": 0.7007648944854736, "rewards/strict_format_reward_func/mean": 0.21875, "rewards/strict_format_reward_func/std": 0.25, "step": 65, "time_profile/curr_logprobs_and_update": 0.26148034693324007, "time_profile/old_logps": 33.658358512446284, "time_profile/ref_logps": 33.74039072729647, "time_profile/reward": 0.0044120242819190025, "time_profile/text_rollout": 473.54834595508873, "train/gen_step": 4160.0, "und/advantages_abs_mean": 0.516845703125, "und/advantages_max": 2.0625, "und/advantages_mean": 0.0, "und/advantages_min": -1.6875, "und/advantages_std": 0.7237750291824341, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.75390625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.03125, "und/kl": 0.061325576913077384, "und/loss": 0.0018601922784000635, "und/prompt/max_length": 156.0, "und/prompt/mean_length": 109.5, "und/prompt/min_length": 63.0, "und/reward": 0.5234375, "und/reward_std": 0.8563321828842163 }, { "epoch": 0.8454763811048839, "grad_norm": 0.040835775434970856, "learning_rate": 1e-05, "loss": 0.0047, "step": 66, "time_profile/curr_logprobs_and_update": 0.2613908527127933, "train/gen_step": 4224.0, "und/clip_ratio/high_mean": 0.0002187356585636735, "und/clip_ratio/low_mean": 0.0002555298342485912, "und/clip_ratio/region_mean": 0.0004742654928122647, "und/kl": 0.11611133666156093, "und/loss": 0.0033431433839723468 }, { "epoch": 0.8582866293034428, "grad_norm": 0.04072283208370209, "learning_rate": 1e-05, "loss": 0.0042, "rewards/correctness_reward_func/mean": 0.875, "rewards/correctness_reward_func/std": 1.0, "rewards/strict_format_reward_func/mean": 0.3828125, "rewards/strict_format_reward_func/std": 0.21347814798355103, "step": 67, "time_profile/curr_logprobs_and_update": 0.26157006529683713, "time_profile/old_logps": 33.639826339669526, "time_profile/ref_logps": 33.74620936065912, "time_profile/reward": 0.004473298788070679, "time_profile/text_rollout": 454.7873819489032, "train/gen_step": 4288.0, "und/advantages_abs_mean": 0.275390625, "und/advantages_max": 0.9375, "und/advantages_mean": 0.0, "und/advantages_min": -2.1875, "und/advantages_std": 0.4825730323791504, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.744140625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.203125, "und/kl": 0.10618040141707752, "und/loss": 0.002721756522078067, "und/prompt/max_length": 53.0, "und/prompt/mean_length": 49.5, "und/prompt/min_length": 46.0, "und/reward": 1.283203125, "und/reward_std": 1.1394332647323608 }, { "epoch": 0.8710968775020016, "grad_norm": 0.023537058383226395, "learning_rate": 1e-05, "loss": 0.0019, "step": 68, "time_profile/curr_logprobs_and_update": 0.2615145814634161, "train/gen_step": 4352.0, "und/clip_ratio/high_mean": 0.0006931504431122448, "und/clip_ratio/low_mean": 0.000173399384948425, "und/clip_ratio/region_mean": 0.000866549824422691, "und/kl": 0.05065767532505561, "und/loss": 0.0008179922006092966 }, { "epoch": 0.8839071257005604, "grad_norm": 0.00545235862955451, "learning_rate": 1e-05, "loss": 0.0019, "rewards/correctness_reward_func/mean": 0.03125, "rewards/correctness_reward_func/std": 0.25, "rewards/strict_format_reward_func/mean": 0.0703125, "rewards/strict_format_reward_func/std": 0.1751912236213684, "step": 69, "time_profile/curr_logprobs_and_update": 0.261740165471565, "time_profile/old_logps": 33.70919596590102, "time_profile/ref_logps": 33.77206904441118, "time_profile/reward": 0.004499014467000961, "time_profile/text_rollout": 501.05089360289276, "train/gen_step": 4416.0, "und/advantages_abs_mean": 0.09765625, "und/advantages_max": 2.125, "und/advantages_mean": 0.0, "und/advantages_min": -0.4375, "und/advantages_std": 0.18104934692382812, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.92578125, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.53125, "und/kl": 0.04702442421694286, "und/loss": 0.0023319427573369467, "und/prompt/max_length": 313.0, "und/prompt/mean_length": 185.5, "und/prompt/min_length": 58.0, "und/reward": 0.083984375, "und/reward_std": 0.21164102852344513 }, { "epoch": 0.8967173738991193, "grad_norm": 0.015413722954690456, "learning_rate": 1e-05, "loss": 0.0029, "step": 70, "time_profile/curr_logprobs_and_update": 0.26160615020489786, "train/gen_step": 4480.0, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 9.790100375539623e-05, "und/clip_ratio/region_mean": 9.790100375539623e-05, "und/kl": 0.0741199756594142, "und/loss": 0.002338052549376357 }, { "epoch": 0.9095276220976781, "grad_norm": 0.013947661966085434, "learning_rate": 1e-05, "loss": 0.0017, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.09375, "rewards/strict_format_reward_func/std": 0.19669894874095917, "step": 71, "time_profile/curr_logprobs_and_update": 0.26171521554351784, "time_profile/old_logps": 33.66377367079258, "time_profile/ref_logps": 33.751427008770406, "time_profile/reward": 0.004377297125756741, "time_profile/text_rollout": 500.3317451989278, "train/gen_step": 4544.0, "und/advantages_abs_mean": 0.107666015625, "und/advantages_max": 0.4375, "und/advantages_mean": 0.0, "und/advantages_min": -0.4375, "und/advantages_std": 0.16422295570373535, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.857421875, "und/completion/min_length": 254.0, "und/frac_reward_zero_std": 0.5, "und/kl": 0.04332471443194663, "und/loss": 0.0016425758730918005, "und/prompt/max_length": 313.0, "und/prompt/mean_length": 184.0, "und/prompt/min_length": 55.0, "und/reward": 0.1005859375, "und/reward_std": 0.2006341516971588 }, { "epoch": 0.922337870296237, "grad_norm": 0.012040519155561924, "learning_rate": 1e-05, "loss": 0.0016, "step": 72, "time_profile/curr_logprobs_and_update": 0.2617163411487127, "train/gen_step": 4608.0, "und/clip_ratio/high_mean": 0.00010259100599796511, "und/clip_ratio/low_mean": 0.00037950047408230603, "und/clip_ratio/region_mean": 0.00048209148008027114, "und/kl": 0.03941097212373279, "und/loss": 0.0017510293993154846 }, { "epoch": 0.9351481184947958, "grad_norm": 0.025353815406560898, "learning_rate": 1e-05, "loss": 0.0082, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.234375, "rewards/strict_format_reward_func/std": 0.2514837086200714, "step": 73, "time_profile/curr_logprobs_and_update": 0.26159169392485637, "time_profile/old_logps": 33.65948408655822, "time_profile/ref_logps": 33.75273651909083, "time_profile/reward": 0.004401316866278648, "time_profile/text_rollout": 487.3236538115889, "train/gen_step": 4672.0, "und/advantages_abs_mean": 0.070556640625, "und/advantages_max": 2.1875, "und/advantages_mean": 0.0, "und/advantages_min": -0.6875, "und/advantages_std": 0.2424243539571762, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.818359375, "und/completion/min_length": 254.0, "und/frac_reward_zero_std": 0.703125, "und/kl": 0.20393660698027816, "und/loss": 0.004929899203489185, "und/prompt/max_length": 247.0, "und/prompt/mean_length": 148.0, "und/prompt/min_length": 49.0, "und/reward": 0.2666015625, "und/reward_std": 0.35626593232154846 }, { "epoch": 0.9479583666933546, "grad_norm": 0.013712815009057522, "learning_rate": 1e-05, "loss": 0.0149, "step": 74, "time_profile/curr_logprobs_and_update": 0.26143009605584666, "train/gen_step": 4736.0, "und/clip_ratio/high_mean": 0.00025846589414868504, "und/clip_ratio/low_mean": 5.278716344037093e-05, "und/clip_ratio/region_mean": 0.00031125305758905597, "und/kl": 0.36927700173691846, "und/loss": 0.0025039048924071494 }, { "epoch": 0.9607686148919136, "grad_norm": 0.009972140192985535, "learning_rate": 1e-05, "loss": 0.0136, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.21875, "rewards/strict_format_reward_func/std": 0.25, "step": 75, "time_profile/curr_logprobs_and_update": 0.26166287070373073, "time_profile/old_logps": 33.67625562660396, "time_profile/ref_logps": 33.7687153192237, "time_profile/reward": 0.0043565016239881516, "time_profile/text_rollout": 473.3151173014194, "train/gen_step": 4800.0, "und/advantages_abs_mean": 0.205810546875, "und/advantages_max": 0.4375, "und/advantages_mean": 0.0, "und/advantages_min": -0.375, "und/advantages_std": 0.22705358266830444, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.791015625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.015625, "und/kl": 0.33891669298463967, "und/loss": 0.0008663814514875412, "und/prompt/max_length": 162.0, "und/prompt/mean_length": 107.0, "und/prompt/min_length": 52.0, "und/reward": 0.1904296875, "und/reward_std": 0.24303650856018066 }, { "epoch": 0.9735788630904724, "grad_norm": 0.009620222263038158, "learning_rate": 1e-05, "loss": 0.0059, "step": 76, "time_profile/curr_logprobs_and_update": 0.2629899470921373, "train/gen_step": 4864.0, "und/clip_ratio/high_mean": 0.00046749321700190194, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.00046749321700190194, "und/kl": 0.1477012149625807, "und/loss": 0.0013083760859444737 }, { "epoch": 0.9863891112890312, "grad_norm": 0.010119947604835033, "learning_rate": 1e-05, "loss": 0.0234, "rewards/correctness_reward_func/mean": 0.03125, "rewards/correctness_reward_func/std": 0.25, "rewards/strict_format_reward_func/mean": 0.2578125, "rewards/strict_format_reward_func/std": 0.25185325741767883, "step": 77, "time_profile/curr_logprobs_and_update": 0.2616106412606314, "time_profile/old_logps": 33.68558882549405, "time_profile/ref_logps": 33.77748038619757, "time_profile/reward": 0.0041413139551877975, "time_profile/text_rollout": 473.13247388228774, "train/gen_step": 4928.0, "und/advantages_abs_mean": 0.216064453125, "und/advantages_max": 2.125, "und/advantages_mean": 0.0, "und/advantages_min": -0.5, "und/advantages_std": 0.2934376895427704, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.806640625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.0625, "und/kl": 0.5857936546090059, "und/loss": 0.0010301140270740916, "und/prompt/max_length": 162.0, "und/prompt/mean_length": 106.5, "und/prompt/min_length": 51.0, "und/reward": 0.220703125, "und/reward_std": 0.31764960289001465 }, { "epoch": 0.9991993594875901, "grad_norm": 0.012176643125712872, "learning_rate": 1e-05, "loss": 0.0303, "step": 78, "time_profile/curr_logprobs_and_update": 0.2613839999830816, "train/gen_step": 4992.0, "und/clip_ratio/high_mean": 0.00017109220061684027, "und/clip_ratio/low_mean": 0.000871925825776998, "und/clip_ratio/region_mean": 0.0010430180263938382, "und/kl": 0.7574968582484871, "und/loss": 0.0010476876539087243 }, { "epoch": 1.0, "grad_norm": 0.010883414186537266, "learning_rate": 1e-05, "loss": 0.001, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.2890625, "rewards/strict_format_reward_func/std": 0.24888142943382263, "step": 79, "time_profile/curr_logprobs_and_update": 0.26290554786100984, "time_profile/old_logps": 33.66828224621713, "time_profile/ref_logps": 33.729235981591046, "time_profile/reward": 0.004360824823379517, "time_profile/text_rollout": 456.6069682110101, "train/gen_step": 4996.0, "und/advantages_abs_mean": 0.123779296875, "und/advantages_max": 0.4375, "und/advantages_mean": 0.0, "und/advantages_min": -0.4375, "und/advantages_std": 0.17608344554901123, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.671875, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.3125, "und/kl": 0.018394303042441607, "und/loss": 0.12604861333966255, "und/prompt/max_length": 64.0, "und/prompt/mean_length": 57.0, "und/prompt/min_length": 50.0, "und/reward": 0.3056640625, "und/reward_std": 0.2439626157283783 }, { "epoch": 1.0128102481985588, "grad_norm": 0.011202736757695675, "learning_rate": 1e-05, "loss": 0.0026, "step": 80, "time_profile/curr_logprobs_and_update": 0.26144698687130585, "train/gen_step": 5060.0, "und/clip_ratio/high_mean": 0.00020480685998336412, "und/clip_ratio/low_mean": 0.0010958394632325508, "und/clip_ratio/region_mean": 0.001300646319577936, "und/kl": 0.06380279999575578, "und/loss": 0.001128781703300774 }, { "epoch": 1.0256204963971176, "grad_norm": 0.011037485674023628, "learning_rate": 1e-05, "loss": -0.0013, "rewards/correctness_reward_func/mean": 0.96875, "rewards/correctness_reward_func/std": 1.0074130296707153, "rewards/strict_format_reward_func/mean": 0.296875, "rewards/strict_format_reward_func/std": 0.24750742316246033, "step": 81, "time_profile/curr_logprobs_and_update": 0.2614856250002049, "time_profile/old_logps": 33.64920091070235, "time_profile/ref_logps": 33.70459945779294, "time_profile/reward": 0.004259305074810982, "time_profile/text_rollout": 454.86890551261604, "train/gen_step": 5124.0, "und/advantages_abs_mean": 0.21484375, "und/advantages_max": 0.9375, "und/advantages_mean": 0.0, "und/advantages_min": -2.1875, "und/advantages_std": 0.4081484079360962, "und/clip_ratio/high_mean": 0.0003931849460059311, "und/clip_ratio/low_mean": 0.0013027411841903813, "und/clip_ratio/region_mean": 0.0016959261301963124, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.779296875, "und/completion/min_length": 222.0, "und/frac_reward_zero_std": 0.328125, "und/kl": 0.06465533758455422, "und/loss": -0.005637970802126802, "und/prompt/max_length": 58.0, "und/prompt/mean_length": 51.0, "und/prompt/min_length": 44.0, "und/reward": 1.2578125, "und/reward_std": 1.1828593015670776 }, { "epoch": 1.0384307445956766, "grad_norm": 0.012660922482609749, "learning_rate": 1e-05, "loss": 0.0052, "step": 82, "time_profile/curr_logprobs_and_update": 0.2613754039048217, "train/gen_step": 5188.0, "und/clip_ratio/high_mean": 0.00011789133350248449, "und/clip_ratio/low_mean": 0.0001570866115798708, "und/clip_ratio/region_mean": 0.0002749779450823553, "und/kl": 0.13132090101134963, "und/loss": 0.0017235139853255532 }, { "epoch": 1.0512409927942354, "grad_norm": 0.04931657388806343, "learning_rate": 1e-05, "loss": 0.0348, "rewards/correctness_reward_func/mean": 0.3125, "rewards/correctness_reward_func/std": 0.7319250702857971, "rewards/strict_format_reward_func/mean": 0.390625, "rewards/strict_format_reward_func/std": 0.2083333432674408, "step": 83, "time_profile/curr_logprobs_and_update": 0.2614861790498253, "time_profile/old_logps": 33.64623137563467, "time_profile/ref_logps": 33.70282917190343, "time_profile/reward": 0.004459716379642487, "time_profile/text_rollout": 454.369918580167, "train/gen_step": 5252.0, "und/advantages_abs_mean": 0.470703125, "und/advantages_max": 1.75, "und/advantages_mean": 0.0, "und/advantages_min": -1.4375, "und/advantages_std": 0.6308677792549133, "und/clip_ratio/high_mean": 0.00037126274764887057, "und/clip_ratio/low_mean": 0.0008432684480794705, "und/clip_ratio/region_mean": 0.0012145311957283411, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.71875, "und/completion/min_length": 220.0, "und/frac_reward_zero_std": 0.046875, "und/kl": 0.8189952824468492, "und/loss": 0.002741719154414568, "und/prompt/max_length": 51.0, "und/prompt/mean_length": 47.5, "und/prompt/min_length": 44.0, "und/reward": 0.689453125, "und/reward_std": 0.7827932238578796 }, { "epoch": 1.0640512409927942, "grad_norm": 0.0498473197221756, "learning_rate": 1e-05, "loss": 0.0023, "step": 84, "time_profile/curr_logprobs_and_update": 0.26133355250931345, "train/gen_step": 5316.0, "und/clip_ratio/high_mean": 0.0004264039162080735, "und/clip_ratio/low_mean": 0.0006009412281855475, "und/clip_ratio/region_mean": 0.0010273451480315998, "und/kl": 0.060061977404984646, "und/loss": 0.002192927919622889 }, { "epoch": 1.076861489191353, "grad_norm": 0.037328220903873444, "learning_rate": 1e-05, "loss": 0.0067, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.4921875, "rewards/strict_format_reward_func/std": 0.0625, "step": 85, "time_profile/curr_logprobs_and_update": 0.26134839945007116, "time_profile/old_logps": 33.615224280394614, "time_profile/ref_logps": 33.697587930597365, "time_profile/reward": 0.004523534327745438, "time_profile/text_rollout": 454.07083107624203, "train/gen_step": 5380.0, "und/advantages_abs_mean": 0.103759765625, "und/advantages_max": 1.8125, "und/advantages_mean": 0.0, "und/advantages_min": -0.6875, "und/advantages_std": 0.28764939308166504, "und/clip_ratio/high_mean": 0.0008836273773340508, "und/clip_ratio/low_mean": 0.001003391091217054, "und/clip_ratio/region_mean": 0.0018870184685511049, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.63671875, "und/completion/min_length": 254.0, "und/frac_reward_zero_std": 0.625, "und/kl": 0.05825820894096978, "und/loss": 0.002590392931324459, "und/prompt/max_length": 47.0, "und/prompt/mean_length": 45.0, "und/prompt/min_length": 43.0, "und/reward": 0.5224609375, "und/reward_std": 0.30963554978370667 }, { "epoch": 1.0896717373899119, "grad_norm": 0.029800761491060257, "learning_rate": 1e-05, "loss": 0.0081, "step": 86, "time_profile/curr_logprobs_and_update": 0.2612544225266902, "train/gen_step": 5444.0, "und/clip_ratio/high_mean": 9.385726662003435e-05, "und/clip_ratio/low_mean": 9.646531907492317e-05, "und/clip_ratio/region_mean": 0.00019032258569495752, "und/kl": 0.20097231600084342, "und/loss": 0.0032604012124011206 }, { "epoch": 1.1024819855884709, "grad_norm": 0.03139861673116684, "learning_rate": 1e-05, "loss": 0.0314, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.09375, "rewards/strict_format_reward_func/std": 0.19669894874095917, "step": 87, "time_profile/curr_logprobs_and_update": 0.26139444966975134, "time_profile/old_logps": 33.700514083728194, "time_profile/ref_logps": 33.78526249341667, "time_profile/reward": 0.004218182526528835, "time_profile/text_rollout": 518.5401958124712, "train/gen_step": 5508.0, "und/advantages_abs_mean": 0.0966796875, "und/advantages_max": 0.4375, "und/advantages_mean": 0.0, "und/advantages_min": -0.3125, "und/advantages_std": 0.1556188315153122, "und/clip_ratio/high_mean": 0.00026991102640749887, "und/clip_ratio/low_mean": 0.0002991936562466435, "und/clip_ratio/region_mean": 0.0005691046826541424, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.720703125, "und/completion/min_length": 159.0, "und/frac_reward_zero_std": 0.5, "und/kl": 0.7756888958901982, "und/loss": 0.01109763615556858, "und/prompt/max_length": 313.0, "und/prompt/mean_length": 237.5, "und/prompt/min_length": 162.0, "und/reward": 0.080078125, "und/reward_std": 0.18355479836463928 }, { "epoch": 1.1152922337870297, "grad_norm": 0.007225779816508293, "learning_rate": 1e-05, "loss": 0.0023, "step": 88, "time_profile/curr_logprobs_and_update": 0.2618225736368913, "train/gen_step": 5572.0, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 0.0, "und/kl": 0.05710822065520915, "und/loss": 0.0013941282181804127 }, { "epoch": 1.1281024819855885, "grad_norm": 0.006840178277343512, "learning_rate": 1e-05, "loss": 0.0449, "rewards/correctness_reward_func/mean": 0.28125, "rewards/correctness_reward_func/std": 0.7007648944854736, "rewards/strict_format_reward_func/mean": 0.2109375, "rewards/strict_format_reward_func/std": 0.24888142943382263, "step": 89, "time_profile/curr_logprobs_and_update": 0.2619360370154027, "time_profile/old_logps": 33.66616447176784, "time_profile/ref_logps": 33.762949594296515, "time_profile/reward": 0.004555100575089455, "time_profile/text_rollout": 499.7487023388967, "train/gen_step": 5636.0, "und/advantages_abs_mean": 0.455078125, "und/advantages_max": 1.8125, "und/advantages_mean": 0.0, "und/advantages_min": -1.9375, "und/advantages_std": 0.6984922289848328, "und/clip_ratio/high_mean": 0.00011725750300684012, "und/clip_ratio/low_mean": 0.0001583614903211128, "und/clip_ratio/region_mean": 0.0002756189933279529, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.771484375, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.46875, "und/kl": 1.232714807476441, "und/loss": -0.006696377290722921, "und/prompt/max_length": 313.0, "und/prompt/mean_length": 180.0, "und/prompt/min_length": 47.0, "und/reward": 0.734375, "und/reward_std": 1.0485988855361938 }, { "epoch": 1.1409127301841473, "grad_norm": 0.006123845931142569, "learning_rate": 1e-05, "loss": 0.004, "step": 90, "time_profile/curr_logprobs_and_update": 0.26148606633068994, "train/gen_step": 5700.0, "und/clip_ratio/high_mean": 0.0, "und/clip_ratio/low_mean": 0.00026530873219599016, "und/clip_ratio/region_mean": 0.00026530873219599016, "und/kl": 0.09945900924503803, "und/loss": 0.004926550779600802 }, { "epoch": 1.153722978382706, "grad_norm": 0.009329823777079582, "learning_rate": 1e-05, "loss": 0.0647, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.3046875, "rewards/strict_format_reward_func/std": 0.24587368965148926, "step": 91, "time_profile/curr_logprobs_and_update": 0.26152575001469813, "time_profile/old_logps": 33.65729128010571, "time_profile/ref_logps": 33.73677711561322, "time_profile/reward": 0.004335583187639713, "time_profile/text_rollout": 472.53139427211136, "train/gen_step": 5764.0, "und/advantages_abs_mean": 0.19921875, "und/advantages_max": 1.8125, "und/advantages_mean": 0.0, "und/advantages_min": -0.6875, "und/advantages_std": 0.35682472586631775, "und/clip_ratio/high_mean": 0.0003500454331515357, "und/clip_ratio/low_mean": 0.00026731862817541696, "und/clip_ratio/region_mean": 0.0006173640649649315, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.72265625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.265625, "und/kl": 1.6728466484637465, "und/loss": 0.006491393034089299, "und/prompt/max_length": 162.0, "und/prompt/mean_length": 104.0, "und/prompt/min_length": 46.0, "und/reward": 0.359375, "und/reward_std": 0.4551832377910614 }, { "epoch": 1.1665332265812651, "grad_norm": 3.300291061401367, "learning_rate": 1e-05, "loss": 0.021, "step": 92, "time_profile/curr_logprobs_and_update": 0.2615345583180897, "train/gen_step": 5828.0, "und/clip_ratio/high_mean": 0.00015658686243114062, "und/clip_ratio/low_mean": 0.000412728051742306, "und/clip_ratio/region_mean": 0.0005693149105354678, "und/kl": 0.5270771795330802, "und/loss": 0.021786472902746823 }, { "epoch": 1.179343474779824, "grad_norm": 0.04499354213476181, "learning_rate": 1e-05, "loss": 0.0101, "rewards/correctness_reward_func/mean": 0.78125, "rewards/correctness_reward_func/std": 0.983494758605957, "rewards/strict_format_reward_func/mean": 0.359375, "rewards/strict_format_reward_func/std": 0.22658175230026245, "step": 93, "time_profile/curr_logprobs_and_update": 0.2617357307317434, "time_profile/old_logps": 33.68263782840222, "time_profile/ref_logps": 33.77571779489517, "time_profile/reward": 0.004069678485393524, "time_profile/text_rollout": 456.24606967996806, "train/gen_step": 5892.0, "und/advantages_abs_mean": 0.408447265625, "und/advantages_max": 0.875, "und/advantages_mean": 0.0, "und/advantages_min": -2.1875, "und/advantages_std": 0.5925272703170776, "und/clip_ratio/high_mean": 0.0006427463522413746, "und/clip_ratio/low_mean": 0.0003597673749027308, "und/clip_ratio/region_mean": 0.0010025137307820842, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.634765625, "und/completion/min_length": 173.0, "und/frac_reward_zero_std": 0.078125, "und/kl": 0.16817912326951046, "und/loss": 0.0022755718414586568, "und/prompt/max_length": 65.0, "und/prompt/mean_length": 54.0, "und/prompt/min_length": 43.0, "und/reward": 1.1474609375, "und/reward_std": 1.1275041103363037 }, { "epoch": 1.1921537229783827, "grad_norm": 0.06060933321714401, "learning_rate": 1e-05, "loss": 0.0042, "step": 94, "time_profile/curr_logprobs_and_update": 0.26131482362688985, "train/gen_step": 5956.0, "und/clip_ratio/high_mean": 0.0019653168055810966, "und/clip_ratio/low_mean": 0.00016475993470521644, "und/clip_ratio/region_mean": 0.002130076743924292, "und/kl": 0.10070258447376546, "und/loss": 0.005654156702803448 }, { "epoch": 1.2049639711769415, "grad_norm": 0.03084016963839531, "learning_rate": 1e-05, "loss": 0.0413, "rewards/correctness_reward_func/mean": 0.03125, "rewards/correctness_reward_func/std": 0.25, "rewards/strict_format_reward_func/mean": 0.265625, "rewards/strict_format_reward_func/std": 0.2514837086200714, "step": 95, "time_profile/curr_logprobs_and_update": 0.26137867687793914, "time_profile/old_logps": 33.66459304280579, "time_profile/ref_logps": 33.77252321410924, "time_profile/reward": 0.004362174309790134, "time_profile/text_rollout": 486.8325478192419, "train/gen_step": 6020.0, "und/advantages_abs_mean": 0.0693359375, "und/advantages_max": 2.1875, "und/advantages_mean": 0.0, "und/advantages_min": -0.625, "und/advantages_std": 0.27044573426246643, "und/clip_ratio/high_mean": 0.0046464357401418965, "und/clip_ratio/low_mean": 0.0008429612207692116, "und/clip_ratio/region_mean": 0.005489396939083235, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.8828125, "und/completion/min_length": 254.0, "und/frac_reward_zero_std": 0.8125, "und/kl": 1.0781144749053055, "und/loss": 0.00424630767565759, "und/prompt/max_length": 247.0, "und/prompt/mean_length": 145.0, "und/prompt/min_length": 43.0, "und/reward": 0.287109375, "und/reward_std": 0.3735242784023285 }, { "epoch": 1.2177742193755003, "grad_norm": 0.008688083849847317, "learning_rate": 1e-05, "loss": 0.0111, "step": 96, "time_profile/curr_logprobs_and_update": 0.2616057716368232, "train/gen_step": 6084.0, "und/clip_ratio/high_mean": 0.00013542917804443277, "und/clip_ratio/low_mean": 4.882812572759576e-05, "und/clip_ratio/region_mean": 0.00018425730377202854, "und/kl": 0.27596294990507886, "und/loss": 0.002798294794956746 }, { "epoch": 1.2305844675740594, "grad_norm": 0.003840341465547681, "learning_rate": 1e-05, "loss": 0.0055, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0546875, "rewards/strict_format_reward_func/std": 0.15728822350502014, "step": 97, "time_profile/curr_logprobs_and_update": 0.26164151464763563, "time_profile/old_logps": 33.694718721322715, "time_profile/ref_logps": 33.787934163585305, "time_profile/reward": 0.004452788271009922, "time_profile/text_rollout": 518.2917092712596, "train/gen_step": 6148.0, "und/advantages_abs_mean": 0.088623046875, "und/advantages_max": 0.4375, "und/advantages_mean": 0.0, "und/advantages_min": -0.3125, "und/advantages_std": 0.14899368584156036, "und/clip_ratio/high_mean": 0.000445808833319461, "und/clip_ratio/low_mean": 0.0001805975625757128, "und/clip_ratio/region_mean": 0.0006264063958951738, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.91015625, "und/completion/min_length": 253.0, "und/frac_reward_zero_std": 0.546875, "und/kl": 0.08030579777550884, "und/loss": 0.008490518508324385, "und/prompt/max_length": 313.0, "und/prompt/mean_length": 237.5, "und/prompt/min_length": 162.0, "und/reward": 0.0673828125, "und/reward_std": 0.1709035038948059 }, { "epoch": 1.2433947157726182, "grad_norm": 0.002296620514243841, "learning_rate": 1e-05, "loss": 0.0021, "step": 98, "time_profile/curr_logprobs_and_update": 0.2616005388117628, "train/gen_step": 6212.0, "und/clip_ratio/high_mean": 0.0002009032541536726, "und/clip_ratio/low_mean": 5.5803571740398183e-05, "und/clip_ratio/region_mean": 0.0002567068258940708, "und/kl": 0.05221230116148945, "und/loss": 0.0016822464920096536 }, { "epoch": 1.256204963971177, "grad_norm": 0.0028974718879908323, "learning_rate": 1e-05, "loss": 0.0042, "rewards/correctness_reward_func/mean": 0.0, "rewards/correctness_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0546875, "rewards/strict_format_reward_func/std": 0.15728822350502014, "step": 99, "time_profile/curr_logprobs_and_update": 0.26193443170632236, "time_profile/old_logps": 33.68688790965825, "time_profile/ref_logps": 33.75524262525141, "time_profile/reward": 0.004406670108437538, "time_profile/text_rollout": 518.5918508702889, "train/gen_step": 6276.0, "und/advantages_abs_mean": 0.095458984375, "und/advantages_max": 0.4375, "und/advantages_mean": 0.0, "und/advantages_min": -0.3125, "und/advantages_std": 0.1546332687139511, "und/clip_ratio/high_mean": 0.00018194289805251174, "und/clip_ratio/low_mean": 4.340277882874943e-05, "und/clip_ratio/region_mean": 0.00022534567688126117, "und/completion/max_length": 255.0, "und/completion/mean_length": 254.91796875, "und/completion/min_length": 254.0, "und/frac_reward_zero_std": 0.515625, "und/kl": 0.10376544366590679, "und/loss": 0.0064854817031232415, "und/prompt/max_length": 313.0, "und/prompt/mean_length": 237.5, "und/prompt/min_length": 162.0, "und/reward": 0.0732421875, "und/reward_std": 0.17696848511695862 }, { "epoch": 1.2690152121697358, "grad_norm": 0.002257066546007991, "learning_rate": 1e-05, "loss": 0.0048, "step": 100, "time_profile/curr_logprobs_and_update": 0.26184689076035284, "train/gen_step": 6340.0, "und/clip_ratio/high_mean": 5.425347262644209e-05, "und/clip_ratio/low_mean": 0.0, "und/clip_ratio/region_mean": 5.425347262644209e-05, "und/kl": 0.12037181136838626, "und/loss": 0.0016251774967486199 } ], "logging_steps": 1.0, "max_steps": 790, "num_input_tokens_seen": 0, "num_train_epochs": 10, "save_steps": 50, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 1, "trial_name": null, "trial_params": null }