{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 100, "global_step": 877, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5112152099609375, "epoch": 0.0011402508551881414, "frac_reward_zero_std": 0.984375, "grad_norm": 0.000928188266698271, "kl": 0.0, "learning_rate": 0.0, "loss": -0.0, "num_tokens": 322056.0, "reward": 0.01435546949505806, "reward_std": 0.18717169761657715, "rewards/code_complexity_reward/mean": 0.005566406063735485, "rewards/code_complexity_reward/std": 0.07257677614688873, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 1, "step_time": 48.01857057400048 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5017318725585938, "epoch": 0.002280501710376283, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0009395060478709638, "kl": 0.0, "learning_rate": 5.681818181818182e-08, "loss": 0.0, "num_tokens": 644704.0, "reward": 0.013867187313735485, "reward_std": 0.18081432580947876, "rewards/code_complexity_reward/mean": 0.005078124813735485, "rewards/code_complexity_reward/std": 0.0662350207567215, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 2, "step_time": 49.314027764834464 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5292491912841797, "epoch": 0.0034207525655644243, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0009974789572879672, "kl": 0.00010262429714202881, "learning_rate": 1.1363636363636364e-07, "loss": 0.0, "num_tokens": 968084.0, "reward": 0.013867187313735485, "reward_std": 0.18081432580947876, "rewards/code_complexity_reward/mean": 0.005078124813735485, "rewards/code_complexity_reward/std": 0.0662350207567215, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 3, "step_time": 50.980388288386166 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5012245178222656, "epoch": 0.004561003420752566, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0016204880084842443, "kl": 0.0006495863199234009, "learning_rate": 1.7045454545454545e-07, "loss": 0.0, "num_tokens": 1290684.0, "reward": 0.01992187649011612, "reward_std": 0.20638912916183472, "rewards/code_complexity_reward/mean": 0.00917968712747097, "rewards/code_complexity_reward/std": 0.09254876524209976, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 4, "step_time": 51.30323136691004 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5167083740234375, "epoch": 0.005701254275940707, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0011615775292739272, "kl": 0.0007846653461456299, "learning_rate": 2.2727272727272729e-07, "loss": 0.0, "num_tokens": 1610276.0, "reward": 0.009570312686264515, "reward_std": 0.15297509729862213, "rewards/code_complexity_reward/mean": 0.0037109374534338713, "rewards/code_complexity_reward/std": 0.05931687355041504, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 5, "step_time": 44.86139403562993 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5129203796386719, "epoch": 0.0068415051311288486, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001355443848297, "kl": 0.0008092373609542847, "learning_rate": 2.840909090909091e-07, "loss": 0.0, "num_tokens": 1933736.0, "reward": 0.01689453050494194, "reward_std": 0.19455082714557648, "rewards/code_complexity_reward/mean": 0.007128905970603228, "rewards/code_complexity_reward/std": 0.08050087094306946, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 6, "step_time": 51.19557042513043 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 511.8359375, "completions/mean_terminated_length": 428.0, "completions/min_length": 428.0, "completions/min_terminated_length": 428.0, "entropy": 0.5167092299088836, "epoch": 0.00798175598631699, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0018476588884368539, "kl": 0.0008008307386262459, "learning_rate": 3.409090909090909e-07, "loss": 0.0005, "num_tokens": 2256924.0, "reward": 0.0283203125, "reward_std": 0.26036012172698975, "rewards/code_complexity_reward/mean": 0.0107421875, "rewards/code_complexity_reward/std": 0.09882795065641403, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 7, "step_time": 51.38472785707563 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5054416656494141, "epoch": 0.009122006841505131, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0007894547306932509, "kl": 0.0007839351892471313, "learning_rate": 3.9772727272727276e-07, "loss": 0.0, "num_tokens": 2577560.0, "reward": 0.0047851563431322575, "reward_std": 0.10827571898698807, "rewards/code_complexity_reward/mean": 0.0018554687267169356, "rewards/code_complexity_reward/std": 0.04198446497321129, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 8, "step_time": 49.739017243497074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5171127319335938, "epoch": 0.010262257696693273, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.001547661842778325, "kl": 0.0007987022399902344, "learning_rate": 4.5454545454545457e-07, "loss": 0.0, "num_tokens": 2898880.0, "reward": 0.03271484375, "reward_std": 0.2782195508480072, "rewards/code_complexity_reward/mean": 0.01220703125, "rewards/code_complexity_reward/std": 0.103992760181427, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 9, "step_time": 51.26508019026369 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5169639587402344, "epoch": 0.011402508551881414, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0017646212363615632, "kl": 0.0008037537336349487, "learning_rate": 5.113636363636364e-07, "loss": 0.0, "num_tokens": 3220876.0, "reward": 0.02587890625, "reward_std": 0.24442754685878754, "rewards/code_complexity_reward/mean": 0.01025390625, "rewards/code_complexity_reward/std": 0.09558416903018951, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 10, "step_time": 44.12486486788839 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5135402679443359, "epoch": 0.012542759407069556, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0006548846722580492, "kl": 0.0007953494787216187, "learning_rate": 5.681818181818182e-07, "loss": 0.0, "num_tokens": 3544580.0, "reward": 0.0028320313431322575, "reward_std": 0.06408155709505081, "rewards/code_complexity_reward/mean": 0.0018554687267169356, "rewards/code_complexity_reward/std": 0.04198446497321129, "rewards/code_execution_reward/mean": 0.0, "rewards/code_execution_reward/std": 0.0, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 11, "step_time": 44.61772104538977 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5177841186523438, "epoch": 0.013683010262257697, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001364148105494678, "kl": 0.0007954537868499756, "learning_rate": 6.25e-07, "loss": 0.0, "num_tokens": 3867252.0, "reward": 0.02861328236758709, "reward_std": 0.2630296051502228, "rewards/code_complexity_reward/mean": 0.011035156436264515, "rewards/code_complexity_reward/std": 0.10145855695009232, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 12, "step_time": 45.31378670129925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5069789886474609, "epoch": 0.014823261117445839, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0013727092882618308, "kl": 0.0008048564195632935, "learning_rate": 6.818181818181818e-07, "loss": 0.0, "num_tokens": 4187496.0, "reward": 0.01953125, "reward_std": 0.20385083556175232, "rewards/code_complexity_reward/mean": 0.0087890625, "rewards/code_complexity_reward/std": 0.08892100304365158, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 13, "step_time": 50.75423573423177 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5169792175292969, "epoch": 0.01596351197263398, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.001513324212282896, "kl": 0.000804901123046875, "learning_rate": 7.386363636363638e-07, "loss": 0.0, "num_tokens": 4507536.0, "reward": 0.03339843824505806, "reward_std": 0.2839609682559967, "rewards/code_complexity_reward/mean": 0.012890624813735485, "rewards/code_complexity_reward/std": 0.10961524397134781, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 14, "step_time": 44.678417798131704 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 511.939453125, "completions/mean_terminated_length": 481.0, "completions/min_length": 481.0, "completions/min_terminated_length": 481.0, "entropy": 0.5174384005367756, "epoch": 0.01710376282782212, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0018425935413688421, "kl": 0.000791448657764704, "learning_rate": 7.954545454545455e-07, "loss": 0.0002, "num_tokens": 4827817.0, "reward": 0.03769531473517418, "reward_std": 0.2995021343231201, "rewards/code_complexity_reward/mean": 0.014257811941206455, "rewards/code_complexity_reward/std": 0.11331094056367874, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 15, "step_time": 50.87936371937394 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5083122253417969, "epoch": 0.018244013683010263, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014342929935082793, "kl": 0.0007891356945037842, "learning_rate": 8.522727272727273e-07, "loss": 0.0, "num_tokens": 5151673.0, "reward": 0.02910156548023224, "reward_std": 0.25324270129203796, "rewards/code_complexity_reward/mean": 0.01249999925494194, "rewards/code_complexity_reward/std": 0.10630790144205093, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 16, "step_time": 45.315268891863525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5028038024902344, "epoch": 0.019384264538198404, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0012504939222708344, "kl": 0.0007841289043426514, "learning_rate": 9.090909090909091e-07, "loss": 0.0, "num_tokens": 5472445.0, "reward": 0.01884765550494194, "reward_std": 0.2126186639070511, "rewards/code_complexity_reward/mean": 0.007128905970603228, "rewards/code_complexity_reward/std": 0.08044007420539856, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 17, "step_time": 45.4323341017589 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.939453125, "completions/mean_terminated_length": 496.5, "completions/min_length": 486.0, "completions/min_terminated_length": 486.0, "entropy": 0.5279938308522105, "epoch": 0.020524515393386546, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0016279332339763641, "kl": 0.0008054297986745951, "learning_rate": 9.65909090909091e-07, "loss": 0.0002, "num_tokens": 5795134.0, "reward": 0.03769531473517418, "reward_std": 0.2995184659957886, "rewards/code_complexity_reward/mean": 0.01425781287252903, "rewards/code_complexity_reward/std": 0.11335410922765732, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 18, "step_time": 51.13050156831741 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 511.86328125, "completions/mean_terminated_length": 477.0, "completions/min_length": 458.0, "completions/min_terminated_length": 458.0, "entropy": 0.5091929119080305, "epoch": 0.021664766248574687, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.0019620079547166824, "kl": 0.0007929565854283283, "learning_rate": 1.0227272727272729e-06, "loss": 0.0003, "num_tokens": 6117300.0, "reward": 0.04824218899011612, "reward_std": 0.3305644094944, "rewards/code_complexity_reward/mean": 0.01992187462747097, "rewards/code_complexity_reward/std": 0.1347375214099884, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 19, "step_time": 50.607015871442854 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5194282531738281, "epoch": 0.02280501710376283, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0010497239418327808, "kl": 0.0008287280797958374, "learning_rate": 1.0795454545454546e-06, "loss": 0.0, "num_tokens": 6440084.0, "reward": 0.009570312686264515, "reward_std": 0.15297509729862213, "rewards/code_complexity_reward/mean": 0.0037109374534338713, "rewards/code_complexity_reward/std": 0.05931687355041504, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 20, "step_time": 45.71797686628997 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5174217224121094, "epoch": 0.02394526795895097, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0011697729350998998, "kl": 0.0008073151111602783, "learning_rate": 1.1363636363636364e-06, "loss": 0.0, "num_tokens": 6760248.0, "reward": 0.01699218899011612, "reward_std": 0.19523265957832336, "rewards/code_complexity_reward/mean": 0.0072265625931322575, "rewards/code_complexity_reward/std": 0.08154886960983276, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 21, "step_time": 45.373082681559026 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5179214477539062, "epoch": 0.02508551881413911, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0017554357182234526, "kl": 0.0007967948913574219, "learning_rate": 1.1931818181818183e-06, "loss": 0.0, "num_tokens": 7082108.0, "reward": 0.02753906324505806, "reward_std": 0.24143528938293457, "rewards/code_complexity_reward/mean": 0.012890624813735485, "rewards/code_complexity_reward/std": 0.1096152514219284, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 22, "step_time": 50.12867009919137 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.998046875, "completions/mean_terminated_length": 511.0, "completions/min_length": 511.0, "completions/min_terminated_length": 511.0, "entropy": 0.5365309370681643, "epoch": 0.026225769669327253, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.0020867169369012117, "kl": 0.0008162720214386354, "learning_rate": 1.25e-06, "loss": 0.0, "num_tokens": 7403487.0, "reward": 0.04335937649011612, "reward_std": 0.3128601014614105, "rewards/code_complexity_reward/mean": 0.01796875149011612, "rewards/code_complexity_reward/std": 0.12755942344665527, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 23, "step_time": 44.71202396415174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 511.712890625, "completions/mean_terminated_length": 365.0, "completions/min_length": 365.0, "completions/min_terminated_length": 365.0, "entropy": 0.5132100093178451, "epoch": 0.027366020524515394, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0016562737291678786, "kl": 0.0008009661551113822, "learning_rate": 1.3068181818181819e-06, "loss": 0.0008, "num_tokens": 7726264.0, "reward": 0.03115234524011612, "reward_std": 0.26817721128463745, "rewards/code_complexity_reward/mean": 0.01259765587747097, "rewards/code_complexity_reward/std": 0.10714443773031235, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 24, "step_time": 44.949530518613756 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5088729858398438, "epoch": 0.028506271379703536, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0013304443564265966, "kl": 0.0008032172918319702, "learning_rate": 1.3636363636363636e-06, "loss": 0.0, "num_tokens": 8049092.0, "reward": 0.01162109337747097, "reward_std": 0.15742145478725433, "rewards/code_complexity_reward/mean": 0.0047851563431322575, "rewards/code_complexity_reward/std": 0.06288523226976395, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 25, "step_time": 45.34271411690861 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5193710327148438, "epoch": 0.029646522234891677, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0009804965229704976, "kl": 0.0008179247379302979, "learning_rate": 1.4204545454545458e-06, "loss": 0.0, "num_tokens": 8371876.0, "reward": 0.00732421875, "reward_std": 0.12005407363176346, "rewards/code_complexity_reward/mean": 0.00341796875, "rewards/code_complexity_reward/std": 0.05483507737517357, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 26, "step_time": 45.40594596043229 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5266876220703125, "epoch": 0.03078677309007982, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0015175772132351995, "kl": 0.000812232494354248, "learning_rate": 1.4772727272727275e-06, "loss": 0.0, "num_tokens": 8693504.0, "reward": 0.014941406436264515, "reward_std": 0.1745455265045166, "rewards/code_complexity_reward/mean": 0.007128905970603228, "rewards/code_complexity_reward/std": 0.08044007420539856, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 27, "step_time": 45.46625022869557 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5110454559326172, "epoch": 0.03192702394526796, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0011097366223111749, "kl": 0.0008052438497543335, "learning_rate": 1.5340909090909093e-06, "loss": 0.0, "num_tokens": 9015180.0, "reward": 0.01435546949505806, "reward_std": 0.18717169761657715, "rewards/code_complexity_reward/mean": 0.005566406063735485, "rewards/code_complexity_reward/std": 0.07257677614688873, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 28, "step_time": 44.99307359568775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5223236083984375, "epoch": 0.0330672748004561, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0008341497159563005, "kl": 0.0008229911327362061, "learning_rate": 1.590909090909091e-06, "loss": 0.0, "num_tokens": 9339148.0, "reward": 0.00937500037252903, "reward_std": 0.1498531550168991, "rewards/code_complexity_reward/mean": 0.0035156249068677425, "rewards/code_complexity_reward/std": 0.056194934993982315, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 29, "step_time": 44.85392099339515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 511.98046875, "completions/mean_terminated_length": 502.0, "completions/min_length": 502.0, "completions/min_terminated_length": 502.0, "entropy": 0.5218262006528676, "epoch": 0.03420752565564424, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014307784149423242, "kl": 0.0008145257297655917, "learning_rate": 1.6477272727272728e-06, "loss": 0.0001, "num_tokens": 9660930.0, "reward": 0.02861328236758709, "reward_std": 0.2630295753479004, "rewards/code_complexity_reward/mean": 0.01103515550494194, "rewards/code_complexity_reward/std": 0.10145855695009232, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 30, "step_time": 51.10716211237013 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5156669616699219, "epoch": 0.03534777651083238, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0016982245724648237, "kl": 0.0008130669593811035, "learning_rate": 1.7045454545454546e-06, "loss": 0.0, "num_tokens": 9981790.0, "reward": 0.02236328274011612, "reward_std": 0.2138155847787857, "rewards/code_complexity_reward/mean": 0.01064453087747097, "rewards/code_complexity_reward/std": 0.09801840037107468, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 31, "step_time": 44.90444824937731 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.510009765625, "epoch": 0.036488027366020526, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0009442435693927109, "kl": 0.0007992833852767944, "learning_rate": 1.7613636363636365e-06, "loss": 0.0, "num_tokens": 10302382.0, "reward": 0.00937500037252903, "reward_std": 0.1498531550168991, "rewards/code_complexity_reward/mean": 0.0035156249068677425, "rewards/code_complexity_reward/std": 0.056194934993982315, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 32, "step_time": 44.600112142041326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 511.9609375, "completions/mean_terminated_length": 492.0, "completions/min_length": 492.0, "completions/min_terminated_length": 492.0, "entropy": 0.5139688942581415, "epoch": 0.037628278221208664, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0012904007453471422, "kl": 0.0007926567304821219, "learning_rate": 1.8181818181818183e-06, "loss": 0.0001, "num_tokens": 10624390.0, "reward": 0.01914062537252903, "reward_std": 0.21591484546661377, "rewards/code_complexity_reward/mean": 0.0074218749068677425, "rewards/code_complexity_reward/std": 0.0837220847606659, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 33, "step_time": 45.52419452928007 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5211219787597656, "epoch": 0.03876852907639681, "frac_reward_zero_std": 1.0, "grad_norm": 8.541852366761304e-06, "kl": 0.0008100569248199463, "learning_rate": 1.8750000000000003e-06, "loss": 0.0, "num_tokens": 10946142.0, "reward": 0.0, "reward_std": 0.0, "rewards/code_complexity_reward/mean": 0.0, "rewards/code_complexity_reward/std": 0.0, "rewards/code_execution_reward/mean": 0.0, "rewards/code_execution_reward/std": 0.0, "rewards/code_syntax_reward/mean": 0.0, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 34, "step_time": 45.319515343755484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5173721313476562, "epoch": 0.039908779931584946, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0014867030549794436, "kl": 0.0007915347814559937, "learning_rate": 1.931818181818182e-06, "loss": 0.0, "num_tokens": 11267862.0, "reward": 0.02373046986758709, "reward_std": 0.23920601606369019, "rewards/code_complexity_reward/mean": 0.009082031436264515, "rewards/code_complexity_reward/std": 0.09157534688711166, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 35, "step_time": 45.595273833721876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5191555023193359, "epoch": 0.04104903078677309, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0006888291682116687, "kl": 0.000832676887512207, "learning_rate": 1.9886363636363638e-06, "loss": 0.0, "num_tokens": 11588714.0, "reward": 0.0047851563431322575, "reward_std": 0.10827571898698807, "rewards/code_complexity_reward/mean": 0.0018554687267169356, "rewards/code_complexity_reward/std": 0.04198446497321129, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 36, "step_time": 45.82450713403523 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.520599365234375, "epoch": 0.04218928164196123, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0015760608948767185, "kl": 0.0008025020360946655, "learning_rate": 2.0454545454545457e-06, "loss": 0.0, "num_tokens": 11910162.0, "reward": 0.02255859412252903, "reward_std": 0.21368058025836945, "rewards/code_complexity_reward/mean": 0.010839843191206455, "rewards/code_complexity_reward/std": 0.09972897917032242, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 37, "step_time": 51.00037721730769 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5080204010009766, "epoch": 0.043329532497149374, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0009549854439683259, "kl": 0.0007791370153427124, "learning_rate": 2.1022727272727277e-06, "loss": 0.0, "num_tokens": 12232046.0, "reward": 0.009570312686264515, "reward_std": 0.15297509729862213, "rewards/code_complexity_reward/mean": 0.0037109374534338713, "rewards/code_complexity_reward/std": 0.05931687355041504, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 38, "step_time": 50.82266180869192 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 511.97265625, "completions/mean_terminated_length": 498.0, "completions/min_length": 498.0, "completions/min_terminated_length": 498.0, "entropy": 0.5211005229502916, "epoch": 0.04446978335233751, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0013338972348719835, "kl": 0.00080246942161466, "learning_rate": 2.1590909090909092e-06, "loss": 0.0001, "num_tokens": 12553392.0, "reward": 0.02001953125, "reward_std": 0.20752620697021484, "rewards/code_complexity_reward/mean": 0.00927734375, "rewards/code_complexity_reward/std": 0.09351195394992828, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 39, "step_time": 44.5429763328284 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 511.876953125, "completions/mean_terminated_length": 480.5, "completions/min_length": 457.0, "completions/min_terminated_length": 457.0, "entropy": 0.5308737577870488, "epoch": 0.04561003420752566, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0015828817849978805, "kl": 0.0008200870324799325, "learning_rate": 2.2159090909090912e-06, "loss": 0.0003, "num_tokens": 12874589.0, "reward": 0.02841796912252903, "reward_std": 0.2612592577934265, "rewards/code_complexity_reward/mean": 0.01083984412252903, "rewards/code_complexity_reward/std": 0.09972897917032242, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 40, "step_time": 45.28833164088428 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 511.9765625, "completions/mean_terminated_length": 500.0, "completions/min_length": 500.0, "completions/min_terminated_length": 500.0, "entropy": 0.5173209840431809, "epoch": 0.046750285062713795, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0010899591725319624, "kl": 0.0008001563546713442, "learning_rate": 2.2727272727272728e-06, "loss": 0.0, "num_tokens": 13196973.0, "reward": 0.01523437537252903, "reward_std": 0.17745302617549896, "rewards/code_complexity_reward/mean": 0.0074218749068677425, "rewards/code_complexity_reward/std": 0.0837220847606659, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 41, "step_time": 45.432716919109225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 511.5078125, "completions/mean_terminated_length": 386.0, "completions/min_length": 375.0, "completions/min_terminated_length": 375.0, "entropy": 0.5257157552987337, "epoch": 0.04789053591790194, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.0018305275589227676, "kl": 0.0008057071017901762, "learning_rate": 2.3295454545454547e-06, "loss": 0.0008, "num_tokens": 13517881.0, "reward": 0.0400390662252903, "reward_std": 0.3026600778102875, "rewards/code_complexity_reward/mean": 0.015624999068677425, "rewards/code_complexity_reward/std": 0.11750004440546036, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 42, "step_time": 45.85349241644144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.51092529296875, "epoch": 0.04903078677309008, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0010073003359138966, "kl": 0.0007970333099365234, "learning_rate": 2.3863636363636367e-06, "loss": 0.0, "num_tokens": 13840669.0, "reward": 0.01914062537252903, "reward_std": 0.21591484546661377, "rewards/code_complexity_reward/mean": 0.0074218749068677425, "rewards/code_complexity_reward/std": 0.0837220847606659, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 43, "step_time": 45.328043133951724 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5050544738769531, "epoch": 0.05017103762827822, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0011301900958642364, "kl": 0.0007976144552230835, "learning_rate": 2.4431818181818182e-06, "loss": 0.0, "num_tokens": 14162377.0, "reward": 0.00927734375, "reward_std": 0.14836639165878296, "rewards/code_complexity_reward/mean": 0.00341796875, "rewards/code_complexity_reward/std": 0.05483507737517357, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 44, "step_time": 45.07921890169382 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5187301635742188, "epoch": 0.05131128848346636, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0012788443127647042, "kl": 0.0008012354373931885, "learning_rate": 2.5e-06, "loss": 0.0, "num_tokens": 14482993.0, "reward": 0.01992187649011612, "reward_std": 0.206862673163414, "rewards/code_complexity_reward/mean": 0.00917968712747097, "rewards/code_complexity_reward/std": 0.09254877269268036, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 45, "step_time": 51.35232986416668 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5213508605957031, "epoch": 0.052451539338654506, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001383924507535994, "kl": 0.0008253753185272217, "learning_rate": 2.556818181818182e-06, "loss": 0.0, "num_tokens": 14806137.0, "reward": 0.01699218712747097, "reward_std": 0.19625738263130188, "rewards/code_complexity_reward/mean": 0.00722656212747097, "rewards/code_complexity_reward/std": 0.0816088393330574, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 46, "step_time": 45.21146249305457 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5218849182128906, "epoch": 0.053591790193842644, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.000614947872236371, "kl": 0.000811353325843811, "learning_rate": 2.6136363636363637e-06, "loss": 0.0, "num_tokens": 15128693.0, "reward": 0.00937500037252903, "reward_std": 0.1498531550168991, "rewards/code_complexity_reward/mean": 0.0035156249068677425, "rewards/code_complexity_reward/std": 0.056194934993982315, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 47, "step_time": 45.53867436479777 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5125198364257812, "epoch": 0.05473204104903079, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0005773964803665876, "kl": 0.0007997602224349976, "learning_rate": 2.6704545454545457e-06, "loss": 0.0, "num_tokens": 15451793.0, "reward": 0.0027343749534338713, "reward_std": 0.06187184154987335, "rewards/code_complexity_reward/mean": 0.0017578124534338713, "rewards/code_complexity_reward/std": 0.039774756878614426, "rewards/code_execution_reward/mean": 0.0, "rewards/code_execution_reward/std": 0.0, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 48, "step_time": 51.66811321396381 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5131683349609375, "epoch": 0.055872291904218926, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0015757797518745065, "kl": 0.0008024126291275024, "learning_rate": 2.7272727272727272e-06, "loss": 0.0, "num_tokens": 15773353.0, "reward": 0.02373046986758709, "reward_std": 0.23920601606369019, "rewards/code_complexity_reward/mean": 0.00908203050494194, "rewards/code_complexity_reward/std": 0.09157534688711166, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 49, "step_time": 44.69605880603194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5207347869873047, "epoch": 0.05701254275940707, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0011314681032672524, "kl": 0.0008026957511901855, "learning_rate": 2.784090909090909e-06, "loss": 0.0, "num_tokens": 16093333.0, "reward": 0.01425781287252903, "reward_std": 0.18590718507766724, "rewards/code_complexity_reward/mean": 0.0054687499068677425, "rewards/code_complexity_reward/std": 0.0713263675570488, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 50, "step_time": 51.095568262040615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5158653259277344, "epoch": 0.05815279361459521, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0019937690813094378, "kl": 0.0008038431406021118, "learning_rate": 2.8409090909090916e-06, "loss": 0.0, "num_tokens": 16415505.0, "reward": 0.04082031548023224, "reward_std": 0.30821070075035095, "rewards/code_complexity_reward/mean": 0.01640624925494194, "rewards/code_complexity_reward/std": 0.12281106412410736, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 51, "step_time": 45.37204051669687 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 511.404296875, "completions/mean_terminated_length": 410.3333435058594, "completions/min_length": 281.0, "completions/min_terminated_length": 281.0, "entropy": 0.5024195956066251, "epoch": 0.059293044469783354, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.0019817177671939135, "kl": 0.0007676757659282885, "learning_rate": 2.897727272727273e-06, "loss": 0.001, "num_tokens": 16737660.0, "reward": 0.04257812723517418, "reward_std": 0.31866833567619324, "rewards/code_complexity_reward/mean": 0.01621093787252903, "rewards/code_complexity_reward/std": 0.12143507599830627, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 52, "step_time": 50.70893675740808 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 511.9140625, "completions/mean_terminated_length": 490.0, "completions/min_length": 488.0, "completions/min_terminated_length": 488.0, "entropy": 0.5042452295310795, "epoch": 0.06043329532497149, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.0020700653549283743, "kl": 0.0007887799974923837, "learning_rate": 2.954545454545455e-06, "loss": 0.0001, "num_tokens": 17060268.0, "reward": 0.04072265326976776, "reward_std": 0.3077709376811981, "rewards/code_complexity_reward/mean": 0.01630859263241291, "rewards/code_complexity_reward/std": 0.12208496779203415, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 53, "step_time": 45.96823632996529 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5076694488525391, "epoch": 0.06157354618015964, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0009545396897010505, "kl": 0.0007885098457336426, "learning_rate": 3.0113636363636366e-06, "loss": 0.0, "num_tokens": 17381460.0, "reward": 0.01435546949505806, "reward_std": 0.18717168271541595, "rewards/code_complexity_reward/mean": 0.005566406063735485, "rewards/code_complexity_reward/std": 0.07257677614688873, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 54, "step_time": 46.10059131216258 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5006923675537109, "epoch": 0.06271379703534778, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0008144342573359609, "kl": 0.0007875263690948486, "learning_rate": 3.0681818181818186e-06, "loss": 0.0, "num_tokens": 17702376.0, "reward": 0.0047851563431322575, "reward_std": 0.10827571898698807, "rewards/code_complexity_reward/mean": 0.0018554687267169356, "rewards/code_complexity_reward/std": 0.04198446497321129, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 55, "step_time": 45.45572846569121 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5130405426025391, "epoch": 0.06385404789053592, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001453711069189012, "kl": 0.0008146464824676514, "learning_rate": 3.125e-06, "loss": 0.0, "num_tokens": 18022296.0, "reward": 0.012939453125, "reward_std": 0.16482453048229218, "rewards/code_complexity_reward/mean": 0.00537109375, "rewards/code_complexity_reward/std": 0.07012330740690231, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 56, "step_time": 50.502467853948474 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 511.923828125, "completions/mean_terminated_length": 492.5, "completions/min_length": 489.0, "completions/min_terminated_length": 489.0, "entropy": 0.5087363719940186, "epoch": 0.06499429874572406, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0017609602073207498, "kl": 0.0007867640943004517, "learning_rate": 3.181818181818182e-06, "loss": 0.0002, "num_tokens": 18344161.0, "reward": 0.021484375, "reward_std": 0.1869189441204071, "rewards/code_complexity_reward/mean": 0.0126953125, "rewards/code_complexity_reward/std": 0.10797441005706787, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 57, "step_time": 50.60492390021682 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5138282775878906, "epoch": 0.0661345496009122, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0008789357962086797, "kl": 0.0008032172918319702, "learning_rate": 3.2386363636363637e-06, "loss": 0.0, "num_tokens": 18664941.0, "reward": 0.009570312686264515, "reward_std": 0.15297509729862213, "rewards/code_complexity_reward/mean": 0.0037109374534338713, "rewards/code_complexity_reward/std": 0.05931687355041504, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 58, "step_time": 45.072539716959 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5053024291992188, "epoch": 0.06727480045610035, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0012223056983202696, "kl": 0.0007817298173904419, "learning_rate": 3.2954545454545456e-06, "loss": 0.0, "num_tokens": 18986801.0, "reward": 0.0234375, "reward_std": 0.23624072968959808, "rewards/code_complexity_reward/mean": 0.0087890625, "rewards/code_complexity_reward/std": 0.08859027177095413, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 59, "step_time": 45.34158855024725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5078372955322266, "epoch": 0.06841505131128849, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0010403324849903584, "kl": 0.0007797479629516602, "learning_rate": 3.352272727272727e-06, "loss": 0.0, "num_tokens": 19308393.0, "reward": 0.01406249962747097, "reward_std": 0.18343189358711243, "rewards/code_complexity_reward/mean": 0.0052734375931322575, "rewards/code_complexity_reward/std": 0.06897008419036865, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 60, "step_time": 45.67875342257321 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5160083770751953, "epoch": 0.06955530216647662, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0013135956833139062, "kl": 0.0007977187633514404, "learning_rate": 3.409090909090909e-06, "loss": 0.0, "num_tokens": 19630185.0, "reward": 0.01865234412252903, "reward_std": 0.2104155272245407, "rewards/code_complexity_reward/mean": 0.0069335936568677425, "rewards/code_complexity_reward/std": 0.07823750376701355, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 61, "step_time": 44.63997431099415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 511.904296875, "completions/mean_terminated_length": 463.0, "completions/min_length": 463.0, "completions/min_terminated_length": 463.0, "entropy": 0.5141611117869616, "epoch": 0.07069555302166476, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0010444809449836612, "kl": 0.000803096119852853, "learning_rate": 3.4659090909090915e-06, "loss": 0.0002, "num_tokens": 19953196.0, "reward": 0.01894531399011612, "reward_std": 0.2137230783700943, "rewards/code_complexity_reward/mean": 0.00722656212747097, "rewards/code_complexity_reward/std": 0.08154886960983276, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 62, "step_time": 50.80704717151821 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 511.970703125, "completions/mean_terminated_length": 497.0, "completions/min_length": 497.0, "completions/min_terminated_length": 497.0, "entropy": 0.5080030928365886, "epoch": 0.07183580387685291, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0010742857120931149, "kl": 0.000782505007009604, "learning_rate": 3.522727272727273e-06, "loss": 0.0001, "num_tokens": 20275357.0, "reward": 0.01435546949505806, "reward_std": 0.18717169761657715, "rewards/code_complexity_reward/mean": 0.005566406063735485, "rewards/code_complexity_reward/std": 0.07257678359746933, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 63, "step_time": 50.60806827060878 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5217208862304688, "epoch": 0.07297605473204105, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0009564663632772863, "kl": 0.0008136481046676636, "learning_rate": 3.579545454545455e-06, "loss": 0.0, "num_tokens": 20599105.0, "reward": 0.007519531529396772, "reward_std": 0.12381374835968018, "rewards/code_complexity_reward/mean": 0.003613281063735485, "rewards/code_complexity_reward/std": 0.057777076959609985, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 64, "step_time": 50.75970184523612 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 511.951171875, "completions/mean_terminated_length": 487.0, "completions/min_length": 487.0, "completions/min_terminated_length": 487.0, "entropy": 0.5167407086119056, "epoch": 0.07411630558722919, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001351800630800426, "kl": 0.0008016267838684143, "learning_rate": 3.6363636363636366e-06, "loss": 0.0001, "num_tokens": 20922016.0, "reward": 0.03339843824505806, "reward_std": 0.2839609682559967, "rewards/code_complexity_reward/mean": 0.012890624813735485, "rewards/code_complexity_reward/std": 0.10961524397134781, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 65, "step_time": 51.05286444257945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5175743103027344, "epoch": 0.07525655644241733, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0012636483879759908, "kl": 0.0008223950862884521, "learning_rate": 3.6931818181818186e-06, "loss": 0.0, "num_tokens": 21244504.0, "reward": 0.01904296875, "reward_std": 0.21482177078723907, "rewards/code_complexity_reward/mean": 0.00732421875, "rewards/code_complexity_reward/std": 0.08264267444610596, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 66, "step_time": 51.70834381971508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 511.701171875, "completions/mean_terminated_length": 435.5, "completions/min_length": 407.0, "completions/min_terminated_length": 407.0, "entropy": 0.5094133042730391, "epoch": 0.07639680729760548, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0016778953140601516, "kl": 0.0007881049095885828, "learning_rate": 3.7500000000000005e-06, "loss": 0.0006, "num_tokens": 21566327.0, "reward": 0.03183593973517418, "reward_std": 0.2615850865840912, "rewards/code_complexity_reward/mean": 0.01425781287252903, "rewards/code_complexity_reward/std": 0.11348351836204529, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 67, "step_time": 46.09761200193316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 511.96484375, "completions/mean_terminated_length": 494.0, "completions/min_length": 494.0, "completions/min_terminated_length": 494.0, "entropy": 0.5203476417809725, "epoch": 0.07753705815279362, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0016249080654233694, "kl": 0.000790370379036176, "learning_rate": 3.806818181818182e-06, "loss": 0.0001, "num_tokens": 21889797.0, "reward": 0.02431640401482582, "reward_std": 0.22905300557613373, "rewards/code_complexity_reward/mean": 0.01064453087747097, "rewards/code_complexity_reward/std": 0.09796847403049469, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 68, "step_time": 49.8536473903805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5133476257324219, "epoch": 0.07867730900798175, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0006047082715667784, "kl": 0.0007890164852142334, "learning_rate": 3.863636363636364e-06, "loss": 0.0, "num_tokens": 22210889.0, "reward": 0.004589843563735485, "reward_std": 0.10385630279779434, "rewards/code_complexity_reward/mean": 0.0016601562965661287, "rewards/code_complexity_reward/std": 0.037565045058727264, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 69, "step_time": 50.79237159062177 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5172843933105469, "epoch": 0.07981755986316989, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014741484774276614, "kl": 0.000806853175163269, "learning_rate": 3.9204545454545456e-06, "loss": 0.0, "num_tokens": 22533853.0, "reward": 0.01972656324505806, "reward_std": 0.2050524204969406, "rewards/code_complexity_reward/mean": 0.008984374813735485, "rewards/code_complexity_reward/std": 0.09059135615825653, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 70, "step_time": 45.16645230166614 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5040111541748047, "epoch": 0.08095781071835804, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0012623153161257505, "kl": 0.0007951259613037109, "learning_rate": 3.9772727272727275e-06, "loss": 0.0, "num_tokens": 22855789.0, "reward": 0.01718750037252903, "reward_std": 0.19763152301311493, "rewards/code_complexity_reward/mean": 0.0074218749068677425, "rewards/code_complexity_reward/std": 0.0837220847606659, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 71, "step_time": 45.45582844968885 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5215797424316406, "epoch": 0.08209806157354618, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0009845010936260223, "kl": 0.0008087456226348877, "learning_rate": 4.0340909090909095e-06, "loss": 0.0, "num_tokens": 23176357.0, "reward": 0.00947265699505806, "reward_std": 0.15142220258712769, "rewards/code_complexity_reward/mean": 0.003613281063735485, "rewards/code_complexity_reward/std": 0.05777707323431969, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 72, "step_time": 45.82586772646755 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 511.939453125, "completions/mean_terminated_length": 481.0, "completions/min_length": 481.0, "completions/min_terminated_length": 481.0, "entropy": 0.5099092987366021, "epoch": 0.08323831242873432, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0015078980941325426, "kl": 0.0008019313927434268, "learning_rate": 4.0909090909090915e-06, "loss": 0.0001, "num_tokens": 23496782.0, "reward": 0.03144531324505806, "reward_std": 0.2704230546951294, "rewards/code_complexity_reward/mean": 0.012890624813735485, "rewards/code_complexity_reward/std": 0.10961524397134781, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 73, "step_time": 49.88486846163869 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 511.935546875, "completions/mean_terminated_length": 479.0, "completions/min_length": 479.0, "completions/min_terminated_length": 479.0, "entropy": 0.5314634321257472, "epoch": 0.08437856328392246, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0020856959745287895, "kl": 0.000814704470030847, "learning_rate": 4.1477272727272734e-06, "loss": 0.0002, "num_tokens": 23817129.0, "reward": 0.03574218973517418, "reward_std": 0.27384668588638306, "rewards/code_complexity_reward/mean": 0.01621093787252903, "rewards/code_complexity_reward/std": 0.12147535383701324, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 74, "step_time": 45.45444482099265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.527252197265625, "epoch": 0.08551881413911061, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0008569829515181482, "kl": 0.0008324682712554932, "learning_rate": 4.204545454545455e-06, "loss": 0.0, "num_tokens": 24139965.0, "reward": 0.00947265699505806, "reward_std": 0.15142220258712769, "rewards/code_complexity_reward/mean": 0.003613281063735485, "rewards/code_complexity_reward/std": 0.057777076959609985, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 75, "step_time": 50.37532264646143 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 511.94140625, "completions/mean_terminated_length": 482.0, "completions/min_length": 482.0, "completions/min_terminated_length": 482.0, "entropy": 0.5234187114983797, "epoch": 0.08665906499429875, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0021641289349645376, "kl": 0.0007968365653141518, "learning_rate": 4.2613636363636365e-06, "loss": 0.0002, "num_tokens": 24460455.0, "reward": 0.03378906473517418, "reward_std": 0.27487924695014954, "rewards/code_complexity_reward/mean": 0.01425781287252903, "rewards/code_complexity_reward/std": 0.1135697066783905, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 76, "step_time": 44.81216438766569 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5080604553222656, "epoch": 0.08779931584948689, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0015510414959862828, "kl": 0.0007940679788589478, "learning_rate": 4.3181818181818185e-06, "loss": 0.0, "num_tokens": 24781163.0, "reward": 0.02861328050494194, "reward_std": 0.2630295753479004, "rewards/code_complexity_reward/mean": 0.011035156436264515, "rewards/code_complexity_reward/std": 0.10145855695009232, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 77, "step_time": 50.73064348101616 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 511.83984375, "completions/mean_terminated_length": 471.0, "completions/min_length": 452.0, "completions/min_terminated_length": 452.0, "entropy": 0.5155928544700146, "epoch": 0.08893956670467502, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0012122971238568425, "kl": 0.0008161963150996598, "learning_rate": 4.3750000000000005e-06, "loss": 0.0002, "num_tokens": 25102429.0, "reward": 0.02656250074505806, "reward_std": 0.24699269235134125, "rewards/code_complexity_reward/mean": 0.010937499813735485, "rewards/code_complexity_reward/std": 0.10062183439731598, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 78, "step_time": 45.62870597653091 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5099411010742188, "epoch": 0.09007981755986318, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0020872210152447224, "kl": 0.0008039325475692749, "learning_rate": 4.4318181818181824e-06, "loss": 0.0, "num_tokens": 25425861.0, "reward": 0.06630859524011612, "reward_std": 0.3958970904350281, "rewards/code_complexity_reward/mean": 0.02529296651482582, "rewards/code_complexity_reward/std": 0.1510883867740631, "rewards/code_execution_reward/mean": 0.02734375, "rewards/code_execution_reward/std": 0.16324250400066376, "rewards/code_syntax_reward/mean": 0.013671875, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 79, "step_time": 45.125071617774665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 511.80078125, "completions/mean_terminated_length": 410.0, "completions/min_length": 410.0, "completions/min_terminated_length": 410.0, "entropy": 0.5061400812119246, "epoch": 0.09122006841505131, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014438808429986238, "kl": 0.0008016444562599645, "learning_rate": 4.4886363636363636e-06, "loss": 0.0003, "num_tokens": 25747439.0, "reward": 0.02158203162252903, "reward_std": 0.22166872024536133, "rewards/code_complexity_reward/mean": 0.00888671912252903, "rewards/code_complexity_reward/std": 0.08965104818344116, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 80, "step_time": 44.618858862668276 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 511.95703125, "completions/mean_terminated_length": 490.0, "completions/min_length": 490.0, "completions/min_terminated_length": 490.0, "entropy": 0.5241127479821444, "epoch": 0.09236031927023945, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0020064199343323708, "kl": 0.000801085208877339, "learning_rate": 4.5454545454545455e-06, "loss": 0.0001, "num_tokens": 26070313.0, "reward": 0.03828125074505806, "reward_std": 0.2925181984901428, "rewards/code_complexity_reward/mean": 0.01582031138241291, "rewards/code_complexity_reward/std": 0.11867545545101166, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 81, "step_time": 44.35164962243289 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 511.86328125, "completions/mean_terminated_length": 442.0, "completions/min_length": 442.0, "completions/min_terminated_length": 442.0, "entropy": 0.5148605592548847, "epoch": 0.09350057012542759, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0016232281923294067, "kl": 0.0007973802839842392, "learning_rate": 4.6022727272727275e-06, "loss": 0.0004, "num_tokens": 26391563.0, "reward": 0.033203125, "reward_std": 0.282307893037796, "rewards/code_complexity_reward/mean": 0.0126953125, "rewards/code_complexity_reward/std": 0.10797441005706787, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 82, "step_time": 44.876578114926815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.531005859375, "epoch": 0.09464082098061574, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0016402078326791525, "kl": 0.0008084923028945923, "learning_rate": 4.6590909090909095e-06, "loss": 0.0, "num_tokens": 26712759.0, "reward": 0.02646484598517418, "reward_std": 0.24681493639945984, "rewards/code_complexity_reward/mean": 0.01083984412252903, "rewards/code_complexity_reward/std": 0.09967990219593048, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 83, "step_time": 45.26349873561412 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 511.94921875, "completions/mean_terminated_length": 486.0, "completions/min_length": 486.0, "completions/min_terminated_length": 486.0, "entropy": 0.512430340051651, "epoch": 0.09578107183580388, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0018435453530400991, "kl": 0.0008080440420599189, "learning_rate": 4.715909090909091e-06, "loss": 0.0002, "num_tokens": 27033377.0, "reward": 0.02666015736758709, "reward_std": 0.24831560254096985, "rewards/code_complexity_reward/mean": 0.01103515550494194, "rewards/code_complexity_reward/std": 0.10145854949951172, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 84, "step_time": 44.644844179973006 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5149993896484375, "epoch": 0.09692132269099202, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0013406892539933324, "kl": 0.0008103996515274048, "learning_rate": 4.772727272727273e-06, "loss": 0.0, "num_tokens": 27355593.0, "reward": 0.01718750037252903, "reward_std": 0.19763152301311493, "rewards/code_complexity_reward/mean": 0.0074218749068677425, "rewards/code_complexity_reward/std": 0.0837220847606659, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 85, "step_time": 45.31094349734485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5211753845214844, "epoch": 0.09806157354618016, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0013213605852797627, "kl": 0.0008022487163543701, "learning_rate": 4.829545454545455e-06, "loss": 0.0, "num_tokens": 27677929.0, "reward": 0.02167968824505806, "reward_std": 0.22228731215000153, "rewards/code_complexity_reward/mean": 0.008984374813735485, "rewards/code_complexity_reward/std": 0.09064534306526184, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 86, "step_time": 45.530151853337884 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 511.94140625, "completions/mean_terminated_length": 482.0, "completions/min_length": 482.0, "completions/min_terminated_length": 482.0, "entropy": 0.5191793795675039, "epoch": 0.09920182440136831, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0017781669739633799, "kl": 0.000806792395451339, "learning_rate": 4.8863636363636365e-06, "loss": 0.0002, "num_tokens": 28000291.0, "reward": 0.02656249888241291, "reward_std": 0.247368723154068, "rewards/code_complexity_reward/mean": 0.010937499813735485, "rewards/code_complexity_reward/std": 0.10057320445775986, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 87, "step_time": 45.95823614951223 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5239524841308594, "epoch": 0.10034207525655645, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0016954478342086077, "kl": 0.0008341223001480103, "learning_rate": 4.9431818181818184e-06, "loss": 0.0, "num_tokens": 28322979.0, "reward": 0.03134765848517418, "reward_std": 0.26955559849739075, "rewards/code_complexity_reward/mean": 0.012792968191206455, "rewards/code_complexity_reward/std": 0.10879796743392944, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 88, "step_time": 46.096476702950895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5148105621337891, "epoch": 0.10148232611174458, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0012546635698527098, "kl": 0.0008029788732528687, "learning_rate": 5e-06, "loss": 0.0, "num_tokens": 28645371.0, "reward": 0.01220703125, "reward_std": 0.16341635584831238, "rewards/code_complexity_reward/mean": 0.00537109375, "rewards/code_complexity_reward/std": 0.07005351036787033, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 89, "step_time": 45.03944506589323 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5150718688964844, "epoch": 0.10262257696693272, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0013418194139376283, "kl": 0.0008109062910079956, "learning_rate": 4.999980182212003e-06, "loss": 0.0, "num_tokens": 28968275.0, "reward": 0.02832031436264515, "reward_std": 0.26036012172698975, "rewards/code_complexity_reward/mean": 0.0107421875, "rewards/code_complexity_reward/std": 0.09882794320583344, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 90, "step_time": 45.04765653330833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 511.771484375, "completions/mean_terminated_length": 395.0, "completions/min_length": 395.0, "completions/min_terminated_length": 395.0, "entropy": 0.5013239122927189, "epoch": 0.10376282782212087, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0012906616320833564, "kl": 0.0007872669002608745, "learning_rate": 4.999920729162207e-06, "loss": 0.0007, "num_tokens": 29290394.0, "reward": 0.01416015625, "reward_std": 0.18466046452522278, "rewards/code_complexity_reward/mean": 0.00537109375, "rewards/code_complexity_reward/std": 0.07012330740690231, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 91, "step_time": 45.311531184241176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 511.802734375, "completions/mean_terminated_length": 411.0, "completions/min_length": 411.0, "completions/min_terminated_length": 411.0, "entropy": 0.5096945846453309, "epoch": 0.10490307867730901, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.001659944886341691, "kl": 0.0008135000716720242, "learning_rate": 4.999821641793195e-06, "loss": 0.0006, "num_tokens": 29610625.0, "reward": 0.02949218824505806, "reward_std": 0.2565374970436096, "rewards/code_complexity_reward/mean": 0.012890624813735485, "rewards/code_complexity_reward/std": 0.10961524397134781, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 92, "step_time": 44.71872239001095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 511.986328125, "completions/mean_terminated_length": 505.0, "completions/min_length": 505.0, "completions/min_terminated_length": 505.0, "entropy": 0.5054364046081901, "epoch": 0.10604332953249715, "frac_reward_zero_std": 0.984375, "grad_norm": 0.000987074337899685, "kl": 0.0008129073548843735, "learning_rate": 4.999682921675919e-06, "loss": 0.0, "num_tokens": 29930954.0, "reward": 0.00742187537252903, "reward_std": 0.12352295219898224, "rewards/code_complexity_reward/mean": 0.0035156249068677425, "rewards/code_complexity_reward/std": 0.056281927973032, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 93, "step_time": 46.41400856245309 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 511.67578125, "completions/mean_terminated_length": 456.66668701171875, "completions/min_length": 435.0, "completions/min_terminated_length": 435.0, "entropy": 0.513271713629365, "epoch": 0.10718358038768529, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0018530217930674553, "kl": 0.0008160939441950177, "learning_rate": 4.999504571009682e-06, "loss": 0.0009, "num_tokens": 30253224.0, "reward": 0.04042968899011612, "reward_std": 0.30552032589912415, "rewards/code_complexity_reward/mean": 0.01601562462747097, "rewards/code_complexity_reward/std": 0.11996143311262131, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 94, "step_time": 45.36229538731277 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 511.943359375, "completions/mean_terminated_length": 497.5, "completions/min_length": 496.0, "completions/min_terminated_length": 496.0, "entropy": 0.499619385227561, "epoch": 0.10832383124287344, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0012111930409446359, "kl": 0.0007960296425153501, "learning_rate": 4.999286592622096e-06, "loss": 0.0001, "num_tokens": 30576287.0, "reward": 0.02382812649011612, "reward_std": 0.24018622934818268, "rewards/code_complexity_reward/mean": 0.00917968712747097, "rewards/code_complexity_reward/std": 0.09254877269268036, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 95, "step_time": 50.99943410139531 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 511.98046875, "completions/mean_terminated_length": 502.0, "completions/min_length": 502.0, "completions/min_terminated_length": 502.0, "entropy": 0.5179265849292278, "epoch": 0.10946408209806158, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0017258968437090516, "kl": 0.0008124707946990384, "learning_rate": 4.99902898996904e-06, "loss": 0.0001, "num_tokens": 30899697.0, "reward": 0.02158203162252903, "reward_std": 0.22265969216823578, "rewards/code_complexity_reward/mean": 0.00888671912252903, "rewards/code_complexity_reward/std": 0.08992349356412888, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 96, "step_time": 45.71308558061719 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.791015625, "completions/mean_terminated_length": 476.3333435058594, "completions/min_length": 450.0, "completions/min_terminated_length": 450.0, "entropy": 0.5244129449129105, "epoch": 0.11060433295324971, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0016772000817582011, "kl": 0.0008273767589344061, "learning_rate": 4.998731767134606e-06, "loss": 0.0006, "num_tokens": 31219414.0, "reward": 0.02851562574505806, "reward_std": 0.2621366083621979, "rewards/code_complexity_reward/mean": 0.010937499813735485, "rewards/code_complexity_reward/std": 0.10057321190834045, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 97, "step_time": 45.312702405266464 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5065078735351562, "epoch": 0.11174458380843785, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0019344384782016277, "kl": 0.0008065998554229736, "learning_rate": 4.998394928831034e-06, "loss": 0.0, "num_tokens": 31540026.0, "reward": 0.04990234598517418, "reward_std": 0.3395994305610657, "rewards/code_complexity_reward/mean": 0.01962890475988388, "rewards/code_complexity_reward/std": 0.1327875852584839, "rewards/code_execution_reward/mean": 0.01953125, "rewards/code_execution_reward/std": 0.1385180652141571, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 98, "step_time": 44.46913767233491 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 511.9609375, "completions/mean_terminated_length": 492.0, "completions/min_length": 492.0, "completions/min_terminated_length": 492.0, "entropy": 0.5093090711161494, "epoch": 0.11288483466362599, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014462806284427643, "kl": 0.0008141570460793446, "learning_rate": 4.998018480398635e-06, "loss": 0.0001, "num_tokens": 31861742.0, "reward": 0.01337890699505806, "reward_std": 0.15556131303310394, "rewards/code_complexity_reward/mean": 0.007519531063735485, "rewards/code_complexity_reward/std": 0.08484531939029694, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 99, "step_time": 45.477089506573975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.516204833984375, "epoch": 0.11402508551881414, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014144235756248236, "kl": 0.0008196085691452026, "learning_rate": 4.99760242780571e-06, "loss": 0.0, "num_tokens": 32182978.0, "reward": 0.01894531212747097, "reward_std": 0.21374596655368805, "rewards/code_complexity_reward/mean": 0.0072265625931322575, "rewards/code_complexity_reward/std": 0.0816088393330574, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 100, "step_time": 44.71392563264817 }, { "epoch": 0.11402508551881414, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.9975, "eval_completions/max_length": 512.0, "eval_completions/max_terminated_length": 9.24, "eval_completions/mean_length": 511.875, "eval_completions/mean_terminated_length": 9.24, "eval_completions/min_length": 511.0, "eval_completions/min_terminated_length": 9.24, "eval_entropy": 0.5354266226291656, "eval_frac_reward_zero_std": 0.96, "eval_kl": 0.0008520149462856353, "eval_loss": 0.00037871685344725847, "eval_num_tokens": 32182978.0, "eval_reward": 0.03012499988079071, "eval_reward_std": 0.0737291157245636, "eval_rewards/code_complexity_reward/mean": 0.011375000029802322, "eval_rewards/code_complexity_reward/std": 0.028022014498710633, "eval_rewards/code_execution_reward/mean": 0.0125, "eval_rewards/code_execution_reward/std": 0.03047140419483185, "eval_rewards/code_syntax_reward/mean": 0.00625, "eval_rewards/code_syntax_reward/std": 0.015235702097415925, "eval_rewards/reasoning_present_reward_func/mean": 0.0, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.0, "eval_rewards/xmlcount_reward_func/std": 0.0, "eval_runtime": 1260.7748, "eval_samples_per_second": 0.079, "eval_steps_per_second": 0.01, "step": 100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 511.861328125, "completions/mean_terminated_length": 441.0, "completions/min_length": 441.0, "completions/min_terminated_length": 441.0, "entropy": 0.5202585165388882, "epoch": 0.11516533637400228, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0009131806436926126, "kl": 0.000819154810415057, "learning_rate": 4.9971467776484526e-06, "loss": 0.0004, "num_tokens": 32506511.0, "reward": 0.009570312686264515, "reward_std": 0.15297509729862213, "rewards/code_complexity_reward/mean": 0.0037109374534338713, "rewards/code_complexity_reward/std": 0.05931687355041504, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 101, "step_time": 45.001069891266525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 511.970703125, "completions/mean_terminated_length": 497.0, "completions/min_length": 497.0, "completions/min_terminated_length": 497.0, "entropy": 0.524513628333807, "epoch": 0.11630558722919042, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0007165050483308733, "kl": 0.0008146761647367384, "learning_rate": 4.9966515371508445e-06, "loss": 0.0001, "num_tokens": 32828944.0, "reward": 0.004687500186264515, "reward_std": 0.1060660183429718, "rewards/code_complexity_reward/mean": 0.0017578124534338713, "rewards/code_complexity_reward/std": 0.039774756878614426, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 102, "step_time": 44.69509568065405 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5109291076660156, "epoch": 0.11744583808437856, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0015022146981209517, "kl": 0.0008076578378677368, "learning_rate": 4.9961167141645435e-06, "loss": 0.0, "num_tokens": 33151080.0, "reward": 0.02392578125, "reward_std": 0.24116241931915283, "rewards/code_complexity_reward/mean": 0.00927734375, "rewards/code_complexity_reward/std": 0.09351195394992828, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 103, "step_time": 45.176013990305364 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 511.912109375, "completions/mean_terminated_length": 467.0, "completions/min_length": 467.0, "completions/min_terminated_length": 467.0, "entropy": 0.5126009825617075, "epoch": 0.11858608893956671, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0016499029006808996, "kl": 0.0008173532605724176, "learning_rate": 4.995542317168756e-06, "loss": 0.0002, "num_tokens": 33472567.0, "reward": 0.04042968899011612, "reward_std": 0.30518385767936707, "rewards/code_complexity_reward/mean": 0.01601562462747097, "rewards/code_complexity_reward/std": 0.11992064118385315, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 104, "step_time": 51.18978889565915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5047969818115234, "epoch": 0.11972633979475485, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0009626607643440366, "kl": 0.00080910325050354, "learning_rate": 4.994928355270105e-06, "loss": 0.0, "num_tokens": 33794355.0, "reward": 0.014160157181322575, "reward_std": 0.18463397026062012, "rewards/code_complexity_reward/mean": 0.00537109375, "rewards/code_complexity_reward/std": 0.07005351036787033, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 105, "step_time": 44.89533361326903 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5162124633789062, "epoch": 0.12086659064994298, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0014022118411958218, "kl": 0.0008143037557601929, "learning_rate": 4.994274838202483e-06, "loss": 0.0, "num_tokens": 34116487.0, "reward": 0.01015624962747097, "reward_std": 0.1359332948923111, "rewards/code_complexity_reward/mean": 0.0052734375931322575, "rewards/code_complexity_reward/std": 0.06897008419036865, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 106, "step_time": 45.33267491310835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.502349853515625, "epoch": 0.12200684150513112, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.000720928015653044, "kl": 0.0008103102445602417, "learning_rate": 4.993581776326901e-06, "loss": 0.0, "num_tokens": 34437283.0, "reward": 0.004687500186264515, "reward_std": 0.1060660183429718, "rewards/code_complexity_reward/mean": 0.0017578124534338713, "rewards/code_complexity_reward/std": 0.039774756878614426, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 107, "step_time": 45.2509846938774 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 511.884765625, "completions/mean_terminated_length": 482.5, "completions/min_length": 457.0, "completions/min_terminated_length": 457.0, "entropy": 0.5135140465572476, "epoch": 0.12314709236031927, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002173924818634987, "kl": 0.0008220580493798479, "learning_rate": 4.9928491806313216e-06, "loss": 0.0003, "num_tokens": 34758924.0, "reward": 0.0478515625, "reward_std": 0.3280065953731537, "rewards/code_complexity_reward/mean": 0.01953125, "rewards/code_complexity_reward/std": 0.13200758397579193, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 108, "step_time": 45.54669271316379 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.4970054626464844, "epoch": 0.12428734321550741, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0016885660588741302, "kl": 0.0008108019828796387, "learning_rate": 4.992077062730485e-06, "loss": 0.0, "num_tokens": 35079752.0, "reward": 0.03105469048023224, "reward_std": 0.2673572599887848, "rewards/code_complexity_reward/mean": 0.012500000186264515, "rewards/code_complexity_reward/std": 0.10644587129354477, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 109, "step_time": 45.07953772414476 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 512.0, "completions/min_length": 512.0, "completions/min_terminated_length": 512.0, "entropy": 0.5094947814941406, "epoch": 0.12542759407069556, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0013806310016661882, "kl": 0.0008247643709182739, "learning_rate": 4.991265434865726e-06, "loss": 0.0, "num_tokens": 35400620.0, "reward": 0.0146484375, "reward_std": 0.17167378962039948, "rewards/code_complexity_reward/mean": 0.006835937034338713, "rewards/code_complexity_reward/std": 0.07720755785703659, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 110, "step_time": 44.69590571988374 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5085258483886719, "epoch": 0.1265678449258837, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0016849417006596923, "kl": 0.0008086860179901123, "learning_rate": 4.990414309904781e-06, "loss": 0.0, "num_tokens": 35724236.0, "reward": 0.02177734300494194, "reward_std": 0.22332078218460083, "rewards/code_complexity_reward/mean": 0.00908203050494194, "rewards/code_complexity_reward/std": 0.09157534688711166, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 111, "step_time": 45.085434310138226 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 511.87890625, "completions/mean_terminated_length": 481.0, "completions/min_length": 478.0, "completions/min_terminated_length": 478.0, "entropy": 0.49953233264386654, "epoch": 0.12770809578107184, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.0019491632701829076, "kl": 0.0007984147659954033, "learning_rate": 4.98952370134158e-06, "loss": 0.0001, "num_tokens": 36044658.0, "reward": 0.04677734524011612, "reward_std": 0.33186060190200806, "rewards/code_complexity_reward/mean": 0.01748046651482582, "rewards/code_complexity_reward/std": 0.12426731735467911, "rewards/code_execution_reward/mean": 0.01953125, "rewards/code_execution_reward/std": 0.1385180652141571, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 112, "step_time": 50.1518177902326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5146026611328125, "epoch": 0.12884834663625996, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0016642095288261771, "kl": 0.0008296370506286621, "learning_rate": 4.988593623296038e-06, "loss": 0.0, "num_tokens": 36366194.0, "reward": 0.02421875111758709, "reward_std": 0.22933021187782288, "rewards/code_complexity_reward/mean": 0.010546875186264515, "rewards/code_complexity_reward/std": 0.09710130095481873, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 113, "step_time": 45.306757321581244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5112361907958984, "epoch": 0.12998859749144812, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0012394506484270096, "kl": 0.0008357316255569458, "learning_rate": 4.987624090513825e-06, "loss": 0.0, "num_tokens": 36688586.0, "reward": 0.014160157181322575, "reward_std": 0.1846339851617813, "rewards/code_complexity_reward/mean": 0.00537109375, "rewards/code_complexity_reward/std": 0.07005351036787033, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 114, "step_time": 51.31405379995704 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.990234375, "completions/mean_terminated_length": 507.0, "completions/min_length": 507.0, "completions/min_terminated_length": 507.0, "entropy": 0.5308621875010431, "epoch": 0.13112884834663627, "frac_reward_zero_std": 0.984375, "grad_norm": 0.001028083497658372, "kl": 0.0008397088540732511, "learning_rate": 4.986615118366138e-06, "loss": 0.0, "num_tokens": 37010557.0, "reward": 0.00732421875, "reward_std": 0.12247475236654282, "rewards/code_complexity_reward/mean": 0.00341796875, "rewards/code_complexity_reward/std": 0.05483507737517357, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 115, "step_time": 45.03840523213148 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5136184692382812, "epoch": 0.1322690992018244, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0007663095602765679, "kl": 0.0008162111043930054, "learning_rate": 4.985566722849454e-06, "loss": 0.0, "num_tokens": 37332585.0, "reward": 0.004687500186264515, "reward_std": 0.1060660183429718, "rewards/code_complexity_reward/mean": 0.0017578124534338713, "rewards/code_complexity_reward/std": 0.039774756878614426, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 116, "step_time": 45.005584380589426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5200920104980469, "epoch": 0.13340935005701254, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0007265130407176912, "kl": 0.0008261650800704956, "learning_rate": 4.984478920585277e-06, "loss": 0.0, "num_tokens": 37653393.0, "reward": 0.0047851563431322575, "reward_std": 0.10827571898698807, "rewards/code_complexity_reward/mean": 0.0018554687267169356, "rewards/code_complexity_reward/std": 0.04198446497321129, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 117, "step_time": 50.66639079526067 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5183658599853516, "epoch": 0.1345496009122007, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.001119456603191793, "kl": 0.0008418411016464233, "learning_rate": 4.983351728819874e-06, "loss": 0.0, "num_tokens": 37974849.0, "reward": 0.01435546949505806, "reward_std": 0.18717169761657715, "rewards/code_complexity_reward/mean": 0.005566406063735485, "rewards/code_complexity_reward/std": 0.07257677614688873, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 118, "step_time": 50.76223050896078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 511.857421875, "completions/mean_terminated_length": 439.0, "completions/min_length": 439.0, "completions/min_terminated_length": 439.0, "entropy": 0.5131807867437601, "epoch": 0.13568985176738882, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014368777628988028, "kl": 0.0008302947380798287, "learning_rate": 4.9821851654240025e-06, "loss": 0.0004, "num_tokens": 38295032.0, "reward": 0.01904296875, "reward_std": 0.21482178568840027, "rewards/code_complexity_reward/mean": 0.00732421875, "rewards/code_complexity_reward/std": 0.08264267444610596, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 119, "step_time": 44.80542318802327 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.990234375, "completions/mean_terminated_length": 507.0, "completions/min_length": 507.0, "completions/min_terminated_length": 507.0, "entropy": 0.5091308504343033, "epoch": 0.13683010262257697, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001425169873982668, "kl": 0.0008174032855094993, "learning_rate": 4.98097924889263e-06, "loss": 0.0, "num_tokens": 38617711.0, "reward": 0.0166015625, "reward_std": 0.19049778580665588, "rewards/code_complexity_reward/mean": 0.0068359375, "rewards/code_complexity_reward/std": 0.077397421002388, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 120, "step_time": 45.88382957410067 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5152244567871094, "epoch": 0.1379703534777651, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.001760784536600113, "kl": 0.0008256286382675171, "learning_rate": 4.979733998344632e-06, "loss": 0.0, "num_tokens": 38939667.0, "reward": 0.01806640625, "reward_std": 0.1879458725452423, "rewards/code_complexity_reward/mean": 0.00927734375, "rewards/code_complexity_reward/std": 0.09356426447629929, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 121, "step_time": 44.56587606202811 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5068550109863281, "epoch": 0.13911060433295325, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0015204038936644793, "kl": 0.0008173882961273193, "learning_rate": 4.9784494335225e-06, "loss": 0.0, "num_tokens": 39259139.0, "reward": 0.02177734486758709, "reward_std": 0.2228822112083435, "rewards/code_complexity_reward/mean": 0.00908203050494194, "rewards/code_complexity_reward/std": 0.09157534688711166, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 122, "step_time": 50.336371577344835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5035552978515625, "epoch": 0.1402508551881414, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001359418616630137, "kl": 0.0008091181516647339, "learning_rate": 4.977125574792018e-06, "loss": 0.0, "num_tokens": 39581895.0, "reward": 0.02841797098517418, "reward_std": 0.2612592577934265, "rewards/code_complexity_reward/mean": 0.01083984412252903, "rewards/code_complexity_reward/std": 0.09972897171974182, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 123, "step_time": 51.16977543011308 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5104866027832031, "epoch": 0.14139110604332952, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.001472910982556641, "kl": 0.0008240044116973877, "learning_rate": 4.975762443141949e-06, "loss": 0.0, "num_tokens": 39903307.0, "reward": 0.01416015625, "reward_std": 0.18466046452522278, "rewards/code_complexity_reward/mean": 0.00537109375, "rewards/code_complexity_reward/std": 0.07012330740690231, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 124, "step_time": 45.60321255773306 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 511.69921875, "completions/mean_terminated_length": 358.0, "completions/min_length": 358.0, "completions/min_terminated_length": 358.0, "entropy": 0.5090659051202238, "epoch": 0.14253135689851767, "frac_reward_zero_std": 0.953125, "grad_norm": 0.001815532916225493, "kl": 0.000809515904620639, "learning_rate": 4.9743600601836974e-06, "loss": 0.0005, "num_tokens": 40225613.0, "reward": 0.03359375149011612, "reward_std": 0.27238234877586365, "rewards/code_complexity_reward/mean": 0.01406249962747097, "rewards/code_complexity_reward/std": 0.11181432753801346, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 125, "step_time": 50.643147067166865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 511.99609375, "completions/mean_terminated_length": 510.0, "completions/min_length": 510.0, "completions/min_terminated_length": 510.0, "entropy": 0.5166338076815009, "epoch": 0.14367160775370583, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002214421285316348, "kl": 0.0008212990032916423, "learning_rate": 4.9729184481509644e-06, "loss": 0.0, "num_tokens": 40547891.0, "reward": 0.04345703125, "reward_std": 0.3039388656616211, "rewards/code_complexity_reward/mean": 0.01904296875, "rewards/code_complexity_reward/std": 0.12963013350963593, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 126, "step_time": 45.2356364056468 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5139732360839844, "epoch": 0.14481185860889395, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0010790930828079581, "kl": 0.0008401274681091309, "learning_rate": 4.971437629899399e-06, "loss": 0.0, "num_tokens": 40867615.0, "reward": 0.007617187686264515, "reward_std": 0.125709667801857, "rewards/code_complexity_reward/mean": 0.0037109374534338713, "rewards/code_complexity_reward/std": 0.05931687355041504, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 127, "step_time": 45.314213016070426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5177040100097656, "epoch": 0.1459521094640821, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014088472817093134, "kl": 0.000829845666885376, "learning_rate": 4.969917628906234e-06, "loss": 0.0, "num_tokens": 41191107.0, "reward": 0.01865234412252903, "reward_std": 0.2104387879371643, "rewards/code_complexity_reward/mean": 0.006933593191206455, "rewards/code_complexity_reward/std": 0.07830001413822174, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 128, "step_time": 44.85944141820073 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 511.865234375, "completions/mean_terminated_length": 443.0, "completions/min_length": 443.0, "completions/min_terminated_length": 443.0, "entropy": 0.508387559093535, "epoch": 0.14709236031927023, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0017566318856552243, "kl": 0.0008136915967043024, "learning_rate": 4.968358469269917e-06, "loss": 0.0002, "num_tokens": 41514270.0, "reward": 0.04501952975988388, "reward_std": 0.32185494899749756, "rewards/code_complexity_reward/mean": 0.01767578162252903, "rewards/code_complexity_reward/std": 0.1256103217601776, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 129, "step_time": 46.628164794296026 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5168113708496094, "epoch": 0.14823261117445838, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0013751458609476686, "kl": 0.000826910138130188, "learning_rate": 4.966760175709725e-06, "loss": 0.0, "num_tokens": 41836654.0, "reward": 0.03632812574505806, "reward_std": 0.2915787994861603, "rewards/code_complexity_reward/mean": 0.014843749813735485, "rewards/code_complexity_reward/std": 0.11793383210897446, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 130, "step_time": 51.01861369237304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5096931457519531, "epoch": 0.14937286202964653, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014400240033864975, "kl": 0.0008255541324615479, "learning_rate": 4.96512277356537e-06, "loss": 0.0, "num_tokens": 42158822.0, "reward": 0.02158203162252903, "reward_std": 0.22265969216823578, "rewards/code_complexity_reward/mean": 0.00888671912252903, "rewards/code_complexity_reward/std": 0.08992348611354828, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 131, "step_time": 45.44392273016274 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 511.98828125, "completions/mean_terminated_length": 506.0, "completions/min_length": 506.0, "completions/min_terminated_length": 506.0, "entropy": 0.5157711571082473, "epoch": 0.15051311288483465, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0015799105167388916, "kl": 0.0008461144952889299, "learning_rate": 4.963446288796605e-06, "loss": 0.0, "num_tokens": 42480872.0, "reward": 0.0283203125, "reward_std": 0.2603413164615631, "rewards/code_complexity_reward/mean": 0.0107421875, "rewards/code_complexity_reward/std": 0.09877842664718628, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 132, "step_time": 50.48934603109956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5108833312988281, "epoch": 0.1516533637400228, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0014840296935290098, "kl": 0.0008135735988616943, "learning_rate": 4.961730747982804e-06, "loss": 0.0, "num_tokens": 42803360.0, "reward": 0.02167968824505806, "reward_std": 0.22228732705116272, "rewards/code_complexity_reward/mean": 0.008984374813735485, "rewards/code_complexity_reward/std": 0.09064534306526184, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 133, "step_time": 45.634389768354595 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 511.896484375, "completions/mean_terminated_length": 459.0, "completions/min_length": 459.0, "completions/min_terminated_length": 459.0, "entropy": 0.4990629553794861, "epoch": 0.15279361459521096, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0018501834711059928, "kl": 0.0008262104656751035, "learning_rate": 4.9599761783225465e-06, "loss": 0.0003, "num_tokens": 43123895.0, "reward": 0.02939453162252903, "reward_std": 0.25566041469573975, "rewards/code_complexity_reward/mean": 0.01279296912252903, "rewards/code_complexity_reward/std": 0.10888786613941193, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 134, "step_time": 50.69638724159449 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5127906799316406, "epoch": 0.15393386545039908, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.002020485932007432, "kl": 0.0008588284254074097, "learning_rate": 4.9581826076331854e-06, "loss": 0.0, "num_tokens": 43444979.0, "reward": 0.04531250149011612, "reward_std": 0.32422947883605957, "rewards/code_complexity_reward/mean": 0.01796874962747097, "rewards/code_complexity_reward/std": 0.12748266756534576, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 135, "step_time": 45.66088761296123 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5175323486328125, "epoch": 0.15507411630558723, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.002025904133915901, "kl": 0.0008394867181777954, "learning_rate": 4.956350064350403e-06, "loss": 0.0, "num_tokens": 43767115.0, "reward": 0.03017578087747097, "reward_std": 0.24799107015132904, "rewards/code_complexity_reward/mean": 0.01455078087747097, "rewards/code_complexity_reward/std": 0.11564585566520691, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 136, "step_time": 44.90792300179601 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5122718811035156, "epoch": 0.15621436716077536, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0010483484948053956, "kl": 0.0008419007062911987, "learning_rate": 4.954478577527761e-06, "loss": 0.0, "num_tokens": 44087899.0, "reward": 0.014062500558793545, "reward_std": 0.18335185945034027, "rewards/code_complexity_reward/mean": 0.00527343712747097, "rewards/code_complexity_reward/std": 0.06875694543123245, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 137, "step_time": 45.404835814610124 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.49855804443359375, "epoch": 0.1573546180159635, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0017122459830716252, "kl": 0.0008249878883361816, "learning_rate": 4.952568176836246e-06, "loss": 0.0, "num_tokens": 44411367.0, "reward": 0.04062499850988388, "reward_std": 0.30669310688972473, "rewards/code_complexity_reward/mean": 0.01621093600988388, "rewards/code_complexity_reward/std": 0.12135446816682816, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 138, "step_time": 51.69954443629831 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5169143676757812, "epoch": 0.15849486887115166, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0015655140159651637, "kl": 0.0008568912744522095, "learning_rate": 4.9506188925637885e-06, "loss": 0.0, "num_tokens": 44734879.0, "reward": 0.03330077975988388, "reward_std": 0.28313565254211426, "rewards/code_complexity_reward/mean": 0.012792968191206455, "rewards/code_complexity_reward/std": 0.10879796743392944, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 139, "step_time": 44.76744126994163 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5211658477783203, "epoch": 0.15963511972633979, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0015596316661685705, "kl": 0.0008570700883865356, "learning_rate": 4.948630755614792e-06, "loss": 0.0, "num_tokens": 45054739.0, "reward": 0.02314453013241291, "reward_std": 0.233342245221138, "rewards/code_complexity_reward/mean": 0.008496093563735485, "rewards/code_complexity_reward/std": 0.08578567206859589, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 140, "step_time": 45.05908948928118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5267276763916016, "epoch": 0.16077537058152794, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0015306572895497084, "kl": 0.000849410891532898, "learning_rate": 4.946603797509635e-06, "loss": 0.0, "num_tokens": 45376431.0, "reward": 0.02617187425494194, "reward_std": 0.24362435936927795, "rewards/code_complexity_reward/mean": 0.010546875186264515, "rewards/code_complexity_reward/std": 0.09715167433023453, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 141, "step_time": 44.62845978233963 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5191802978515625, "epoch": 0.1619156214367161, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0007574994233436882, "kl": 0.0008433312177658081, "learning_rate": 4.944538050384181e-06, "loss": 0.0, "num_tokens": 45699155.0, "reward": 0.00937500037252903, "reward_std": 0.1498531550168991, "rewards/code_complexity_reward/mean": 0.0035156249068677425, "rewards/code_complexity_reward/std": 0.056194934993982315, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 142, "step_time": 45.38007913995534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5111465454101562, "epoch": 0.1630558722919042, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001752528827637434, "kl": 0.0008470714092254639, "learning_rate": 4.94243354698926e-06, "loss": 0.0, "num_tokens": 46020743.0, "reward": 0.02158203162252903, "reward_std": 0.22038491070270538, "rewards/code_complexity_reward/mean": 0.008886718191206455, "rewards/code_complexity_reward/std": 0.08976011723279953, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 143, "step_time": 50.690470767207444 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 511.970703125, "completions/mean_terminated_length": 497.0, "completions/min_length": 497.0, "completions/min_terminated_length": 497.0, "entropy": 0.5124783888459206, "epoch": 0.16419612314709237, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0016272732755169272, "kl": 0.0008478434510834632, "learning_rate": 4.9402903206901535e-06, "loss": 0.0001, "num_tokens": 46343620.0, "reward": 0.03789062798023224, "reward_std": 0.30107414722442627, "rewards/code_complexity_reward/mean": 0.014453125186264515, "rewards/code_complexity_reward/std": 0.11491549760103226, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 144, "step_time": 45.341447808779776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.49447059631347656, "epoch": 0.1653363740022805, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0010925206588581204, "kl": 0.0008126944303512573, "learning_rate": 4.938108405466065e-06, "loss": 0.0, "num_tokens": 46665480.0, "reward": 0.00937500037252903, "reward_std": 0.1498531550168991, "rewards/code_complexity_reward/mean": 0.0035156249068677425, "rewards/code_complexity_reward/std": 0.05619493126869202, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 145, "step_time": 45.092533267103136 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5182685852050781, "epoch": 0.16647662485746864, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0011238281149417162, "kl": 0.0008600950241088867, "learning_rate": 4.935887835909581e-06, "loss": 0.0, "num_tokens": 46988144.0, "reward": 0.01425781287252903, "reward_std": 0.18590719997882843, "rewards/code_complexity_reward/mean": 0.0054687499068677425, "rewards/code_complexity_reward/std": 0.0713263750076294, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 146, "step_time": 44.73334057535976 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5115547180175781, "epoch": 0.1676168757126568, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0009299111552536488, "kl": 0.000846564769744873, "learning_rate": 4.933628647226123e-06, "loss": 0.0, "num_tokens": 47311264.0, "reward": 0.00947265699505806, "reward_std": 0.15142220258712769, "rewards/code_complexity_reward/mean": 0.003613281063735485, "rewards/code_complexity_reward/std": 0.057777076959609985, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 147, "step_time": 50.52522438112646 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5163078308105469, "epoch": 0.16875712656784492, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001493960153311491, "kl": 0.0008568465709686279, "learning_rate": 4.93133087523339e-06, "loss": 0.0, "num_tokens": 47631616.0, "reward": 0.02617187425494194, "reward_std": 0.24360427260398865, "rewards/code_complexity_reward/mean": 0.010546875186264515, "rewards/code_complexity_reward/std": 0.09710130095481873, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 148, "step_time": 45.13776252232492 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5097770690917969, "epoch": 0.16989737742303307, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0016446707304567099, "kl": 0.0008662045001983643, "learning_rate": 4.928994556360787e-06, "loss": 0.0, "num_tokens": 47951632.0, "reward": 0.01914062537252903, "reward_std": 0.21591484546661377, "rewards/code_complexity_reward/mean": 0.0074218749068677425, "rewards/code_complexity_reward/std": 0.0837220847606659, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 149, "step_time": 45.616613923572004 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5037651062011719, "epoch": 0.17103762827822122, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002112990478053689, "kl": 0.0008357316255569458, "learning_rate": 4.926619727648852e-06, "loss": 0.0, "num_tokens": 48272744.0, "reward": 0.05205078050494194, "reward_std": 0.35165345668792725, "rewards/code_complexity_reward/mean": 0.01982421800494194, "rewards/code_complexity_reward/std": 0.1340056210756302, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 150, "step_time": 51.258536946959794 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 511.912109375, "completions/mean_terminated_length": 467.0, "completions/min_length": 467.0, "completions/min_terminated_length": 467.0, "entropy": 0.5169482175260782, "epoch": 0.17217787913340935, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0010946865659207106, "kl": 0.0008528277248842642, "learning_rate": 4.924206426748668e-06, "loss": -0.0, "num_tokens": 48595315.0, "reward": 0.015136719681322575, "reward_std": 0.17667394876480103, "rewards/code_complexity_reward/mean": 0.00732421875, "rewards/code_complexity_reward/std": 0.08264267444610596, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 151, "step_time": 45.402513169683516 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 511.970703125, "completions/mean_terminated_length": 504.5, "completions/min_length": 500.0, "completions/min_terminated_length": 500.0, "entropy": 0.5083491876721382, "epoch": 0.1733181299885975, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.002083960920572281, "kl": 0.0008521317104168702, "learning_rate": 4.921754691921262e-06, "loss": 0.0001, "num_tokens": 48915652.0, "reward": 0.03632812574505806, "reward_std": 0.27970942854881287, "rewards/code_complexity_reward/mean": 0.01582031324505806, "rewards/code_complexity_reward/std": 0.11851044744253159, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 152, "step_time": 45.4901397228241 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 511.94921875, "completions/mean_terminated_length": 486.0, "completions/min_length": 486.0, "completions/min_terminated_length": 486.0, "entropy": 0.5095183760859072, "epoch": 0.17445838084378562, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0010628963354974985, "kl": 0.0008552065846743062, "learning_rate": 4.919264562037003e-06, "loss": 0.0002, "num_tokens": 49235690.0, "reward": 0.00722656212747097, "reward_std": 0.1196722611784935, "rewards/code_complexity_reward/mean": 0.0033203125931322575, "rewards/code_complexity_reward/std": 0.05307299271225929, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 153, "step_time": 45.389462077990174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 511.849609375, "completions/mean_terminated_length": 435.0, "completions/min_length": 435.0, "completions/min_terminated_length": 435.0, "entropy": 0.503321435302496, "epoch": 0.17559863169897377, "frac_reward_zero_std": 0.984375, "grad_norm": 0.001089703757315874, "kl": 0.0008390088532905793, "learning_rate": 4.9167360765749845e-06, "loss": 0.0004, "num_tokens": 49557621.0, "reward": 0.00947265699505806, "reward_std": 0.15142220258712769, "rewards/code_complexity_reward/mean": 0.003613281063735485, "rewards/code_complexity_reward/std": 0.05777707323431969, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 154, "step_time": 50.75770638138056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5173835754394531, "epoch": 0.17673888255416192, "frac_reward_zero_std": 0.984375, "grad_norm": 0.001026336569339037, "kl": 0.0008803009986877441, "learning_rate": 4.914169275622397e-06, "loss": 0.0, "num_tokens": 49879801.0, "reward": 0.03349609300494194, "reward_std": 0.28478386998176575, "rewards/code_complexity_reward/mean": 0.01298828050494194, "rewards/code_complexity_reward/std": 0.1104263886809349, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 155, "step_time": 44.73164187837392 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5238246917724609, "epoch": 0.17787913340935005, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.00173387979157269, "kl": 0.0008756071329116821, "learning_rate": 4.911564199873894e-06, "loss": 0.0, "num_tokens": 50201673.0, "reward": 0.01953125, "reward_std": 0.20291268825531006, "rewards/code_complexity_reward/mean": 0.0087890625, "rewards/code_complexity_reward/std": 0.08897601068019867, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 156, "step_time": 45.5778310764581 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 511.9296875, "completions/mean_terminated_length": 476.0, "completions/min_length": 476.0, "completions/min_terminated_length": 476.0, "entropy": 0.4927749950438738, "epoch": 0.1790193842645382, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0013344171456992626, "kl": 0.0008345635751538794, "learning_rate": 4.908920890630947e-06, "loss": 0.0002, "num_tokens": 50520861.0, "reward": 0.01230468787252903, "reward_std": 0.16485467553138733, "rewards/code_complexity_reward/mean": 0.0054687499068677425, "rewards/code_complexity_reward/std": 0.0713263675570488, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 157, "step_time": 45.536417414434254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5094356536865234, "epoch": 0.18015963511972635, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0017977970419451594, "kl": 0.0008627623319625854, "learning_rate": 4.906239389801191e-06, "loss": 0.0, "num_tokens": 50842993.0, "reward": 0.04277344048023224, "reward_std": 0.320097416639328, "rewards/code_complexity_reward/mean": 0.01640624925494194, "rewards/code_complexity_reward/std": 0.12281107157468796, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 158, "step_time": 45.15818437654525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 511.90625, "completions/mean_terminated_length": 464.0, "completions/min_length": 464.0, "completions/min_terminated_length": 464.0, "entropy": 0.5046401843428612, "epoch": 0.18129988597491448, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.001981848618015647, "kl": 0.0008562890616303775, "learning_rate": 4.903519739897755e-06, "loss": 0.0002, "num_tokens": 51163057.0, "reward": 0.03388671949505806, "reward_std": 0.2627550959587097, "rewards/code_complexity_reward/mean": 0.015332031063735485, "rewards/code_complexity_reward/std": 0.11490780115127563, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 159, "step_time": 44.6368059925735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5205650329589844, "epoch": 0.18244013683010263, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.001693099969998002, "kl": 0.0008680820465087891, "learning_rate": 4.9007619840385975e-06, "loss": 0.0, "num_tokens": 51485013.0, "reward": 0.03134765475988388, "reward_std": 0.26955556869506836, "rewards/code_complexity_reward/mean": 0.012792968191206455, "rewards/code_complexity_reward/std": 0.10879797488451004, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 160, "step_time": 45.3606073083356 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5223903656005859, "epoch": 0.18358038768529075, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.001267051324248314, "kl": 0.0009025484323501587, "learning_rate": 4.897966165945815e-06, "loss": 0.0, "num_tokens": 51806009.0, "reward": 0.01210937462747097, "reward_std": 0.16145086288452148, "rewards/code_complexity_reward/mean": 0.0052734375931322575, "rewards/code_complexity_reward/std": 0.06897008419036865, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 161, "step_time": 45.058435768820345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5029735565185547, "epoch": 0.1847206385404789, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.001274747191928327, "kl": 0.0008481144905090332, "learning_rate": 4.8951323299449514e-06, "loss": 0.0, "num_tokens": 52126593.0, "reward": 0.012011718936264515, "reward_std": 0.16180627048015594, "rewards/code_complexity_reward/mean": 0.005175781436264515, "rewards/code_complexity_reward/std": 0.0676526203751564, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 162, "step_time": 51.23715472780168 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5268821716308594, "epoch": 0.18586088939566706, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0012159274192526937, "kl": 0.0008790940046310425, "learning_rate": 4.892260520964295e-06, "loss": 0.0, "num_tokens": 52448521.0, "reward": 0.02167968824505806, "reward_std": 0.22182464599609375, "rewards/code_complexity_reward/mean": 0.008984374813735485, "rewards/code_complexity_reward/std": 0.09059135615825653, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 163, "step_time": 45.26611603889614 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5156097412109375, "epoch": 0.18700114025085518, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0016372904647141695, "kl": 0.0008674710988998413, "learning_rate": 4.889350784534168e-06, "loss": 0.0, "num_tokens": 52769501.0, "reward": 0.021484375, "reward_std": 0.2216009646654129, "rewards/code_complexity_reward/mean": 0.0087890625, "rewards/code_complexity_reward/std": 0.08892100304365158, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 164, "step_time": 45.14737217128277 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5094261169433594, "epoch": 0.18814139110604333, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0020395969040691853, "kl": 0.0008666813373565674, "learning_rate": 4.886403166786203e-06, "loss": 0.0, "num_tokens": 53092005.0, "reward": 0.04082031548023224, "reward_std": 0.3085438907146454, "rewards/code_complexity_reward/mean": 0.01640624925494194, "rewards/code_complexity_reward/std": 0.12285089492797852, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 165, "step_time": 51.29111194424331 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5016822814941406, "epoch": 0.18928164196123148, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001453536213375628, "kl": 0.0008566975593566895, "learning_rate": 4.883417714452607e-06, "loss": 0.0, "num_tokens": 53412021.0, "reward": 0.01875000074505806, "reward_std": 0.21157778799533844, "rewards/code_complexity_reward/mean": 0.007031249813735485, "rewards/code_complexity_reward/std": 0.0795004814863205, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 166, "step_time": 51.209510523825884 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 511.861328125, "completions/mean_terminated_length": 441.0, "completions/min_length": 441.0, "completions/min_terminated_length": 441.0, "entropy": 0.5171741736121476, "epoch": 0.1904218928164196, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0008707086089998484, "kl": 0.0008877603977452964, "learning_rate": 4.880394474865433e-06, "loss": 0.0004, "num_tokens": 53733746.0, "reward": 0.0047851563431322575, "reward_std": 0.10827571898698807, "rewards/code_complexity_reward/mean": 0.0018554687267169356, "rewards/code_complexity_reward/std": 0.04198446497321129, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 167, "step_time": 45.058390656486154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5212211608886719, "epoch": 0.19156214367160776, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0015360290417447686, "kl": 0.0008740723133087158, "learning_rate": 4.8773334959558165e-06, "loss": 0.0, "num_tokens": 54056230.0, "reward": 0.01220703125, "reward_std": 0.16341634094715118, "rewards/code_complexity_reward/mean": 0.00537109375, "rewards/code_complexity_reward/std": 0.07005351036787033, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 168, "step_time": 46.193039630539715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 512.0, "completions/min_length": 512.0, "completions/min_terminated_length": 512.0, "entropy": 0.5013370513916016, "epoch": 0.19270239452679588, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0015544253401458263, "kl": 0.0008657872676849365, "learning_rate": 4.874234826253223e-06, "loss": 0.0, "num_tokens": 54379590.0, "reward": 0.01435546949505806, "reward_std": 0.18717169761657715, "rewards/code_complexity_reward/mean": 0.005566406063735485, "rewards/code_complexity_reward/std": 0.07257678359746933, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 169, "step_time": 45.406352542340755 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5183982849121094, "epoch": 0.19384264538198404, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0013081238139420748, "kl": 0.0008715540170669556, "learning_rate": 4.871098514884675e-06, "loss": 0.0, "num_tokens": 54699846.0, "reward": 0.014160157181322575, "reward_std": 0.18463397026062012, "rewards/code_complexity_reward/mean": 0.00537109375, "rewards/code_complexity_reward/std": 0.07005351036787033, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 170, "step_time": 44.72631658799946 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 437.0, "completions/mean_length": 511.853515625, "completions/mean_terminated_length": 437.0, "completions/min_length": 437.0, "completions/min_terminated_length": 437.0, "entropy": 0.5141449412330985, "epoch": 0.1949828962371722, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002395492047071457, "kl": 0.0008878407479642192, "learning_rate": 4.867924611573977e-06, "loss": 0.0004, "num_tokens": 55019695.0, "reward": 0.05507812649011612, "reward_std": 0.3583141565322876, "rewards/code_complexity_reward/mean": 0.02187499962747097, "rewards/code_complexity_reward/std": 0.1413867473602295, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 171, "step_time": 50.97627993579954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5030307769775391, "epoch": 0.1961231470923603, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.001726021757349372, "kl": 0.0008632838726043701, "learning_rate": 4.864713166640921e-06, "loss": 0.0, "num_tokens": 55341091.0, "reward": 0.02373046986758709, "reward_std": 0.23920603096485138, "rewards/code_complexity_reward/mean": 0.00908203050494194, "rewards/code_complexity_reward/std": 0.09157534688711166, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 172, "step_time": 45.61157275736332 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.998046875, "completions/mean_terminated_length": 511.0, "completions/min_length": 511.0, "completions/min_terminated_length": 511.0, "entropy": 0.491765771061182, "epoch": 0.19726339794754846, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0017577604157850146, "kl": 0.000841739097268146, "learning_rate": 4.8614642310004975e-06, "loss": 0.0, "num_tokens": 55663190.0, "reward": 0.03261718899011612, "reward_std": 0.27739453315734863, "rewards/code_complexity_reward/mean": 0.012109375558793545, "rewards/code_complexity_reward/std": 0.10317770391702652, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 173, "step_time": 45.312715247273445 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 511.6953125, "completions/mean_terminated_length": 434.0, "completions/min_length": 419.0, "completions/min_terminated_length": 419.0, "entropy": 0.5003171116113663, "epoch": 0.19840364880273662, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002367063658311963, "kl": 0.0008496613827446708, "learning_rate": 4.8581778561620785e-06, "loss": 0.0005, "num_tokens": 55984110.0, "reward": 0.06162109225988388, "reward_std": 0.3821695148944855, "rewards/code_complexity_reward/mean": 0.02353515662252903, "rewards/code_complexity_reward/std": 0.14600954949855804, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.0126953125, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 174, "step_time": 46.07304630894214 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.499725341796875, "epoch": 0.19954389965792474, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0015188759425655007, "kl": 0.0008641183376312256, "learning_rate": 4.8548540942286095e-06, "loss": 0.0, "num_tokens": 56304914.0, "reward": 0.02353515662252903, "reward_std": 0.23727455735206604, "rewards/code_complexity_reward/mean": 0.00888671912252903, "rewards/code_complexity_reward/std": 0.08970560133457184, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 175, "step_time": 45.35088956169784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.927734375, "completions/mean_terminated_length": 499.66668701171875, "completions/min_length": 482.0, "completions/min_terminated_length": 482.0, "entropy": 0.5263161924667656, "epoch": 0.2006841505131129, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0020662443712353706, "kl": 0.0009049052387126721, "learning_rate": 4.851492997895777e-06, "loss": 0.0002, "num_tokens": 56626889.0, "reward": 0.03115234524011612, "reward_std": 0.26783037185668945, "rewards/code_complexity_reward/mean": 0.01259765587747097, "rewards/code_complexity_reward/std": 0.1071900874376297, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 176, "step_time": 50.347594984807074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5107345581054688, "epoch": 0.20182440136830102, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0013481603236868978, "kl": 0.0008828341960906982, "learning_rate": 4.848094620451177e-06, "loss": 0.0, "num_tokens": 56948893.0, "reward": 0.013964843936264515, "reward_std": 0.1821144074201584, "rewards/code_complexity_reward/mean": 0.005175781436264515, "rewards/code_complexity_reward/std": 0.06758026778697968, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 177, "step_time": 50.68386408966035 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5128936767578125, "epoch": 0.20296465222348917, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014401464723050594, "kl": 0.0008693784475326538, "learning_rate": 4.844659015773468e-06, "loss": 0.0, "num_tokens": 57268601.0, "reward": 0.01689453050494194, "reward_std": 0.19560416042804718, "rewards/code_complexity_reward/mean": 0.007128906436264515, "rewards/code_complexity_reward/std": 0.08062233030796051, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 178, "step_time": 45.58310294058174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5092144012451172, "epoch": 0.20410490307867732, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0018531496170908213, "kl": 0.0008788704872131348, "learning_rate": 4.841186238331519e-06, "loss": 0.0, "num_tokens": 57591025.0, "reward": 0.02773437649011612, "reward_std": 0.2550514042377472, "rewards/code_complexity_reward/mean": 0.01015624962747097, "rewards/code_complexity_reward/std": 0.0936557725071907, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 179, "step_time": 51.564099095761776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.51348876953125, "epoch": 0.20524515393386544, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0016823490150272846, "kl": 0.0008853226900100708, "learning_rate": 4.8376763431835424e-06, "loss": 0.0, "num_tokens": 57914573.0, "reward": 0.02939453348517418, "reward_std": 0.2552582323551178, "rewards/code_complexity_reward/mean": 0.01279296912252903, "rewards/code_complexity_reward/std": 0.10884292423725128, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 180, "step_time": 45.6275467062369 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 511.9921875, "completions/mean_terminated_length": 508.0, "completions/min_length": 508.0, "completions/min_terminated_length": 508.0, "entropy": 0.5101404264569283, "epoch": 0.2063854047890536, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0017773471772670746, "kl": 0.0008947480855567846, "learning_rate": 4.834129385976227e-06, "loss": 0.0, "num_tokens": 58234937.0, "reward": 0.02158203162252903, "reward_std": 0.22215375304222107, "rewards/code_complexity_reward/mean": 0.00888671912252903, "rewards/code_complexity_reward/std": 0.08976011723279953, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 181, "step_time": 45.055661195889115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.50537109375, "epoch": 0.20752565564424175, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.0018379485700279474, "kl": 0.0009006708860397339, "learning_rate": 4.830545422943847e-06, "loss": 0.0, "num_tokens": 58554041.0, "reward": 0.04033203050494194, "reward_std": 0.30481985211372375, "rewards/code_complexity_reward/mean": 0.01591796800494194, "rewards/code_complexity_reward/std": 0.11938171088695526, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 182, "step_time": 45.39971403311938 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 511.9921875, "completions/mean_terminated_length": 508.0, "completions/min_length": 508.0, "completions/min_terminated_length": 508.0, "entropy": 0.5033586458303034, "epoch": 0.20866590649942987, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0013278094120323658, "kl": 0.0008795808153081452, "learning_rate": 4.8269245109073795e-06, "loss": 0.0, "num_tokens": 58875069.0, "reward": 0.03095703199505806, "reward_std": 0.26642459630966187, "rewards/code_complexity_reward/mean": 0.012402343563735485, "rewards/code_complexity_reward/std": 0.1054646298289299, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 183, "step_time": 45.6072296788916 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5094375610351562, "epoch": 0.20980615735461802, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0017004848923534155, "kl": 0.0008769780397415161, "learning_rate": 4.823266707273596e-06, "loss": 0.0, "num_tokens": 59197337.0, "reward": 0.03134765475988388, "reward_std": 0.26955556869506836, "rewards/code_complexity_reward/mean": 0.012792968191206455, "rewards/code_complexity_reward/std": 0.10879797488451004, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 184, "step_time": 45.6240591686219 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5073318481445312, "epoch": 0.21094640820980615, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.001666470430791378, "kl": 0.0009032636880874634, "learning_rate": 4.819572070034162e-06, "loss": 0.0, "num_tokens": 59519021.0, "reward": 0.04335937649011612, "reward_std": 0.312844455242157, "rewards/code_complexity_reward/mean": 0.01796874962747097, "rewards/code_complexity_reward/std": 0.12752103805541992, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 185, "step_time": 44.97440150659531 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5105819702148438, "epoch": 0.2120866590649943, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0009223066153936088, "kl": 0.0008943825960159302, "learning_rate": 4.815840657764704e-06, "loss": 0.0, "num_tokens": 59840937.0, "reward": 0.007519531529396772, "reward_std": 0.12381374835968018, "rewards/code_complexity_reward/mean": 0.003613281063735485, "rewards/code_complexity_reward/std": 0.057777076959609985, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 186, "step_time": 46.68491775076836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.49814605712890625, "epoch": 0.21322690992018245, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0008894691709429026, "kl": 0.0009068399667739868, "learning_rate": 4.812072529623894e-06, "loss": 0.0, "num_tokens": 60160721.0, "reward": 0.0047851563431322575, "reward_std": 0.10827571898698807, "rewards/code_complexity_reward/mean": 0.0018554687267169356, "rewards/code_complexity_reward/std": 0.04198446497321129, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 187, "step_time": 45.72068732790649 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 511.728515625, "completions/mean_terminated_length": 373.0, "completions/min_length": 373.0, "completions/min_terminated_length": 373.0, "entropy": 0.507531838491559, "epoch": 0.21436716077537057, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.00162513495888561, "kl": 0.0009090428220588365, "learning_rate": 4.808267745352502e-06, "loss": 0.0008, "num_tokens": 60481742.0, "reward": 0.02382812649011612, "reward_std": 0.24018622934818268, "rewards/code_complexity_reward/mean": 0.00917968712747097, "rewards/code_complexity_reward/std": 0.09254876524209976, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 188, "step_time": 45.2843914180994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 511.890625, "completions/mean_terminated_length": 456.0, "completions/min_length": 456.0, "completions/min_terminated_length": 456.0, "entropy": 0.5157494125887752, "epoch": 0.21550741163055873, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014869224978610873, "kl": 0.0009206091626765556, "learning_rate": 4.804426365272455e-06, "loss": 0.0003, "num_tokens": 60803514.0, "reward": 0.01894531399011612, "reward_std": 0.2137230783700943, "rewards/code_complexity_reward/mean": 0.00722656212747097, "rewards/code_complexity_reward/std": 0.08154886960983276, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 189, "step_time": 46.11061259265989 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.904296875, "completions/mean_terminated_length": 487.5, "completions/min_length": 468.0, "completions/min_terminated_length": 468.0, "entropy": 0.5192366656847298, "epoch": 0.21664766248574688, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0016512839356437325, "kl": 0.0009060854263225337, "learning_rate": 4.800548450285878e-06, "loss": 0.0003, "num_tokens": 61125849.0, "reward": 0.02851562574505806, "reward_std": 0.2621552646160126, "rewards/code_complexity_reward/mean": 0.010937499813735485, "rewards/code_complexity_reward/std": 0.10062184184789658, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 190, "step_time": 45.60598842334002 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5127906799316406, "epoch": 0.217787913340935, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0022634633351117373, "kl": 0.0009129196405410767, "learning_rate": 4.79663406187413e-06, "loss": 0.0, "num_tokens": 61447361.0, "reward": 0.02480468899011612, "reward_std": 0.233265683054924, "rewards/code_complexity_reward/mean": 0.01113281212747097, "rewards/code_complexity_reward/std": 0.10238393396139145, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 191, "step_time": 45.97206365503371 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 511.849609375, "completions/mean_terminated_length": 435.0, "completions/min_length": 435.0, "completions/min_terminated_length": 435.0, "entropy": 0.5155760096386075, "epoch": 0.21892816419612315, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014008003054186702, "kl": 0.0009461857807764318, "learning_rate": 4.792683262096825e-06, "loss": 0.0004, "num_tokens": 61768880.0, "reward": 0.02158203348517418, "reward_std": 0.2212047576904297, "rewards/code_complexity_reward/mean": 0.008886718191206455, "rewards/code_complexity_reward/std": 0.0895964577794075, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 192, "step_time": 45.736316714435816 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5126914978027344, "epoch": 0.22006841505131128, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0006203366792760789, "kl": 0.0009405910968780518, "learning_rate": 4.788696113590853e-06, "loss": 0.0, "num_tokens": 62091720.0, "reward": 0.004687500186264515, "reward_std": 0.1060660183429718, "rewards/code_complexity_reward/mean": 0.0017578124534338713, "rewards/code_complexity_reward/std": 0.039774756878614426, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 193, "step_time": 45.832631099037826 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5007076263427734, "epoch": 0.22120866590649943, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0011073497589677572, "kl": 0.0009206831455230713, "learning_rate": 4.7846726795693855e-06, "loss": 0.0, "num_tokens": 62414712.0, "reward": 0.00742187537252903, "reward_std": 0.12352295219898224, "rewards/code_complexity_reward/mean": 0.0035156249068677425, "rewards/code_complexity_reward/std": 0.056281927973032, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 194, "step_time": 51.99448548350483 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5132884979248047, "epoch": 0.22234891676168758, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0019204795826226473, "kl": 0.0009320974349975586, "learning_rate": 4.780613023820872e-06, "loss": 0.0, "num_tokens": 62735824.0, "reward": 0.02382812649011612, "reward_std": 0.24018622934818268, "rewards/code_complexity_reward/mean": 0.00917968712747097, "rewards/code_complexity_reward/std": 0.09254876524209976, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 195, "step_time": 50.63937941286713 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 511.861328125, "completions/mean_terminated_length": 441.0, "completions/min_length": 441.0, "completions/min_terminated_length": 441.0, "entropy": 0.5313697000965476, "epoch": 0.2234891676168757, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0007599671953357756, "kl": 0.0009729499652166851, "learning_rate": 4.776517210708032e-06, "loss": 0.0004, "num_tokens": 63058753.0, "reward": 0.0047851563431322575, "reward_std": 0.10827571898698807, "rewards/code_complexity_reward/mean": 0.0018554687267169356, "rewards/code_complexity_reward/std": 0.04198446497321129, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 196, "step_time": 46.03139459900558 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.7265625, "completions/mean_terminated_length": 442.0, "completions/min_length": 377.0, "completions/min_terminated_length": 377.0, "entropy": 0.5148735390976071, "epoch": 0.22462941847206386, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0018812192138284445, "kl": 0.0009306077472501784, "learning_rate": 4.772385305166828e-06, "loss": 0.0003, "num_tokens": 63380957.0, "reward": 0.03554687649011612, "reward_std": 0.2758301794528961, "rewards/code_complexity_reward/mean": 0.01503906212747097, "rewards/code_complexity_reward/std": 0.11329609155654907, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 197, "step_time": 46.18873414862901 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5067195892333984, "epoch": 0.22576966932725198, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014824041863903403, "kl": 0.0009524673223495483, "learning_rate": 4.768217372705442e-06, "loss": 0.0, "num_tokens": 63702577.0, "reward": 0.01708984375, "reward_std": 0.19643577933311462, "rewards/code_complexity_reward/mean": 0.00732421875, "rewards/code_complexity_reward/std": 0.08264267444610596, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 198, "step_time": 45.80924357101321 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 511.994140625, "completions/mean_terminated_length": 509.0, "completions/min_length": 509.0, "completions/min_terminated_length": 509.0, "entropy": 0.5129384151659906, "epoch": 0.22690992018244013, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0016744559397920966, "kl": 0.0009139564281213097, "learning_rate": 4.764013479403239e-06, "loss": 0.0, "num_tokens": 64024426.0, "reward": 0.03564453125, "reward_std": 0.28625383973121643, "rewards/code_complexity_reward/mean": 0.014160155318677425, "rewards/code_complexity_reward/std": 0.1125217005610466, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 199, "step_time": 45.9311373885721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5070018768310547, "epoch": 0.22805017103762829, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0011926833540201187, "kl": 0.00093802809715271, "learning_rate": 4.759773691909708e-06, "loss": 0.0, "num_tokens": 64344986.0, "reward": 0.01914062537252903, "reward_std": 0.21591484546661377, "rewards/code_complexity_reward/mean": 0.0074218749068677425, "rewards/code_complexity_reward/std": 0.0837220847606659, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 200, "step_time": 45.22977168485522 }, { "epoch": 0.22805017103762829, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.995, "eval_completions/max_length": 512.0, "eval_completions/max_terminated_length": 9.92, "eval_completions/mean_length": 511.615, "eval_completions/mean_terminated_length": 8.7, "eval_completions/min_length": 509.24, "eval_completions/min_terminated_length": 7.48, "eval_entropy": 0.5231816571950912, "eval_frac_reward_zero_std": 0.96, "eval_kl": 0.000946990258526057, "eval_loss": 0.0004066831024829298, "eval_num_tokens": 64344986.0, "eval_reward": 0.03700000047683716, "eval_reward_std": 0.07187317371368408, "eval_rewards/code_complexity_reward/mean": 0.015749999880790712, "eval_rewards/code_complexity_reward/std": 0.030507435202598573, "eval_rewards/code_execution_reward/mean": 0.0125, "eval_rewards/code_execution_reward/std": 0.024493119716644286, "eval_rewards/code_syntax_reward/mean": 0.00875, "eval_rewards/code_syntax_reward/std": 0.01687566041946411, "eval_rewards/reasoning_present_reward_func/mean": 0.0, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.0, "eval_rewards/xmlcount_reward_func/std": 0.0, "eval_runtime": 1246.347, "eval_samples_per_second": 0.08, "eval_steps_per_second": 0.01, "step": 200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5138206481933594, "epoch": 0.2291904218928164, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0014893420739099383, "kl": 0.0009483247995376587, "learning_rate": 4.755498077443419e-06, "loss": 0.0, "num_tokens": 64667238.0, "reward": 0.01972656324505806, "reward_std": 0.20459860563278198, "rewards/code_complexity_reward/mean": 0.008984374813735485, "rewards/code_complexity_reward/std": 0.09064534306526184, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 201, "step_time": 45.421472986228764 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 512.0, "completions/min_length": 512.0, "completions/min_terminated_length": 512.0, "entropy": 0.50909423828125, "epoch": 0.23033067274800456, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.001987625379115343, "kl": 0.0009393244981765747, "learning_rate": 4.7511867037909484e-06, "loss": 0.0, "num_tokens": 64989046.0, "reward": 0.04580078274011612, "reward_std": 0.3275015950202942, "rewards/code_complexity_reward/mean": 0.01845703087747097, "rewards/code_complexity_reward/std": 0.13091638684272766, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 202, "step_time": 50.926354225724936 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 511.828125, "completions/mean_terminated_length": 468.0, "completions/min_length": 427.0, "completions/min_terminated_length": 427.0, "entropy": 0.508735571987927, "epoch": 0.2314709236031927, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0018621303606778383, "kl": 0.0009325241417172947, "learning_rate": 4.746839639305808e-06, "loss": 0.0005, "num_tokens": 65311162.0, "reward": 0.02714843675494194, "reward_std": 0.23881080746650696, "rewards/code_complexity_reward/mean": 0.01249999925494194, "rewards/code_complexity_reward/std": 0.10644587874412537, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 203, "step_time": 50.80099537782371 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 511.80078125, "completions/mean_terminated_length": 410.0, "completions/min_length": 410.0, "completions/min_terminated_length": 410.0, "entropy": 0.4954290995374322, "epoch": 0.23261117445838084, "frac_reward_zero_std": 0.921875, "grad_norm": 0.0026402813382446766, "kl": 0.0009278782254114049, "learning_rate": 4.742456952907358e-06, "loss": 0.0006, "num_tokens": 65632540.0, "reward": 0.04960937798023224, "reward_std": 0.3378298580646515, "rewards/code_complexity_reward/mean": 0.01933593675494194, "rewards/code_complexity_reward/std": 0.1307705193758011, "rewards/code_execution_reward/mean": 0.01953125, "rewards/code_execution_reward/std": 0.1385180652141571, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 204, "step_time": 51.90917009674013 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 418.0, "completions/mean_length": 511.81640625, "completions/mean_terminated_length": 418.0, "completions/min_length": 418.0, "completions/min_terminated_length": 418.0, "entropy": 0.5151623813435435, "epoch": 0.233751425313569, "frac_reward_zero_std": 0.9375, "grad_norm": 0.002094435738399625, "kl": 0.000952089374550269, "learning_rate": 4.73803871407972e-06, "loss": 0.0005, "num_tokens": 65952650.0, "reward": 0.04375000298023224, "reward_std": 0.3158097565174103, "rewards/code_complexity_reward/mean": 0.01835937425494194, "rewards/code_complexity_reward/std": 0.13027459383010864, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 205, "step_time": 50.96155072376132 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 511.810546875, "completions/mean_terminated_length": 479.66668701171875, "completions/min_length": 471.0, "completions/min_terminated_length": 471.0, "entropy": 0.5183755145408213, "epoch": 0.2348916761687571, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0018867775797843933, "kl": 0.0009442642040085047, "learning_rate": 4.733584992870669e-06, "loss": 0.0005, "num_tokens": 66273133.0, "reward": 0.02626953274011612, "reward_std": 0.2462206482887268, "rewards/code_complexity_reward/mean": 0.01064453087747097, "rewards/code_complexity_reward/std": 0.0981680229306221, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 206, "step_time": 51.5479404758662 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 511.8828125, "completions/mean_terminated_length": 452.0, "completions/min_length": 452.0, "completions/min_terminated_length": 452.0, "entropy": 0.5047558471560478, "epoch": 0.23603192702394526, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001547793042846024, "kl": 0.0009503525443506078, "learning_rate": 4.729095859890529e-06, "loss": 0.0001, "num_tokens": 66593329.0, "reward": 0.03134765848517418, "reward_std": 0.26955559849739075, "rewards/code_complexity_reward/mean": 0.012792968191206455, "rewards/code_complexity_reward/std": 0.10879796743392944, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 207, "step_time": 51.32711111754179 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5023097991943359, "epoch": 0.23717217787913342, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0016285256715491414, "kl": 0.0009529590606689453, "learning_rate": 4.724571386311046e-06, "loss": 0.0, "num_tokens": 66913665.0, "reward": 0.02363281324505806, "reward_std": 0.23822173476219177, "rewards/code_complexity_reward/mean": 0.008984374813735485, "rewards/code_complexity_reward/std": 0.09059135615825653, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 208, "step_time": 45.61909634433687 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.509765625, "epoch": 0.23831242873432154, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0016506114043295383, "kl": 0.0009779483079910278, "learning_rate": 4.720011643864268e-06, "loss": 0.0, "num_tokens": 67234449.0, "reward": 0.02861328050494194, "reward_std": 0.2630296051502228, "rewards/code_complexity_reward/mean": 0.011035156436264515, "rewards/code_complexity_reward/std": 0.10145855695009232, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 209, "step_time": 46.53948832768947 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 511.869140625, "completions/mean_terminated_length": 445.0, "completions/min_length": 445.0, "completions/min_terminated_length": 445.0, "entropy": 0.5204551396891475, "epoch": 0.2394526795895097, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0018343842821195722, "kl": 0.0009691867599030957, "learning_rate": 4.715416704841404e-06, "loss": 0.0004, "num_tokens": 67556646.0, "reward": 0.0263671875, "reward_std": 0.2471720427274704, "rewards/code_complexity_reward/mean": 0.0107421875, "rewards/code_complexity_reward/std": 0.09907515347003937, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 210, "step_time": 45.82368289399892 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 511.974609375, "completions/mean_terminated_length": 499.0, "completions/min_length": 499.0, "completions/min_terminated_length": 499.0, "entropy": 0.5068041738122702, "epoch": 0.24059293044469784, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0020554300863295794, "kl": 0.0009718334877106827, "learning_rate": 4.710786642091673e-06, "loss": 0.0001, "num_tokens": 67877197.0, "reward": 0.03095703013241291, "reward_std": 0.2660938501358032, "rewards/code_complexity_reward/mean": 0.012402343563735485, "rewards/code_complexity_reward/std": 0.10555736720561981, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 211, "step_time": 45.25454221572727 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 511.982421875, "completions/mean_terminated_length": 503.0, "completions/min_length": 503.0, "completions/min_terminated_length": 503.0, "entropy": 0.5040599526837468, "epoch": 0.24173318129988597, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0017140806885436177, "kl": 0.000976059889580938, "learning_rate": 4.706121529021158e-06, "loss": 0.0, "num_tokens": 68200052.0, "reward": 0.033203125, "reward_std": 0.28230786323547363, "rewards/code_complexity_reward/mean": 0.0126953125, "rewards/code_complexity_reward/std": 0.10797441005706787, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 212, "step_time": 44.9746730895713 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5012989044189453, "epoch": 0.24287343215507412, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0010017919121310115, "kl": 0.0009545981884002686, "learning_rate": 4.7014214395916355e-06, "loss": 0.0, "num_tokens": 68522120.0, "reward": 0.01435546949505806, "reward_std": 0.18717168271541595, "rewards/code_complexity_reward/mean": 0.005566406063735485, "rewards/code_complexity_reward/std": 0.07257677614688873, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 213, "step_time": 45.64111914113164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5165824890136719, "epoch": 0.24401368301026224, "frac_reward_zero_std": 0.921875, "grad_norm": 0.0025937785394489765, "kl": 0.0009837746620178223, "learning_rate": 4.696686448319408e-06, "loss": 0.0, "num_tokens": 68844848.0, "reward": 0.04599609225988388, "reward_std": 0.3191208839416504, "rewards/code_complexity_reward/mean": 0.01962890475988388, "rewards/code_complexity_reward/std": 0.13289807736873627, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 214, "step_time": 51.11643483582884 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5027446746826172, "epoch": 0.2451539338654504, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.002087595406919718, "kl": 0.0009736865758895874, "learning_rate": 4.691916630274117e-06, "loss": 0.0, "num_tokens": 69164200.0, "reward": 0.03554687649011612, "reward_std": 0.2854873538017273, "rewards/code_complexity_reward/mean": 0.014062500558793545, "rewards/code_complexity_reward/std": 0.11185808479785919, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 215, "step_time": 45.50819264817983 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 511.77734375, "completions/mean_terminated_length": 455.0, "completions/min_length": 409.0, "completions/min_terminated_length": 409.0, "entropy": 0.519929931499064, "epoch": 0.24629418472063855, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0018129657255485654, "kl": 0.0010008495328293066, "learning_rate": 4.687112061077556e-06, "loss": 0.0001, "num_tokens": 69484946.0, "reward": 0.04248046875, "reward_std": 0.3179359436035156, "rewards/code_complexity_reward/mean": 0.01611328125, "rewards/code_complexity_reward/std": 0.12070053815841675, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 216, "step_time": 48.081068970263004 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5125961303710938, "epoch": 0.24743443557582667, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.001770813250914216, "kl": 0.0009892135858535767, "learning_rate": 4.6822728169024735e-06, "loss": 0.0, "num_tokens": 69809290.0, "reward": 0.02402343787252903, "reward_std": 0.22727221250534058, "rewards/code_complexity_reward/mean": 0.01035156287252903, "rewards/code_complexity_reward/std": 0.09524031728506088, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 217, "step_time": 50.592664040625095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.49960899353027344, "epoch": 0.24857468643101482, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0020840929355472326, "kl": 0.0009915977716445923, "learning_rate": 4.67739897447136e-06, "loss": 0.0, "num_tokens": 70130790.0, "reward": 0.02626953274011612, "reward_std": 0.24452587962150574, "rewards/code_complexity_reward/mean": 0.01064453087747097, "rewards/code_complexity_reward/std": 0.09791851788759232, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 218, "step_time": 51.610070396214724 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 511.947265625, "completions/mean_terminated_length": 485.0, "completions/min_length": 485.0, "completions/min_terminated_length": 485.0, "entropy": 0.5009378651157022, "epoch": 0.24971493728620298, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0022250209003686905, "kl": 0.0009917967709043296, "learning_rate": 4.672490611055238e-06, "loss": 0.0002, "num_tokens": 70454039.0, "reward": 0.03623047098517418, "reward_std": 0.29045623540878296, "rewards/code_complexity_reward/mean": 0.014746093191206455, "rewards/code_complexity_reward/std": 0.11721797287464142, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 219, "step_time": 45.311363206245005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5104827880859375, "epoch": 0.2508551881413911, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0019728217739611864, "kl": 0.0010358840227127075, "learning_rate": 4.667547804472431e-06, "loss": 0.0, "num_tokens": 70775671.0, "reward": 0.03115234337747097, "reward_std": 0.2681954503059387, "rewards/code_complexity_reward/mean": 0.01259765587747097, "rewards/code_complexity_reward/std": 0.1071900874376297, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 220, "step_time": 50.910957218147814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5043182373046875, "epoch": 0.2519954389965792, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0018036097753793001, "kl": 0.0010161995887756348, "learning_rate": 4.662570633087333e-06, "loss": 0.0, "num_tokens": 71095667.0, "reward": 0.02177734486758709, "reward_std": 0.2228822261095047, "rewards/code_complexity_reward/mean": 0.00908203050494194, "rewards/code_complexity_reward/std": 0.09157534688711166, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 221, "step_time": 50.92446893453598 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5100135803222656, "epoch": 0.2531356898517674, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0018385122530162334, "kl": 0.0010194778442382812, "learning_rate": 4.657559175809168e-06, "loss": 0.0, "num_tokens": 71417511.0, "reward": 0.02119140699505806, "reward_std": 0.21654905378818512, "rewards/code_complexity_reward/mean": 0.008496093563735485, "rewards/code_complexity_reward/std": 0.08572862297296524, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 222, "step_time": 46.672639379277825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5099353790283203, "epoch": 0.2542759407069555, "frac_reward_zero_std": 0.984375, "grad_norm": 0.001131700468249619, "kl": 0.0010207444429397583, "learning_rate": 4.6525135120907314e-06, "loss": 0.0, "num_tokens": 71739771.0, "reward": 0.00947265699505806, "reward_std": 0.15142220258712769, "rewards/code_complexity_reward/mean": 0.003613281063735485, "rewards/code_complexity_reward/std": 0.057777076959609985, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 223, "step_time": 45.729798916727304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 511.59375, "completions/mean_terminated_length": 408.0, "completions/min_length": 371.0, "completions/min_terminated_length": 371.0, "entropy": 0.5066583063453436, "epoch": 0.2554161915621437, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0017637767596170306, "kl": 0.0010582923368929187, "learning_rate": 4.647433721927139e-06, "loss": 0.0012, "num_tokens": 72060495.0, "reward": 0.02871093899011612, "reward_std": 0.26391950249671936, "rewards/code_complexity_reward/mean": 0.01113281212747097, "rewards/code_complexity_reward/std": 0.10233614593744278, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 224, "step_time": 45.21176161430776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 458.0, "completions/mean_length": 511.765625, "completions/mean_terminated_length": 452.0, "completions/min_length": 446.0, "completions/min_terminated_length": 446.0, "entropy": 0.5109332511201501, "epoch": 0.25655644241733183, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0023225166369229555, "kl": 0.001021475520246895, "learning_rate": 4.642319885854557e-06, "loss": 0.0005, "num_tokens": 72379839.0, "reward": 0.04335937649011612, "reward_std": 0.302330881357193, "rewards/code_complexity_reward/mean": 0.01894531212747097, "rewards/code_complexity_reward/std": 0.12848834693431854, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 225, "step_time": 50.64999053813517 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.8125, "completions/mean_terminated_length": 480.0, "completions/min_length": 459.0, "completions/min_terminated_length": 459.0, "entropy": 0.4984328728169203, "epoch": 0.2576966932725199, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.0024606238584965467, "kl": 0.0010339942455175333, "learning_rate": 4.637172084948917e-06, "loss": 0.0002, "num_tokens": 72699971.0, "reward": 0.08818359673023224, "reward_std": 0.45173293352127075, "rewards/code_complexity_reward/mean": 0.03447265550494194, "rewards/code_complexity_reward/std": 0.17611293494701385, "rewards/code_execution_reward/mean": 0.03515625, "rewards/code_execution_reward/std": 0.1843547374010086, "rewards/code_syntax_reward/mean": 0.0185546875, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 226, "step_time": 45.72880727797747 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5059642791748047, "epoch": 0.2588369441277081, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0013069245032966137, "kl": 0.0010727643966674805, "learning_rate": 4.631990400824643e-06, "loss": 0.0, "num_tokens": 73021675.0, "reward": 0.005664062686264515, "reward_std": 0.09053628146648407, "rewards/code_complexity_reward/mean": 0.0037109374534338713, "rewards/code_complexity_reward/std": 0.05931687355041504, "rewards/code_execution_reward/mean": 0.0, "rewards/code_execution_reward/std": 0.0, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 227, "step_time": 52.400277933105826 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 511.91796875, "completions/mean_terminated_length": 470.0, "completions/min_length": 470.0, "completions/min_terminated_length": 470.0, "entropy": 0.5071133635938168, "epoch": 0.25997719498289623, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002552899532020092, "kl": 0.0010599596062093042, "learning_rate": 4.626774915633349e-06, "loss": 0.0002, "num_tokens": 73343481.0, "reward": 0.05449219048023224, "reward_std": 0.3549886643886566, "rewards/code_complexity_reward/mean": 0.02128906175494194, "rewards/code_complexity_reward/std": 0.13765543699264526, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 228, "step_time": 45.936312831006944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 511.7734375, "completions/mean_terminated_length": 473.3333435058594, "completions/min_length": 446.0, "completions/min_terminated_length": 446.0, "entropy": 0.5043902033939958, "epoch": 0.2611174458380844, "frac_reward_zero_std": 0.875, "grad_norm": 0.0036042523570358753, "kl": 0.0010475053859408945, "learning_rate": 4.621525712062537e-06, "loss": 0.0004, "num_tokens": 73665825.0, "reward": 0.07988281548023224, "reward_std": 0.41705986857414246, "rewards/code_complexity_reward/mean": 0.03398437425494194, "rewards/code_complexity_reward/std": 0.17364893853664398, "rewards/code_execution_reward/mean": 0.02734375, "rewards/code_execution_reward/std": 0.16324250400066376, "rewards/code_syntax_reward/mean": 0.0185546875, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 229, "step_time": 45.34674447402358 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 511.9375, "completions/mean_terminated_length": 480.0, "completions/min_length": 480.0, "completions/min_terminated_length": 480.0, "entropy": 0.4997584028169513, "epoch": 0.26225769669327254, "frac_reward_zero_std": 0.9375, "grad_norm": 0.002178784692659974, "kl": 0.0010688781731005292, "learning_rate": 4.616242873334292e-06, "loss": 0.0002, "num_tokens": 73988053.0, "reward": 0.05253906548023224, "reward_std": 0.35492533445358276, "rewards/code_complexity_reward/mean": 0.02031249925494194, "rewards/code_complexity_reward/std": 0.13723400235176086, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 230, "step_time": 45.755312259308994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 511.986328125, "completions/mean_terminated_length": 505.0, "completions/min_length": 505.0, "completions/min_terminated_length": 505.0, "entropy": 0.49743030313402414, "epoch": 0.2633979475484607, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.001898508402518928, "kl": 0.0011053014513890957, "learning_rate": 4.610926483203954e-06, "loss": 0.0, "num_tokens": 74308026.0, "reward": 0.02451171912252903, "reward_std": 0.23147565126419067, "rewards/code_complexity_reward/mean": 0.010839843191206455, "rewards/code_complexity_reward/std": 0.09967990964651108, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 231, "step_time": 46.686134818941355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5115890502929688, "epoch": 0.2645381984036488, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0014619894791394472, "kl": 0.0011195987462997437, "learning_rate": 4.6055766259588004e-06, "loss": 0.0, "num_tokens": 74628566.0, "reward": 0.01689453050494194, "reward_std": 0.19455082714557648, "rewards/code_complexity_reward/mean": 0.007128906436264515, "rewards/code_complexity_reward/std": 0.08050087094306946, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 232, "step_time": 45.40554733015597 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.50457763671875, "epoch": 0.26567844925883694, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001649815938435495, "kl": 0.0010874569416046143, "learning_rate": 4.600193386416697e-06, "loss": 0.0, "num_tokens": 74952370.0, "reward": 0.02773437276482582, "reward_std": 0.2549746632575989, "rewards/code_complexity_reward/mean": 0.01015624962747097, "rewards/code_complexity_reward/std": 0.09344659000635147, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 233, "step_time": 45.500659140758216 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 511.822265625, "completions/mean_terminated_length": 489.25, "completions/min_length": 467.0, "completions/min_terminated_length": 467.0, "entropy": 0.5110119446180761, "epoch": 0.2668187001140251, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0012610777048394084, "kl": 0.0010868756371564814, "learning_rate": 4.594776849924766e-06, "loss": 0.0001, "num_tokens": 75276479.0, "reward": 0.03642578423023224, "reward_std": 0.2787768840789795, "rewards/code_complexity_reward/mean": 0.01591796800494194, "rewards/code_complexity_reward/std": 0.11921767890453339, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 234, "step_time": 46.28258977364749 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5074310302734375, "epoch": 0.26795895096921324, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0009267532732337713, "kl": 0.0011448711156845093, "learning_rate": 4.589327102358024e-06, "loss": 0.0, "num_tokens": 75597067.0, "reward": 0.0047851563431322575, "reward_std": 0.10827571898698807, "rewards/code_complexity_reward/mean": 0.0018554687267169356, "rewards/code_complexity_reward/std": 0.04198446497321129, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 235, "step_time": 45.17337728384882 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 511.640625, "completions/mean_terminated_length": 420.0, "completions/min_length": 368.0, "completions/min_terminated_length": 368.0, "entropy": 0.49991280445829034, "epoch": 0.2690992018244014, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0023945015855133533, "kl": 0.0011152111837873235, "learning_rate": 4.5838442301180245e-06, "loss": 0.0007, "num_tokens": 75918579.0, "reward": 0.05976562574505806, "reward_std": 0.3729901909828186, "rewards/code_complexity_reward/mean": 0.02363281138241291, "rewards/code_complexity_reward/std": 0.14661237597465515, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.0126953125, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 236, "step_time": 45.335906364023685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5190162658691406, "epoch": 0.2702394526795895, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0016166189452633262, "kl": 0.0011633187532424927, "learning_rate": 4.5783283201314876e-06, "loss": 0.0, "num_tokens": 76239203.0, "reward": 0.02119140699505806, "reward_std": 0.21742835640907288, "rewards/code_complexity_reward/mean": 0.008496093563735485, "rewards/code_complexity_reward/std": 0.08567153662443161, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 237, "step_time": 44.872856609523296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 511.796875, "completions/mean_terminated_length": 408.0, "completions/min_length": 408.0, "completions/min_terminated_length": 408.0, "entropy": 0.5212910594418645, "epoch": 0.27137970353477764, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0021252138540148735, "kl": 0.0011293303723505232, "learning_rate": 4.572779459848922e-06, "loss": 0.0006, "num_tokens": 76561095.0, "reward": 0.02851562574505806, "reward_std": 0.2621366083621979, "rewards/code_complexity_reward/mean": 0.01093749888241291, "rewards/code_complexity_reward/std": 0.10057320445775986, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 238, "step_time": 51.426346712745726 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 511.591796875, "completions/mean_terminated_length": 459.75, "completions/min_length": 426.0, "completions/min_terminated_length": 426.0, "entropy": 0.5086621036753058, "epoch": 0.2725199543899658, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0017038054065778852, "kl": 0.0011211645096409484, "learning_rate": 4.5671977372432355e-06, "loss": 0.0008, "num_tokens": 76881990.0, "reward": 0.02939453162252903, "reward_std": 0.2552390694618225, "rewards/code_complexity_reward/mean": 0.012792968191206455, "rewards/code_complexity_reward/std": 0.10879797488451004, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 239, "step_time": 52.105892181396484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 511.6953125, "completions/mean_terminated_length": 434.0, "completions/min_length": 428.0, "completions/min_terminated_length": 428.0, "entropy": 0.5048919483087957, "epoch": 0.27366020524515394, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0022397206630557775, "kl": 0.00113972579856636, "learning_rate": 4.561583240808344e-06, "loss": 0.0006, "num_tokens": 77201938.0, "reward": 0.04365234076976776, "reward_std": 0.3144338130950928, "rewards/code_complexity_reward/mean": 0.01826171763241291, "rewards/code_complexity_reward/std": 0.12955403327941895, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 240, "step_time": 45.52091764006764 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.49300384521484375, "epoch": 0.2748004561003421, "frac_reward_zero_std": 0.984375, "grad_norm": 0.0010371897369623184, "kl": 0.001106351613998413, "learning_rate": 4.555936059557768e-06, "loss": 0.0, "num_tokens": 77524362.0, "reward": 0.00927734375, "reward_std": 0.14836639165878296, "rewards/code_complexity_reward/mean": 0.00341796875, "rewards/code_complexity_reward/std": 0.05483507737517357, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.001953125, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 241, "step_time": 45.685663039796054 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5145988464355469, "epoch": 0.2759407069555302, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.0022744282614439726, "kl": 0.0011572390794754028, "learning_rate": 4.5502562830232225e-06, "loss": 0.0, "num_tokens": 77847934.0, "reward": 0.02910156361758709, "reward_std": 0.25407207012176514, "rewards/code_complexity_reward/mean": 0.012500000186264515, "rewards/code_complexity_reward/std": 0.10644587129354477, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 242, "step_time": 52.57669567596167 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 511.734375, "completions/mean_terminated_length": 376.0, "completions/min_length": 376.0, "completions/min_terminated_length": 376.0, "entropy": 0.5131364781409502, "epoch": 0.27708095781071834, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.0027040252462029457, "kl": 0.0012090681702829897, "learning_rate": 4.544544001253189e-06, "loss": 0.0008, "num_tokens": 78171058.0, "reward": 0.04843750223517418, "reward_std": 0.3319242000579834, "rewards/code_complexity_reward/mean": 0.02011718600988388, "rewards/code_complexity_reward/std": 0.1359376609325409, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 243, "step_time": 46.189454895444214 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 511.87890625, "completions/mean_terminated_length": 450.0, "completions/min_length": 450.0, "completions/min_terminated_length": 450.0, "entropy": 0.4983954280614853, "epoch": 0.2782212086659065, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.001799220684915781, "kl": 0.0011399673749110661, "learning_rate": 4.538799304811503e-06, "loss": 0.0004, "num_tokens": 78492480.0, "reward": 0.02656250074505806, "reward_std": 0.24817830324172974, "rewards/code_complexity_reward/mean": 0.010937499813735485, "rewards/code_complexity_reward/std": 0.10062183439731598, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 244, "step_time": 46.29562845826149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.681640625, "completions/mean_terminated_length": 479.3999938964844, "completions/min_length": 441.0, "completions/min_terminated_length": 441.0, "entropy": 0.5159510541707277, "epoch": 0.27936145952109465, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.00314102484844625, "kl": 0.0011512272903928533, "learning_rate": 4.533022284775903e-06, "loss": 0.0005, "num_tokens": 78814649.0, "reward": 0.07958984375, "reward_std": 0.4236275553703308, "rewards/code_complexity_reward/mean": 0.03271484375, "rewards/code_complexity_reward/std": 0.17161330580711365, "rewards/code_execution_reward/mean": 0.029296875, "rewards/code_execution_reward/std": 0.16880230605602264, "rewards/code_syntax_reward/mean": 0.017578125, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 245, "step_time": 45.782447619363666 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 511.875, "completions/mean_terminated_length": 448.0, "completions/min_length": 448.0, "completions/min_terminated_length": 448.0, "entropy": 0.508544921875, "epoch": 0.2805017103762828, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0018845826853066683, "kl": 0.0011708651272783754, "learning_rate": 4.527213032736596e-06, "loss": 0.0004, "num_tokens": 79135809.0, "reward": 0.02812500111758709, "reward_std": 0.2585901916027069, "rewards/code_complexity_reward/mean": 0.01054687425494194, "rewards/code_complexity_reward/std": 0.09710130095481873, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 246, "step_time": 46.20157009828836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5096664428710938, "epoch": 0.28164196123147095, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0013288543559610844, "kl": 0.0011495649814605713, "learning_rate": 4.521371640794802e-06, "loss": 0.0, "num_tokens": 79458121.0, "reward": 0.01220703125, "reward_std": 0.16281649470329285, "rewards/code_complexity_reward/mean": 0.00537109375, "rewards/code_complexity_reward/std": 0.07005351036787033, "rewards/code_execution_reward/mean": 0.00390625, "rewards/code_execution_reward/std": 0.06243881583213806, "rewards/code_syntax_reward/mean": 0.0029296875, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 247, "step_time": 45.492227241396904 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.84765625, "completions/mean_terminated_length": 492.5, "completions/min_length": 468.0, "completions/min_terminated_length": 468.0, "entropy": 0.49752742564305663, "epoch": 0.28278221208665905, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.002138364827260375, "kl": 0.0011653899437078508, "learning_rate": 4.5154982015612965e-06, "loss": 0.0002, "num_tokens": 79778771.0, "reward": 0.06464844197034836, "reward_std": 0.38827013969421387, "rewards/code_complexity_reward/mean": 0.02558593638241291, "rewards/code_complexity_reward/std": 0.15285812318325043, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.013671875, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 248, "step_time": 45.25749132037163 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5105628967285156, "epoch": 0.2839224629418472, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0020731010008603334, "kl": 0.0011938661336898804, "learning_rate": 4.509592808154936e-06, "loss": 0.0, "num_tokens": 80098347.0, "reward": 0.0283203125, "reward_std": 0.26036012172698975, "rewards/code_complexity_reward/mean": 0.0107421875, "rewards/code_complexity_reward/std": 0.09882795065641403, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 249, "step_time": 46.39009870309383 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5000839233398438, "epoch": 0.28506271379703535, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.001880340394563973, "kl": 0.0012118667364120483, "learning_rate": 4.50365555420119e-06, "loss": 0.0, "num_tokens": 80419883.0, "reward": 0.02460937388241291, "reward_std": 0.23206688463687897, "rewards/code_complexity_reward/mean": 0.010937499813735485, "rewards/code_complexity_reward/std": 0.10057320445775986, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 250, "step_time": 45.20779289677739 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 511.947265625, "completions/mean_terminated_length": 503.0, "completions/min_length": 498.0, "completions/min_terminated_length": 498.0, "entropy": 0.4944837214425206, "epoch": 0.2862029646522235, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.0035127343144267797, "kl": 0.0011946365357289324, "learning_rate": 4.497686533830648e-06, "loss": 0.0001, "num_tokens": 80741164.0, "reward": 0.06816406548023224, "reward_std": 0.387465238571167, "rewards/code_complexity_reward/mean": 0.02910156175494194, "rewards/code_complexity_reward/std": 0.16230596601963043, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 251, "step_time": 50.47389565221965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5053482055664062, "epoch": 0.28734321550741165, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0013954911846667528, "kl": 0.001198500394821167, "learning_rate": 4.491685841677538e-06, "loss": 0.0, "num_tokens": 81062640.0, "reward": 0.01718750037252903, "reward_std": 0.19763152301311493, "rewards/code_complexity_reward/mean": 0.0074218749068677425, "rewards/code_complexity_reward/std": 0.0837220847606659, "rewards/code_execution_reward/mean": 0.005859375, "rewards/code_execution_reward/std": 0.07639661431312561, "rewards/code_syntax_reward/mean": 0.00390625, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 252, "step_time": 47.099687782116234 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 511.646484375, "completions/mean_terminated_length": 451.66668701171875, "completions/min_length": 420.0, "completions/min_terminated_length": 420.0, "entropy": 0.5074766436591744, "epoch": 0.28848346636259975, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002488102065399289, "kl": 0.0011871223905473016, "learning_rate": 4.485653572878213e-06, "loss": 0.001, "num_tokens": 81385167.0, "reward": 0.04111327975988388, "reward_std": 0.29807624220848083, "rewards/code_complexity_reward/mean": 0.01767577975988388, "rewards/code_complexity_reward/std": 0.1255713701248169, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 253, "step_time": 45.635451570153236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 511.931640625, "completions/mean_terminated_length": 477.0, "completions/min_length": 477.0, "completions/min_terminated_length": 477.0, "entropy": 0.5003771712072194, "epoch": 0.2896237172177879, "frac_reward_zero_std": 0.953125, "grad_norm": 0.002023057546466589, "kl": 0.0012023685121675953, "learning_rate": 4.4795898230696535e-06, "loss": 0.0001, "num_tokens": 81706768.0, "reward": 0.03583984449505806, "reward_std": 0.2879165709018707, "rewards/code_complexity_reward/mean": 0.014355468563735485, "rewards/code_complexity_reward/std": 0.11418037116527557, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 254, "step_time": 45.35125857684761 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 511.900390625, "completions/mean_terminated_length": 461.0, "completions/min_length": 461.0, "completions/min_terminated_length": 461.0, "entropy": 0.5056877378374338, "epoch": 0.29076396807297605, "frac_reward_zero_std": 0.9765625, "grad_norm": 0.0014859229559078813, "kl": 0.0012253881668584654, "learning_rate": 4.473494688387945e-06, "loss": 0.0001, "num_tokens": 82029153.0, "reward": 0.0234375, "reward_std": 0.23626144230365753, "rewards/code_complexity_reward/mean": 0.0087890625, "rewards/code_complexity_reward/std": 0.08864548057317734, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 255, "step_time": 45.02948232553899 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 511.97265625, "completions/mean_terminated_length": 498.0, "completions/min_length": 498.0, "completions/min_terminated_length": 498.0, "entropy": 0.4924317169934511, "epoch": 0.2919042189281642, "frac_reward_zero_std": 0.890625, "grad_norm": 0.0033633664716035128, "kl": 0.001215645646880148, "learning_rate": 4.467368265466759e-06, "loss": 0.0001, "num_tokens": 82350267.0, "reward": 0.07880860567092896, "reward_std": 0.4194105863571167, "rewards/code_complexity_reward/mean": 0.03193359076976776, "rewards/code_complexity_reward/std": 0.16784219443798065, "rewards/code_execution_reward/mean": 0.029296875, "rewards/code_execution_reward/std": 0.16880230605602264, "rewards/code_syntax_reward/mean": 0.017578125, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 256, "step_time": 45.76135726086795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5217132568359375, "epoch": 0.29304446978335236, "frac_reward_zero_std": 0.9375, "grad_norm": 0.002907632617279887, "kl": 0.0012632906436920166, "learning_rate": 4.461210651435814e-06, "loss": 0.0, "num_tokens": 82670823.0, "reward": 0.04550781100988388, "reward_std": 0.3142557740211487, "rewards/code_complexity_reward/mean": 0.01914062537252903, "rewards/code_complexity_reward/std": 0.12963460385799408, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 257, "step_time": 46.0861750934273 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5053520202636719, "epoch": 0.29418472063854045, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0020861979573965073, "kl": 0.0012796074151992798, "learning_rate": 4.4550219439193435e-06, "loss": 0.0, "num_tokens": 82992039.0, "reward": 0.02451171912252903, "reward_std": 0.231919065117836, "rewards/code_complexity_reward/mean": 0.010839843191206455, "rewards/code_complexity_reward/std": 0.09972897917032242, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 258, "step_time": 46.289316647686064 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.904296875, "completions/mean_terminated_length": 487.5, "completions/min_length": 464.0, "completions/min_terminated_length": 464.0, "entropy": 0.5112169263884425, "epoch": 0.2953249714937286, "frac_reward_zero_std": 0.9375, "grad_norm": 0.002525175688788295, "kl": 0.0012940725064254366, "learning_rate": 4.448802241034541e-06, "loss": 0.0001, "num_tokens": 83311106.0, "reward": 0.05234374850988388, "reward_std": 0.3433522582054138, "rewards/code_complexity_reward/mean": 0.02109374850988388, "rewards/code_complexity_reward/std": 0.13647209107875824, "rewards/code_execution_reward/mean": 0.01953125, "rewards/code_execution_reward/std": 0.1385180652141571, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 259, "step_time": 46.4045576620847 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 511.923828125, "completions/mean_terminated_length": 473.0, "completions/min_length": 473.0, "completions/min_terminated_length": 473.0, "entropy": 0.5095434533432126, "epoch": 0.29646522234891676, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0020046094432473183, "kl": 0.0012818490158679197, "learning_rate": 4.4425516413900085e-06, "loss": 0.0001, "num_tokens": 83632319.0, "reward": 0.04052734375, "reward_std": 0.3059634268283844, "rewards/code_complexity_reward/mean": 0.01611328125, "rewards/code_complexity_reward/std": 0.12070053815841675, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 260, "step_time": 45.54678396321833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 511.603515625, "completions/mean_terminated_length": 410.5, "completions/min_length": 381.0, "completions/min_terminated_length": 381.0, "entropy": 0.504764215555042, "epoch": 0.2976054732041049, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.001982493093237281, "kl": 0.0012952966317243408, "learning_rate": 4.4362702440841945e-06, "loss": 0.0009, "num_tokens": 83953108.0, "reward": 0.02646484225988388, "reward_std": 0.24723085761070251, "rewards/code_complexity_reward/mean": 0.01083984412252903, "rewards/code_complexity_reward/std": 0.09972897917032242, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 261, "step_time": 44.85580294393003 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.765625, "completions/mean_terminated_length": 472.0, "completions/min_length": 395.0, "completions/min_terminated_length": 395.0, "entropy": 0.5139048169367015, "epoch": 0.29874572405929306, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0019157440401613712, "kl": 0.0013708654332731385, "learning_rate": 4.429958148703818e-06, "loss": 0.0004, "num_tokens": 84273884.0, "reward": 0.04541015625, "reward_std": 0.3249468207359314, "rewards/code_complexity_reward/mean": 0.01806640625, "rewards/code_complexity_reward/std": 0.12817692756652832, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 262, "step_time": 45.486169777810574 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 511.763671875, "completions/mean_terminated_length": 451.5, "completions/min_length": 404.0, "completions/min_terminated_length": 404.0, "entropy": 0.5026578763499856, "epoch": 0.2998859749144812, "frac_reward_zero_std": 0.96875, "grad_norm": 0.002015832345932722, "kl": 0.0013498747866833583, "learning_rate": 4.423615455322293e-06, "loss": 0.0004, "num_tokens": 84594767.0, "reward": 0.04277344048023224, "reward_std": 0.320097416639328, "rewards/code_complexity_reward/mean": 0.01640624925494194, "rewards/code_complexity_reward/std": 0.12281107157468796, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 263, "step_time": 51.53102887608111 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 511.958984375, "completions/mean_terminated_length": 491.0, "completions/min_length": 491.0, "completions/min_terminated_length": 491.0, "entropy": 0.5039231325499713, "epoch": 0.3010262257696693, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0020617456175386906, "kl": 0.0013701926654903218, "learning_rate": 4.417242264498143e-06, "loss": 0.0001, "num_tokens": 84916034.0, "reward": 0.02900390699505806, "reward_std": 0.2527608275413513, "rewards/code_complexity_reward/mean": 0.012402343563735485, "rewards/code_complexity_reward/std": 0.10560370236635208, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 264, "step_time": 46.11511636804789 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.71484375, "completions/mean_terminated_length": 475.5, "completions/min_length": 436.0, "completions/min_terminated_length": 436.0, "entropy": 0.5054779616184533, "epoch": 0.30216647662485746, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.0023786674719303846, "kl": 0.0013310671620274661, "learning_rate": 4.410838677273403e-06, "loss": 0.0004, "num_tokens": 85239512.0, "reward": 0.07294922322034836, "reward_std": 0.4024578332901001, "rewards/code_complexity_reward/mean": 0.03095703013241291, "rewards/code_complexity_reward/std": 0.1672959178686142, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.0166015625, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 265, "step_time": 45.296559849753976 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 511.525390625, "completions/mean_terminated_length": 431.0, "completions/min_length": 302.0, "completions/min_terminated_length": 302.0, "entropy": 0.5109123671427369, "epoch": 0.3033067274800456, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.002410379471257329, "kl": 0.0013867762845620746, "learning_rate": 4.404404795172022e-06, "loss": 0.0005, "num_tokens": 85561109.0, "reward": 0.05507812276482582, "reward_std": 0.3583551049232483, "rewards/code_complexity_reward/mean": 0.02187499962747097, "rewards/code_complexity_reward/std": 0.14149051904678345, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 266, "step_time": 45.69484478421509 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 511.53125, "completions/mean_terminated_length": 392.0, "completions/min_length": 376.0, "completions/min_terminated_length": 376.0, "entropy": 0.5064319730736315, "epoch": 0.30444697833523376, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0018716827034950256, "kl": 0.0013812315264658537, "learning_rate": 4.397940720198246e-06, "loss": 0.0014, "num_tokens": 85883405.0, "reward": 0.02568359300494194, "reward_std": 0.23902428150177002, "rewards/code_complexity_reward/mean": 0.01005859300494194, "rewards/code_complexity_reward/std": 0.09311629086732864, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 267, "step_time": 45.601907023228705 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 511.923828125, "completions/mean_terminated_length": 492.5, "completions/min_length": 489.0, "completions/min_terminated_length": 489.0, "entropy": 0.5009506959468126, "epoch": 0.3055872291904219, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0016865055076777935, "kl": 0.0013520868687919574, "learning_rate": 4.39144655483501e-06, "loss": 0.0001, "num_tokens": 86204722.0, "reward": 0.04072265699505806, "reward_std": 0.30745285749435425, "rewards/code_complexity_reward/mean": 0.01630859263241291, "rewards/code_complexity_reward/std": 0.12208496779203415, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 268, "step_time": 45.98050310462713 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 511.84375, "completions/mean_terminated_length": 472.0, "completions/min_length": 465.0, "completions/min_terminated_length": 465.0, "entropy": 0.5009899279102683, "epoch": 0.30672748004561, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.002159412018954754, "kl": 0.0013971204753033817, "learning_rate": 4.38492240204231e-06, "loss": 0.0005, "num_tokens": 86525834.0, "reward": 0.02158203162252903, "reward_std": 0.22215373814105988, "rewards/code_complexity_reward/mean": 0.008886718191206455, "rewards/code_complexity_reward/std": 0.08976011723279953, "rewards/code_execution_reward/mean": 0.0078125, "rewards/code_execution_reward/std": 0.08812850713729858, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 269, "step_time": 45.87081395462155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.96484375, "completions/mean_terminated_length": 503.0, "completions/min_length": 494.0, "completions/min_terminated_length": 494.0, "entropy": 0.5049630058929324, "epoch": 0.30786773090079816, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.002394432667642832, "kl": 0.0014504676291835494, "learning_rate": 4.378368365255564e-06, "loss": 0.0, "num_tokens": 86847380.0, "reward": 0.06630859524011612, "reward_std": 0.395934134721756, "rewards/code_complexity_reward/mean": 0.02529296837747097, "rewards/code_complexity_reward/std": 0.15118549764156342, "rewards/code_execution_reward/mean": 0.02734375, "rewards/code_execution_reward/std": 0.16324250400066376, "rewards/code_syntax_reward/mean": 0.013671875, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 270, "step_time": 46.02229320444167 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 511.583984375, "completions/mean_terminated_length": 441.0, "completions/min_length": 381.0, "completions/min_terminated_length": 381.0, "entropy": 0.50326820416376, "epoch": 0.3090079817559863, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002522097434848547, "kl": 0.0014165435532049742, "learning_rate": 4.371784548383985e-06, "loss": 0.0011, "num_tokens": 87169215.0, "reward": 0.04628906399011612, "reward_std": 0.31969884037971497, "rewards/code_complexity_reward/mean": 0.01992187649011612, "rewards/code_complexity_reward/std": 0.134701207280159, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 271, "step_time": 45.99194166623056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 511.7890625, "completions/mean_terminated_length": 476.0, "completions/min_length": 444.0, "completions/min_terminated_length": 444.0, "entropy": 0.49969203351065516, "epoch": 0.31014823261117447, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0025291936472058296, "kl": 0.0014553655728377635, "learning_rate": 4.36517105580892e-06, "loss": 0.0005, "num_tokens": 87490255.0, "reward": 0.03876953199505806, "reward_std": 0.29507678747177124, "rewards/code_complexity_reward/mean": 0.01630859449505806, "rewards/code_complexity_reward/std": 0.12216509133577347, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 272, "step_time": 45.495002718642354 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 511.806640625, "completions/mean_terminated_length": 487.25, "completions/min_length": 455.0, "completions/min_terminated_length": 455.0, "entropy": 0.5081590469926596, "epoch": 0.3112884834663626, "frac_reward_zero_std": 0.921875, "grad_norm": 0.00261874427087605, "kl": 0.0015024568729131715, "learning_rate": 4.358527992382206e-06, "loss": 0.0005, "num_tokens": 87812144.0, "reward": 0.068359375, "reward_std": 0.3962456285953522, "rewards/code_complexity_reward/mean": 0.0263671875, "rewards/code_complexity_reward/std": 0.15208300948143005, "rewards/code_execution_reward/mean": 0.02734375, "rewards/code_execution_reward/std": 0.16324250400066376, "rewards/code_syntax_reward/mean": 0.0146484375, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 273, "step_time": 45.5809542927891 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5036144256591797, "epoch": 0.3124287343215507, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.002278645522892475, "kl": 0.001446351408958435, "learning_rate": 4.351855463424498e-06, "loss": 0.0, "num_tokens": 88136388.0, "reward": 0.03496094048023224, "reward_std": 0.2811048924922943, "rewards/code_complexity_reward/mean": 0.01347656175494194, "rewards/code_complexity_reward/std": 0.10756158828735352, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 274, "step_time": 51.355013623833656 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.4988880157470703, "epoch": 0.31356898517673887, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0018147722585126758, "kl": 0.0014896094799041748, "learning_rate": 4.345153574723611e-06, "loss": 0.0, "num_tokens": 88456536.0, "reward": 0.03349609300494194, "reward_std": 0.28478384017944336, "rewards/code_complexity_reward/mean": 0.01298828050494194, "rewards/code_complexity_reward/std": 0.1104263886809349, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 275, "step_time": 45.935760356485844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5086803436279297, "epoch": 0.314709236031927, "frac_reward_zero_std": 0.9375, "grad_norm": 0.002419000491499901, "kl": 0.001523330807685852, "learning_rate": 4.338422432532829e-06, "loss": 0.0, "num_tokens": 88779656.0, "reward": 0.064453125, "reward_std": 0.3873439431190491, "rewards/code_complexity_reward/mean": 0.025390625, "rewards/code_complexity_reward/std": 0.15173441171646118, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.013671875, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 276, "step_time": 45.20989274978638 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.826171875, "completions/mean_terminated_length": 482.3333435058594, "completions/min_length": 445.0, "completions/min_terminated_length": 445.0, "entropy": 0.5032666511833668, "epoch": 0.31584948688711517, "frac_reward_zero_std": 0.921875, "grad_norm": 0.0028530973941087723, "kl": 0.0015282756958185928, "learning_rate": 4.331662143569235e-06, "loss": 0.0005, "num_tokens": 89099983.0, "reward": 0.05058594048023224, "reward_std": 0.3124608099460602, "rewards/code_complexity_reward/mean": 0.02519531175494194, "rewards/code_complexity_reward/std": 0.15050457417964935, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.013671875, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 277, "step_time": 45.38389061577618 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5083160400390625, "epoch": 0.3169897377423033, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002765029203146696, "kl": 0.0015260428190231323, "learning_rate": 4.324872815012005e-06, "loss": 0.0, "num_tokens": 89421691.0, "reward": 0.04316406697034836, "reward_std": 0.3120259940624237, "rewards/code_complexity_reward/mean": 0.01777343824505806, "rewards/code_complexity_reward/std": 0.12623760104179382, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 278, "step_time": 45.28124610520899 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5134086608886719, "epoch": 0.3181299885974915, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0007650478510186076, "kl": 0.0014593899250030518, "learning_rate": 4.318054554500719e-06, "loss": 0.0, "num_tokens": 89747015.0, "reward": 0.0047851563431322575, "reward_std": 0.10827571898698807, "rewards/code_complexity_reward/mean": 0.0018554687267169356, "rewards/code_complexity_reward/std": 0.04198446497321129, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 279, "step_time": 47.057168307714164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 511.900390625, "completions/mean_terminated_length": 486.5, "completions/min_length": 463.0, "completions/min_terminated_length": 463.0, "entropy": 0.4914084943011403, "epoch": 0.31927023945267957, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.002496498404070735, "kl": 0.001534009645183687, "learning_rate": 4.3112074701336505e-06, "loss": 0.0003, "num_tokens": 90068228.0, "reward": 0.03632812574505806, "reward_std": 0.2915787994861603, "rewards/code_complexity_reward/mean": 0.014843749813735485, "rewards/code_complexity_reward/std": 0.11793383955955505, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 280, "step_time": 51.6085448898375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 511.8515625, "completions/mean_terminated_length": 474.0, "completions/min_length": 452.0, "completions/min_terminated_length": 452.0, "entropy": 0.5027010096237063, "epoch": 0.3204104903078677, "frac_reward_zero_std": 0.953125, "grad_norm": 0.002106096362695098, "kl": 0.0015296433412004262, "learning_rate": 4.304331670466052e-06, "loss": 0.0003, "num_tokens": 90390460.0, "reward": 0.03105468861758709, "reward_std": 0.2680882215499878, "rewards/code_complexity_reward/mean": 0.01249999925494194, "rewards/code_complexity_reward/std": 0.10644587129354477, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 281, "step_time": 52.11970657296479 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 511.873046875, "completions/mean_terminated_length": 490.3333435058594, "completions/min_length": 473.0, "completions/min_terminated_length": 473.0, "entropy": 0.498179966583848, "epoch": 0.3215507411630559, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0027682313229888678, "kl": 0.0015520645556534873, "learning_rate": 4.297427264508436e-06, "loss": 0.0002, "num_tokens": 90711179.0, "reward": 0.0546874962747097, "reward_std": 0.35632047057151794, "rewards/code_complexity_reward/mean": 0.021484375, "rewards/code_complexity_reward/std": 0.13900451362133026, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 282, "step_time": 45.988181034103036 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 511.494140625, "completions/mean_terminated_length": 447.25, "completions/min_length": 359.0, "completions/min_terminated_length": 359.0, "entropy": 0.503894520457834, "epoch": 0.322690992018244, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0020078513771295547, "kl": 0.0015617331882822327, "learning_rate": 4.290494361724844e-06, "loss": 0.0013, "num_tokens": 91033904.0, "reward": 0.04052734375, "reward_std": 0.3059794008731842, "rewards/code_complexity_reward/mean": 0.01611328125, "rewards/code_complexity_reward/std": 0.12074106186628342, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0087890625, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 283, "step_time": 44.73563222400844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.4969043731689453, "epoch": 0.3238312428734322, "frac_reward_zero_std": 0.9921875, "grad_norm": 0.0008058715611696243, "kl": 0.00157088041305542, "learning_rate": 4.283533072031116e-06, "loss": 0.0, "num_tokens": 91355024.0, "reward": 0.004589843563735485, "reward_std": 0.10385630279779434, "rewards/code_complexity_reward/mean": 0.0016601562965661287, "rewards/code_complexity_reward/std": 0.037565045058727264, "rewards/code_execution_reward/mean": 0.001953125, "rewards/code_execution_reward/std": 0.04419417306780815, "rewards/code_syntax_reward/mean": 0.0009765625, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 284, "step_time": 46.12933992128819 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.49991416931152344, "epoch": 0.3249714937286203, "frac_reward_zero_std": 0.9375, "grad_norm": 0.002377862809225917, "kl": 0.00157146155834198, "learning_rate": 4.276543505793142e-06, "loss": 0.0, "num_tokens": 91675436.0, "reward": 0.03701171651482582, "reward_std": 0.29414084553718567, "rewards/code_complexity_reward/mean": 0.01357421837747097, "rewards/code_complexity_reward/std": 0.10807114094495773, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 285, "step_time": 49.69449058920145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 511.94140625, "completions/mean_terminated_length": 482.0, "completions/min_length": 482.0, "completions/min_terminated_length": 482.0, "entropy": 0.4996693404391408, "epoch": 0.3261117445838084, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.0025445607025176287, "kl": 0.0016241358644037973, "learning_rate": 4.269525773825115e-06, "loss": 0.0002, "num_tokens": 91996538.0, "reward": 0.03789062425494194, "reward_std": 0.3010741174221039, "rewards/code_complexity_reward/mean": 0.014453125186264515, "rewards/code_complexity_reward/std": 0.11491549760103226, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 286, "step_time": 45.58901398349553 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5157508850097656, "epoch": 0.3272519954389966, "frac_reward_zero_std": 0.96875, "grad_norm": 0.0018570433603599668, "kl": 0.00159473717212677, "learning_rate": 4.262479987387776e-06, "loss": 0.0, "num_tokens": 92318730.0, "reward": 0.02373046986758709, "reward_std": 0.23920603096485138, "rewards/code_complexity_reward/mean": 0.00908203050494194, "rewards/code_complexity_reward/std": 0.09157534688711166, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0048828125, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 287, "step_time": 44.66526561882347 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.4885749816894531, "epoch": 0.32839224629418473, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.003263752441853285, "kl": 0.0016152411699295044, "learning_rate": 4.255406258186644e-06, "loss": 0.0, "num_tokens": 92640622.0, "reward": 0.07109375298023224, "reward_std": 0.39254847168922424, "rewards/code_complexity_reward/mean": 0.03105468675494194, "rewards/code_complexity_reward/std": 0.16778883337974548, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.0166015625, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 288, "step_time": 45.566337834112346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5048637390136719, "epoch": 0.3295324971493729, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.0018091145902872086, "kl": 0.001649618148803711, "learning_rate": 4.248304698370253e-06, "loss": 0.0, "num_tokens": 92963190.0, "reward": 0.03281249850988388, "reward_std": 0.2790420651435852, "rewards/code_complexity_reward/mean": 0.012304686941206455, "rewards/code_complexity_reward/std": 0.10480137914419174, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 289, "step_time": 45.148697789758444 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 511.875, "completions/mean_terminated_length": 480.0, "completions/min_length": 459.0, "completions/min_terminated_length": 459.0, "entropy": 0.5011190790683031, "epoch": 0.330672748004561, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.0024888706393539906, "kl": 0.0016729029739508405, "learning_rate": 4.241175420528369e-06, "loss": 0.0002, "num_tokens": 93287826.0, "reward": 0.03623047098517418, "reward_std": 0.2907760739326477, "rewards/code_complexity_reward/mean": 0.014746093191206455, "rewards/code_complexity_reward/std": 0.11717622727155685, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 290, "step_time": 45.600165725685656 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.49337196350097656, "epoch": 0.33181299885974913, "frac_reward_zero_std": 0.921875, "grad_norm": 0.0027869688346982002, "kl": 0.0017144232988357544, "learning_rate": 4.234018537690204e-06, "loss": 0.0, "num_tokens": 93608374.0, "reward": 0.06884765625, "reward_std": 0.39990365505218506, "rewards/code_complexity_reward/mean": 0.02685546875, "rewards/code_complexity_reward/std": 0.1550092250108719, "rewards/code_execution_reward/mean": 0.02734375, "rewards/code_execution_reward/std": 0.16324250400066376, "rewards/code_syntax_reward/mean": 0.0146484375, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 291, "step_time": 51.31793292146176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 511.861328125, "completions/mean_terminated_length": 441.0, "completions/min_length": 441.0, "completions/min_terminated_length": 441.0, "entropy": 0.5008015800267458, "epoch": 0.3329532497149373, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.0024227702524513006, "kl": 0.001686139847151935, "learning_rate": 4.226834163322629e-06, "loss": 0.0004, "num_tokens": 93930939.0, "reward": 0.05439453572034836, "reward_std": 0.35378193855285645, "rewards/code_complexity_reward/mean": 0.02119140699505806, "rewards/code_complexity_reward/std": 0.13701152801513672, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 292, "step_time": 45.37548064626753 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.4917011260986328, "epoch": 0.33409350057012543, "frac_reward_zero_std": 0.921875, "grad_norm": 0.0025398728903383017, "kl": 0.001671329140663147, "learning_rate": 4.21962241132837e-06, "loss": 0.0, "num_tokens": 94252843.0, "reward": 0.05263672024011612, "reward_std": 0.3453604578971863, "rewards/code_complexity_reward/mean": 0.02138671651482582, "rewards/code_complexity_reward/std": 0.13836701214313507, "rewards/code_execution_reward/mean": 0.01953125, "rewards/code_execution_reward/std": 0.1385180652141571, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 293, "step_time": 45.18890702165663 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 511.716796875, "completions/mean_terminated_length": 439.5, "completions/min_length": 409.0, "completions/min_terminated_length": 409.0, "entropy": 0.5069033275358379, "epoch": 0.3352337514253136, "frac_reward_zero_std": 0.96875, "grad_norm": 0.001988923642784357, "kl": 0.0017034539523592684, "learning_rate": 4.212383396044204e-06, "loss": 0.0001, "num_tokens": 94571614.0, "reward": 0.04550781473517418, "reward_std": 0.325361967086792, "rewards/code_complexity_reward/mean": 0.01816406100988388, "rewards/code_complexity_reward/std": 0.12886737287044525, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 294, "step_time": 46.59907653648406 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 511.578125, "completions/mean_terminated_length": 440.0, "completions/min_length": 336.0, "completions/min_terminated_length": 336.0, "entropy": 0.48771215975284576, "epoch": 0.3363740022805017, "frac_reward_zero_std": 0.875, "grad_norm": 0.004140528850257397, "kl": 0.0017110719581978628, "learning_rate": 4.205117232239148e-06, "loss": 0.0008, "num_tokens": 94893534.0, "reward": 0.09482421725988388, "reward_std": 0.4631505012512207, "rewards/code_complexity_reward/mean": 0.03720702975988388, "rewards/code_complexity_reward/std": 0.18028412759304047, "rewards/code_execution_reward/mean": 0.037109375, "rewards/code_execution_reward/std": 0.18921469151973724, "rewards/code_syntax_reward/mean": 0.0205078125, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 295, "step_time": 45.146162739023566 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5060348510742188, "epoch": 0.33751425313568983, "frac_reward_zero_std": 0.921875, "grad_norm": 0.0030106459744274616, "kl": 0.0017642080783843994, "learning_rate": 4.197824035112637e-06, "loss": 0.0, "num_tokens": 95216298.0, "reward": 0.05136718600988388, "reward_std": 0.3287854790687561, "rewards/code_complexity_reward/mean": 0.02304687537252903, "rewards/code_complexity_reward/std": 0.14312726259231567, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0126953125, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 296, "step_time": 45.05041354894638 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 511.623046875, "completions/mean_terminated_length": 415.5, "completions/min_length": 360.0, "completions/min_terminated_length": 360.0, "entropy": 0.49223949387669563, "epoch": 0.338654503990878, "frac_reward_zero_std": 0.9609375, "grad_norm": 0.002018147613853216, "kl": 0.0017615566757740453, "learning_rate": 4.190503920292698e-06, "loss": 0.0006, "num_tokens": 95538717.0, "reward": 0.04257812350988388, "reward_std": 0.30757492780685425, "rewards/code_complexity_reward/mean": 0.01718750037252903, "rewards/code_complexity_reward/std": 0.12210443615913391, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 297, "step_time": 45.800191319547594 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 511.75, "completions/mean_terminated_length": 448.0, "completions/min_length": 392.0, "completions/min_terminated_length": 392.0, "entropy": 0.4989001825451851, "epoch": 0.33979475484606614, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.0027862645220011473, "kl": 0.0017633253628446255, "learning_rate": 4.183157003834118e-06, "loss": 0.0003, "num_tokens": 95860337.0, "reward": 0.05458984524011612, "reward_std": 0.3451841175556183, "rewards/code_complexity_reward/mean": 0.02236328087747097, "rewards/code_complexity_reward/std": 0.13888315856456757, "rewards/code_execution_reward/mean": 0.01953125, "rewards/code_execution_reward/std": 0.1385180652141571, "rewards/code_syntax_reward/mean": 0.0126953125, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 298, "step_time": 46.43351404182613 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 511.703125, "completions/mean_terminated_length": 461.3333435058594, "completions/min_length": 409.0, "completions/min_terminated_length": 409.0, "entropy": 0.4990708972327411, "epoch": 0.3409350057012543, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0026542891282588243, "kl": 0.0017459523787692888, "learning_rate": 4.175783402216604e-06, "loss": 0.0, "num_tokens": 96182541.0, "reward": 0.05615234375, "reward_std": 0.35481899976730347, "rewards/code_complexity_reward/mean": 0.02392577938735485, "rewards/code_complexity_reward/std": 0.14840580523014069, "rewards/code_execution_reward/mean": 0.01953125, "rewards/code_execution_reward/std": 0.1385180652141571, "rewards/code_syntax_reward/mean": 0.0126953125, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 299, "step_time": 45.341929806396365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 511.93359375, "completions/mean_terminated_length": 478.0, "completions/min_length": 478.0, "completions/min_terminated_length": 478.0, "entropy": 0.5011986689642072, "epoch": 0.34207525655644244, "frac_reward_zero_std": 0.90625, "grad_norm": 0.003246480133384466, "kl": 0.001798356801373302, "learning_rate": 4.168383232342934e-06, "loss": 0.0001, "num_tokens": 96503755.0, "reward": 0.05693359673023224, "reward_std": 0.35777056217193604, "rewards/code_complexity_reward/mean": 0.02275390550494194, "rewards/code_complexity_reward/std": 0.14129965007305145, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.0126953125, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 300, "step_time": 52.01456078700721 }, { "epoch": 0.34207525655644244, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 1.0, "eval_completions/max_length": 512.0, "eval_completions/max_terminated_length": 0.0, "eval_completions/mean_length": 512.0, "eval_completions/mean_terminated_length": 0.0, "eval_completions/min_length": 512.0, "eval_completions/min_terminated_length": 0.0, "eval_entropy": 0.5102099609375, "eval_frac_reward_zero_std": 0.95, "eval_kl": 0.001831188201904297, "eval_loss": 9.18108162295539e-06, "eval_num_tokens": 96503755.0, "eval_reward": 0.03662500083446503, "eval_reward_std": 0.0916255009174347, "eval_rewards/code_complexity_reward/mean": 0.014124999791383742, "eval_rewards/code_complexity_reward/std": 0.035311794877052306, "eval_rewards/code_execution_reward/mean": 0.015, "eval_rewards/code_execution_reward/std": 0.037542471885681154, "eval_rewards/code_syntax_reward/mean": 0.0075, "eval_rewards/code_syntax_reward/std": 0.018771235942840577, "eval_rewards/reasoning_present_reward_func/mean": 0.0, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.0, "eval_rewards/xmlcount_reward_func/std": 0.0, "eval_runtime": 1273.679, "eval_samples_per_second": 0.079, "eval_steps_per_second": 0.01, "step": 300 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 511.982421875, "completions/mean_terminated_length": 503.0, "completions/min_length": 503.0, "completions/min_terminated_length": 503.0, "entropy": 0.49859532713890076, "epoch": 0.34321550741163054, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.0026555994991213083, "kl": 0.0017811928846640512, "learning_rate": 4.160956611537106e-06, "loss": 0.0, "num_tokens": 96826470.0, "reward": 0.03408203274011612, "reward_std": 0.2759178578853607, "rewards/code_complexity_reward/mean": 0.01455078087747097, "rewards/code_complexity_reward/std": 0.11568816006183624, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 301, "step_time": 46.09044578950852 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.4969444274902344, "epoch": 0.3443557582668187, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002939994679763913, "kl": 0.001876935362815857, "learning_rate": 4.153503657542479e-06, "loss": 0.0, "num_tokens": 97146990.0, "reward": 0.04228515550494194, "reward_std": 0.3051643669605255, "rewards/code_complexity_reward/mean": 0.01787109300494194, "rewards/code_complexity_reward/std": 0.12697733938694, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 302, "step_time": 52.365705418400466 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 511.72265625, "completions/mean_terminated_length": 441.0, "completions/min_length": 427.0, "completions/min_terminated_length": 427.0, "entropy": 0.5010610083118081, "epoch": 0.34549600912200684, "frac_reward_zero_std": 0.921875, "grad_norm": 0.002971597481518984, "kl": 0.001858644656749675, "learning_rate": 4.146024488519901e-06, "loss": 0.0006, "num_tokens": 97468060.0, "reward": 0.04863281548023224, "reward_std": 0.3224297761917114, "rewards/code_complexity_reward/mean": 0.02128906175494194, "rewards/code_complexity_reward/std": 0.1377975344657898, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 303, "step_time": 45.15393395535648 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 511.4375, "completions/mean_terminated_length": 454.3999938964844, "completions/min_length": 376.0, "completions/min_terminated_length": 376.0, "entropy": 0.4954945119097829, "epoch": 0.346636259977195, "frac_reward_zero_std": 0.921875, "grad_norm": 0.002983706071972847, "kl": 0.0018241370289615588, "learning_rate": 4.138519223045842e-06, "loss": 0.0008, "num_tokens": 97789164.0, "reward": 0.07646484673023224, "reward_std": 0.4171914756298065, "rewards/code_complexity_reward/mean": 0.03056640550494194, "rewards/code_complexity_reward/std": 0.16524982452392578, "rewards/code_execution_reward/mean": 0.029296875, "rewards/code_execution_reward/std": 0.16880230605602264, "rewards/code_syntax_reward/mean": 0.0166015625, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 304, "step_time": 46.32473694905639 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5016689300537109, "epoch": 0.34777651083238315, "frac_reward_zero_std": 0.921875, "grad_norm": 0.002747159218415618, "kl": 0.0018627345561981201, "learning_rate": 4.130987980110508e-06, "loss": 0.0, "num_tokens": 98111552.0, "reward": 0.05507812276482582, "reward_std": 0.3583551049232483, "rewards/code_complexity_reward/mean": 0.02187500149011612, "rewards/code_complexity_reward/std": 0.14149051904678345, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 305, "step_time": 45.830692790448666 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 511.876953125, "completions/mean_terminated_length": 480.5, "completions/min_length": 479.0, "completions/min_terminated_length": 479.0, "entropy": 0.5085569354705513, "epoch": 0.34891676168757124, "frac_reward_zero_std": 0.9375, "grad_norm": 0.002494239015504718, "kl": 0.0018951741622004192, "learning_rate": 4.123430879115963e-06, "loss": 0.0003, "num_tokens": 98433201.0, "reward": 0.05029296875, "reward_std": 0.3422826826572418, "rewards/code_complexity_reward/mean": 0.02001953125, "rewards/code_complexity_reward/std": 0.13532088696956635, "rewards/code_execution_reward/mean": 0.01953125, "rewards/code_execution_reward/std": 0.1385180652141571, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 306, "step_time": 46.08429889101535 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.962890625, "completions/mean_terminated_length": 502.5, "completions/min_length": 494.0, "completions/min_terminated_length": 494.0, "entropy": 0.48953741509467363, "epoch": 0.3500570125427594, "frac_reward_zero_std": 0.90625, "grad_norm": 0.0035787178203463554, "kl": 0.0019326919264130993, "learning_rate": 4.115848039874225e-06, "loss": 0.0001, "num_tokens": 98753058.0, "reward": 0.05791015923023224, "reward_std": 0.36383870244026184, "rewards/code_complexity_reward/mean": 0.02373046800494194, "rewards/code_complexity_reward/std": 0.1472126841545105, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.0126953125, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 307, "step_time": 51.67402595747262 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 511.8359375, "completions/mean_terminated_length": 428.0, "completions/min_length": 428.0, "completions/min_terminated_length": 428.0, "entropy": 0.500834028236568, "epoch": 0.35119726339794755, "frac_reward_zero_std": 0.9375, "grad_norm": 0.002682613907381892, "kl": 0.0019467023230390623, "learning_rate": 4.108239582605374e-06, "loss": 0.0005, "num_tokens": 99074126.0, "reward": 0.04287109524011612, "reward_std": 0.3094767928123474, "rewards/code_complexity_reward/mean": 0.01748046651482582, "rewards/code_complexity_reward/std": 0.1241491511464119, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 308, "step_time": 45.98892742116004 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 511.58984375, "completions/mean_terminated_length": 442.0, "completions/min_length": 370.0, "completions/min_terminated_length": 370.0, "entropy": 0.4948719311505556, "epoch": 0.3523375142531357, "frac_reward_zero_std": 0.9375, "grad_norm": 0.0028183446265757084, "kl": 0.001929832280438859, "learning_rate": 4.100605627935647e-06, "loss": 0.0008, "num_tokens": 99394572.0, "reward": 0.04941406100988388, "reward_std": 0.3271768391132355, "rewards/code_complexity_reward/mean": 0.02207031100988388, "rewards/code_complexity_reward/std": 0.14266546070575714, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 309, "step_time": 51.116065566428006 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 511.96484375, "completions/mean_terminated_length": 494.0, "completions/min_length": 494.0, "completions/min_terminated_length": 494.0, "entropy": 0.4868952869437635, "epoch": 0.35347776510832385, "frac_reward_zero_std": 0.90625, "grad_norm": 0.0033090824726969004, "kl": 0.0019249762844992802, "learning_rate": 4.0929462968955176e-06, "loss": 0.0001, "num_tokens": 99717174.0, "reward": 0.0703125, "reward_std": 0.39799830317497253, "rewards/code_complexity_reward/mean": 0.02929687313735485, "rewards/code_complexity_reward/std": 0.16335251927375793, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 310, "step_time": 45.714657488279045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.703125, "completions/mean_terminated_length": 481.6000061035156, "completions/min_length": 437.0, "completions/min_terminated_length": 437.0, "entropy": 0.5019986992701888, "epoch": 0.35461801596351195, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.0030778232030570507, "kl": 0.0019638383128040005, "learning_rate": 4.085261710917786e-06, "loss": 0.0004, "num_tokens": 100038010.0, "reward": 0.07646484673023224, "reward_std": 0.4171914756298065, "rewards/code_complexity_reward/mean": 0.03056640550494194, "rewards/code_complexity_reward/std": 0.16524982452392578, "rewards/code_execution_reward/mean": 0.029296875, "rewards/code_execution_reward/std": 0.16880230605602264, "rewards/code_syntax_reward/mean": 0.0166015625, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 311, "step_time": 45.860006109811366 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 511.96484375, "completions/mean_terminated_length": 494.0, "completions/min_length": 494.0, "completions/min_terminated_length": 494.0, "entropy": 0.5047066491097212, "epoch": 0.3557582668187001, "frac_reward_zero_std": 0.953125, "grad_norm": 0.002237114356830716, "kl": 0.0019850588505505584, "learning_rate": 4.0775519918356486e-06, "loss": 0.0, "num_tokens": 100361260.0, "reward": 0.0341796875, "reward_std": 0.2767644226551056, "rewards/code_complexity_reward/mean": 0.014648436568677425, "rewards/code_complexity_reward/std": 0.11645561456680298, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 312, "step_time": 51.002581658773124 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 511.85546875, "completions/mean_terminated_length": 487.3333435058594, "completions/min_length": 467.0, "completions/min_terminated_length": 467.0, "entropy": 0.49721730686724186, "epoch": 0.35689851767388825, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.003232432994991541, "kl": 0.0019867856226483127, "learning_rate": 4.069817261880769e-06, "loss": 0.0003, "num_tokens": 100682962.0, "reward": 0.0693359375, "reward_std": 0.3927263915538788, "rewards/code_complexity_reward/mean": 0.02832031063735485, "rewards/code_complexity_reward/std": 0.15804798901081085, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 313, "step_time": 45.81312806997448 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 511.431640625, "completions/mean_terminated_length": 453.8000183105469, "completions/min_length": 399.0, "completions/min_terminated_length": 399.0, "entropy": 0.5030635860748589, "epoch": 0.3580387685290764, "frac_reward_zero_std": 0.90625, "grad_norm": 0.0038556321524083614, "kl": 0.002027945514782914, "learning_rate": 4.062057643681335e-06, "loss": 0.0003, "num_tokens": 101006487.0, "reward": 0.08808593451976776, "reward_std": 0.4424896836280823, "rewards/code_complexity_reward/mean": 0.03535156324505806, "rewards/code_complexity_reward/std": 0.17575743794441223, "rewards/code_execution_reward/mean": 0.033203125, "rewards/code_execution_reward/std": 0.17934183776378632, "rewards/code_syntax_reward/mean": 0.01953125, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 314, "step_time": 45.72808407712728 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 511.341796875, "completions/mean_terminated_length": 427.75, "completions/min_length": 394.0, "completions/min_terminated_length": 394.0, "entropy": 0.4932703133672476, "epoch": 0.35917901938426455, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0022690705955028534, "kl": 0.0020296738475735765, "learning_rate": 4.054273260260125e-06, "loss": 0.0, "num_tokens": 101327674.0, "reward": 0.0639648512005806, "reward_std": 0.3846050202846527, "rewards/code_complexity_reward/mean": 0.02490234375, "rewards/code_complexity_reward/std": 0.14880472421646118, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.013671875, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 315, "step_time": 51.687582878395915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5082550048828125, "epoch": 0.3603192702394527, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0024778235238045454, "kl": 0.002092212438583374, "learning_rate": 4.046464235032546e-06, "loss": 0.0, "num_tokens": 101650118.0, "reward": 0.03583984449505806, "reward_std": 0.28789958357810974, "rewards/code_complexity_reward/mean": 0.01435546949505806, "rewards/code_complexity_reward/std": 0.11413751542568207, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 316, "step_time": 51.6809406010434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.77734375, "completions/mean_terminated_length": 489.20001220703125, "completions/min_length": 472.0, "completions/min_terminated_length": 472.0, "entropy": 0.49532340466976166, "epoch": 0.3614595210946408, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.0038204919546842575, "kl": 0.0021379651607276173, "learning_rate": 4.0386306918046815e-06, "loss": 0.0004, "num_tokens": 101970424.0, "reward": 0.06591796875, "reward_std": 0.3776090443134308, "rewards/code_complexity_reward/mean": 0.02880859375, "rewards/code_complexity_reward/std": 0.16079896688461304, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 317, "step_time": 45.831715850159526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 511.37109375, "completions/mean_terminated_length": 404.66668701171875, "completions/min_length": 358.0, "completions/min_terminated_length": 358.0, "entropy": 0.49294879054650664, "epoch": 0.36259977194982895, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.0032326134387403727, "kl": 0.002143983332643984, "learning_rate": 4.0307727547713316e-06, "loss": 0.0009, "num_tokens": 102291758.0, "reward": 0.06689453125, "reward_std": 0.381536602973938, "rewards/code_complexity_reward/mean": 0.02783203125, "rewards/code_complexity_reward/std": 0.1553412526845932, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 318, "step_time": 45.69190956745297 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 511.9921875, "completions/mean_terminated_length": 508.0, "completions/min_length": 508.0, "completions/min_terminated_length": 508.0, "entropy": 0.4988862741738558, "epoch": 0.3637400228050171, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.0029273517429828644, "kl": 0.0021266544208629057, "learning_rate": 4.0228905485140415e-06, "loss": 0.0, "num_tokens": 102612954.0, "reward": 0.06816405802965164, "reward_std": 0.3884866237640381, "rewards/code_complexity_reward/mean": 0.02910156175494194, "rewards/code_complexity_reward/std": 0.1623360961675644, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 319, "step_time": 52.20732183009386 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5034866333007812, "epoch": 0.36488027366020526, "frac_reward_zero_std": 0.921875, "grad_norm": 0.0027272826991975307, "kl": 0.002158746123313904, "learning_rate": 4.014984197999125e-06, "loss": 0.0, "num_tokens": 102935510.0, "reward": 0.07363281399011612, "reward_std": 0.4125455617904663, "rewards/code_complexity_reward/mean": 0.02871093712747097, "rewards/code_complexity_reward/std": 0.1602526754140854, "rewards/code_execution_reward/mean": 0.029296875, "rewards/code_execution_reward/std": 0.16880230605602264, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 320, "step_time": 47.52038327232003 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 511.927734375, "completions/mean_terminated_length": 475.0, "completions/min_length": 475.0, "completions/min_terminated_length": 475.0, "entropy": 0.4949114751070738, "epoch": 0.3660205245153934, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002914642682299018, "kl": 0.0022349314003804466, "learning_rate": 4.007053828575684e-06, "loss": 0.0002, "num_tokens": 103256553.0, "reward": 0.05625000223517418, "reward_std": 0.36348769068717957, "rewards/code_complexity_reward/mean": 0.02109374850988388, "rewards/code_complexity_reward/std": 0.13640038669109344, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 321, "step_time": 51.5972062041983 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.830078125, "completions/mean_terminated_length": 490.25, "completions/min_length": 464.0, "completions/min_terminated_length": 464.0, "entropy": 0.4925583670847118, "epoch": 0.3671607753705815, "frac_reward_zero_std": 0.8671875, "grad_norm": 0.004216661211103201, "kl": 0.002318329552508658, "learning_rate": 3.999099565973623e-06, "loss": 0.0002, "num_tokens": 103578350.0, "reward": 0.11337890475988388, "reward_std": 0.49121612310409546, "rewards/code_complexity_reward/mean": 0.04794921725988388, "rewards/code_complexity_reward/std": 0.2037099450826645, "rewards/code_execution_reward/mean": 0.0390625, "rewards/code_execution_reward/std": 0.1939331740140915, "rewards/code_syntax_reward/mean": 0.0263671875, "rewards/code_syntax_reward/std": 0.11186064779758453, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 322, "step_time": 45.247491153888404 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 511.912109375, "completions/mean_terminated_length": 489.5, "completions/min_length": 475.0, "completions/min_terminated_length": 475.0, "entropy": 0.4909856575541198, "epoch": 0.36830102622576966, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.0023332848213613033, "kl": 0.0022549613277078606, "learning_rate": 3.991121536301653e-06, "loss": 0.0001, "num_tokens": 103900433.0, "reward": 0.04345703125, "reward_std": 0.31297942996025085, "rewards/code_complexity_reward/mean": 0.01806640625, "rewards/code_complexity_reward/std": 0.12825323641300201, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 323, "step_time": 44.9039688128978 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.93359375, "completions/mean_terminated_length": 500.66668701171875, "completions/min_length": 489.0, "completions/min_terminated_length": 489.0, "entropy": 0.49264511466026306, "epoch": 0.3694412770809578, "frac_reward_zero_std": 0.953125, "grad_norm": 0.002298475708812475, "kl": 0.002277117680932861, "learning_rate": 3.983119866045297e-06, "loss": 0.0002, "num_tokens": 104221543.0, "reward": 0.03427734225988388, "reward_std": 0.2775907516479492, "rewards/code_complexity_reward/mean": 0.014746093191206455, "rewards/code_complexity_reward/std": 0.11717622727155685, "rewards/code_execution_reward/mean": 0.01171875, "rewards/code_execution_reward/std": 0.10772226005792618, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 324, "step_time": 51.13270273990929 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 511.85546875, "completions/mean_terminated_length": 475.0, "completions/min_length": 468.0, "completions/min_terminated_length": 468.0, "entropy": 0.49917575530707836, "epoch": 0.37058152793614596, "frac_reward_zero_std": 0.875, "grad_norm": 0.003659483278170228, "kl": 0.002384877974691335, "learning_rate": 3.975094682064875e-06, "loss": 0.0004, "num_tokens": 104541913.0, "reward": 0.09033203125, "reward_std": 0.4465450048446655, "rewards/code_complexity_reward/mean": 0.03662109375, "rewards/code_complexity_reward/std": 0.17797432839870453, "rewards/code_execution_reward/mean": 0.033203125, "rewards/code_execution_reward/std": 0.17934183776378632, "rewards/code_syntax_reward/mean": 0.0205078125, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 325, "step_time": 51.22381878178567 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5008430480957031, "epoch": 0.3717217787913341, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0022492017596960068, "kl": 0.0023585259914398193, "learning_rate": 3.967046111593505e-06, "loss": 0.0, "num_tokens": 104863245.0, "reward": 0.02666015736758709, "reward_std": 0.24831561744213104, "rewards/code_complexity_reward/mean": 0.01103515550494194, "rewards/code_complexity_reward/std": 0.10145854949951172, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.005859375, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 326, "step_time": 46.19941683020443 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 511.875, "completions/mean_terminated_length": 480.0, "completions/min_length": 479.0, "completions/min_terminated_length": 479.0, "entropy": 0.5025170287117362, "epoch": 0.3728620296465222, "frac_reward_zero_std": 0.921875, "grad_norm": 0.0028879987075924873, "kl": 0.002449229355988791, "learning_rate": 3.958974282235079e-06, "loss": 0.0003, "num_tokens": 105184801.0, "reward": 0.06210937350988388, "reward_std": 0.3750014305114746, "rewards/code_complexity_reward/mean": 0.02499999850988388, "rewards/code_complexity_reward/std": 0.14946088194847107, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.013671875, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 327, "step_time": 46.005374828353524 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 511.98828125, "completions/mean_terminated_length": 506.0, "completions/min_length": 506.0, "completions/min_terminated_length": 506.0, "entropy": 0.48715475387871265, "epoch": 0.37400228050171036, "frac_reward_zero_std": 0.890625, "grad_norm": 0.0036161078605800867, "kl": 0.0024060845498752315, "learning_rate": 3.9508793219622375e-06, "loss": 0.0, "num_tokens": 105506275.0, "reward": 0.06953124701976776, "reward_std": 0.39415910840034485, "rewards/code_complexity_reward/mean": 0.02851562574505806, "rewards/code_complexity_reward/std": 0.1591850072145462, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 328, "step_time": 45.81074404809624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.503692626953125, "epoch": 0.3751425313568985, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.002611270174384117, "kl": 0.0024178624153137207, "learning_rate": 3.942761359114345e-06, "loss": 0.0, "num_tokens": 105828663.0, "reward": 0.02910156548023224, "reward_std": 0.25285604596138, "rewards/code_complexity_reward/mean": 0.01249999925494194, "rewards/code_complexity_reward/std": 0.10630790144205093, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0068359375, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 329, "step_time": 46.32880854513496 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 511.9765625, "completions/mean_terminated_length": 506.0, "completions/min_length": 504.0, "completions/min_terminated_length": 504.0, "entropy": 0.5114091453142464, "epoch": 0.37628278221208666, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0023544072173535824, "kl": 0.0023915151214168873, "learning_rate": 3.934620522395458e-06, "loss": 0.0001, "num_tokens": 106150699.0, "reward": 0.04531250149011612, "reward_std": 0.3245461881160736, "rewards/code_complexity_reward/mean": 0.01796874962747097, "rewards/code_complexity_reward/std": 0.12752103805541992, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.009765625, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 330, "step_time": 46.04020596295595 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 511.78515625, "completions/mean_terminated_length": 402.0, "completions/min_length": 402.0, "completions/min_terminated_length": 402.0, "entropy": 0.4971674354746938, "epoch": 0.3774230330672748, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.003101550741121173, "kl": 0.0024284938372147735, "learning_rate": 3.926456940872274e-06, "loss": 0.0006, "num_tokens": 106473033.0, "reward": 0.04555664211511612, "reward_std": 0.31125500798225403, "rewards/code_complexity_reward/mean": 0.02041015587747097, "rewards/code_complexity_reward/std": 0.13787749409675598, "rewards/code_execution_reward/mean": 0.013671875, "rewards/code_execution_reward/std": 0.1162383034825325, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 331, "step_time": 45.84701434336603 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 511.630859375, "completions/mean_terminated_length": 417.5, "completions/min_length": 345.0, "completions/min_terminated_length": 345.0, "entropy": 0.4855421790853143, "epoch": 0.37856328392246297, "frac_reward_zero_std": 0.921875, "grad_norm": 0.0031037095468491316, "kl": 0.0025116715441981796, "learning_rate": 3.918270743972097e-06, "loss": 0.0009, "num_tokens": 106795724.0, "reward": 0.05312500149011612, "reward_std": 0.33707618713378906, "rewards/code_complexity_reward/mean": 0.02285156399011612, "rewards/code_complexity_reward/std": 0.1418885439634323, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.0126953125, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 332, "step_time": 45.9407983282581 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 458.0, "completions/mean_length": 511.728515625, "completions/mean_terminated_length": 442.5, "completions/min_length": 427.0, "completions/min_terminated_length": 427.0, "entropy": 0.49611143954098225, "epoch": 0.37970353477765106, "frac_reward_zero_std": 0.921875, "grad_norm": 0.003175242803990841, "kl": 0.0025544340460328385, "learning_rate": 3.910062061480778e-06, "loss": 0.0008, "num_tokens": 107115581.0, "reward": 0.04765625298023224, "reward_std": 0.32727357745170593, "rewards/code_complexity_reward/mean": 0.01933593861758709, "rewards/code_complexity_reward/std": 0.13088269531726837, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 333, "step_time": 51.845130414702 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 511.798828125, "completions/mean_terminated_length": 409.0, "completions/min_length": 409.0, "completions/min_terminated_length": 409.0, "entropy": 0.49455487355589867, "epoch": 0.3808437856328392, "frac_reward_zero_std": 0.890625, "grad_norm": 0.003905928460881114, "kl": 0.0025665774955996312, "learning_rate": 3.901831023540662e-06, "loss": 0.0004, "num_tokens": 107438294.0, "reward": 0.08281250298023224, "reward_std": 0.42937344312667847, "rewards/code_complexity_reward/mean": 0.03300781175494194, "rewards/code_complexity_reward/std": 0.1691012680530548, "rewards/code_execution_reward/mean": 0.03125, "rewards/code_execution_reward/std": 0.17416280508041382, "rewards/code_syntax_reward/mean": 0.0185546875, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 334, "step_time": 50.474500990472734 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 511.5859375, "completions/mean_terminated_length": 469.6000061035156, "completions/min_length": 411.0, "completions/min_terminated_length": 411.0, "entropy": 0.49638779973611236, "epoch": 0.38198403648802737, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.002993721980601549, "kl": 0.002513106432161294, "learning_rate": 3.89357776064852e-06, "loss": 0.0011, "num_tokens": 107760370.0, "reward": 0.08955077826976776, "reward_std": 0.4424991309642792, "rewards/code_complexity_reward/mean": 0.03779296949505806, "rewards/code_complexity_reward/std": 0.1830979734659195, "rewards/code_execution_reward/mean": 0.03125, "rewards/code_execution_reward/std": 0.17416280508041382, "rewards/code_syntax_reward/mean": 0.0205078125, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 335, "step_time": 44.99571338482201 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 511.880859375, "completions/mean_terminated_length": 481.5, "completions/min_length": 460.0, "completions/min_terminated_length": 460.0, "entropy": 0.49994388734921813, "epoch": 0.3831242873432155, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.003151219105347991, "kl": 0.00261169690566021, "learning_rate": 3.885302403653483e-06, "loss": 0.0002, "num_tokens": 108084233.0, "reward": 0.07158203423023224, "reward_std": 0.4034320116043091, "rewards/code_complexity_reward/mean": 0.02861328050494194, "rewards/code_complexity_reward/std": 0.15964315831661224, "rewards/code_execution_reward/mean": 0.02734375, "rewards/code_execution_reward/std": 0.16324250400066376, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 336, "step_time": 45.849324610084295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 511.630859375, "completions/mean_terminated_length": 474.20001220703125, "completions/min_length": 424.0, "completions/min_terminated_length": 424.0, "entropy": 0.482668858487159, "epoch": 0.38426453819840367, "frac_reward_zero_std": 0.890625, "grad_norm": 0.003611429361626506, "kl": 0.0027076194310211577, "learning_rate": 3.8770050837549675e-06, "loss": 0.0011, "num_tokens": 108406764.0, "reward": 0.060546875, "reward_std": 0.35913586616516113, "rewards/code_complexity_reward/mean": 0.0263671875, "rewards/code_complexity_reward/std": 0.15246856212615967, "rewards/code_execution_reward/mean": 0.01953125, "rewards/code_execution_reward/std": 0.1385180652141571, "rewards/code_syntax_reward/mean": 0.0146484375, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 337, "step_time": 46.85355653986335 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.986328125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 511.431640625, "completions/mean_terminated_length": 470.4285888671875, "completions/min_length": 443.0, "completions/min_terminated_length": 443.0, "entropy": 0.4950226149521768, "epoch": 0.38540478905359177, "frac_reward_zero_std": 0.859375, "grad_norm": 0.00412735203281045, "kl": 0.0026849408459383994, "learning_rate": 3.868685932500596e-06, "loss": 0.0012, "num_tokens": 108727533.0, "reward": 0.10380859673023224, "reward_std": 0.47626662254333496, "rewards/code_complexity_reward/mean": 0.04326171800494194, "rewards/code_complexity_reward/std": 0.19544346630573273, "rewards/code_execution_reward/mean": 0.037109375, "rewards/code_execution_reward/std": 0.18921469151973724, "rewards/code_syntax_reward/mean": 0.0234375, "rewards/code_syntax_reward/std": 0.10578890144824982, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 338, "step_time": 45.89778223447502 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.4833812713623047, "epoch": 0.3865450399087799, "frac_reward_zero_std": 0.9453125, "grad_norm": 0.0028004301711916924, "kl": 0.002698570489883423, "learning_rate": 3.860345081784107e-06, "loss": 0.0, "num_tokens": 109049737.0, "reward": 0.05380859225988388, "reward_std": 0.3499372899532318, "rewards/code_complexity_reward/mean": 0.02060546912252903, "rewards/code_complexity_reward/std": 0.13337492942810059, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 339, "step_time": 52.077924902550876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 511.8046875, "completions/mean_terminated_length": 487.0, "completions/min_length": 443.0, "completions/min_terminated_length": 443.0, "entropy": 0.4942458509467542, "epoch": 0.38768529076396807, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002871341537684202, "kl": 0.002802865557896439, "learning_rate": 3.851982663843272e-06, "loss": 0.0003, "num_tokens": 109369761.0, "reward": 0.05781250074505806, "reward_std": 0.3629455268383026, "rewards/code_complexity_reward/mean": 0.02363281138241291, "rewards/code_complexity_reward/std": 0.14664575457572937, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.0126953125, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 340, "step_time": 46.77297986578196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.974609375, "completions/mean_terminated_length": 507.66668701171875, "completions/min_length": 503.0, "completions/min_terminated_length": 503.0, "entropy": 0.49460627418011427, "epoch": 0.3888255416191562, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.0036864725407212973, "kl": 0.0027731457812478766, "learning_rate": 3.84359881125779e-06, "loss": 0.0001, "num_tokens": 109692948.0, "reward": 0.06748046725988388, "reward_std": 0.3832246661186218, "rewards/code_complexity_reward/mean": 0.02841796912252903, "rewards/code_complexity_reward/std": 0.1589718610048294, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 341, "step_time": 45.71947215218097 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.98828125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 511.18359375, "completions/mean_terminated_length": 442.3333435058594, "completions/min_length": 329.0, "completions/min_terminated_length": 329.0, "entropy": 0.5013349545188248, "epoch": 0.3899657924743444, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.0025929335970431566, "kl": 0.0027615099388640374, "learning_rate": 3.835193656947192e-06, "loss": 0.0007, "num_tokens": 110012906.0, "reward": 0.0595703125, "reward_std": 0.3614579737186432, "rewards/code_complexity_reward/mean": 0.0244140625, "rewards/code_complexity_reward/std": 0.14588165283203125, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.013671875, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 342, "step_time": 46.25398394279182 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.48343849182128906, "epoch": 0.39110604332953247, "frac_reward_zero_std": 0.90625, "grad_norm": 0.003249147441238165, "kl": 0.0030299872159957886, "learning_rate": 3.826767334168731e-06, "loss": 0.0, "num_tokens": 110333534.0, "reward": 0.09091796725988388, "reward_std": 0.447695255279541, "rewards/code_complexity_reward/mean": 0.03720702975988388, "rewards/code_complexity_reward/std": 0.18047396838665009, "rewards/code_execution_reward/mean": 0.033203125, "rewards/code_execution_reward/std": 0.17934183776378632, "rewards/code_syntax_reward/mean": 0.0205078125, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 343, "step_time": 45.332096382044256 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 511.57421875, "completions/mean_terminated_length": 439.3333435058594, "completions/min_length": 394.0, "completions/min_terminated_length": 394.0, "entropy": 0.49290309892967343, "epoch": 0.3922462941847206, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.0032696453854441643, "kl": 0.002887235434172908, "learning_rate": 3.8183199765152704e-06, "loss": 0.0004, "num_tokens": 110655892.0, "reward": 0.08320312947034836, "reward_std": 0.43836459517478943, "rewards/code_complexity_reward/mean": 0.03242187574505806, "rewards/code_complexity_reward/std": 0.17019498348236084, "rewards/code_execution_reward/mean": 0.033203125, "rewards/code_execution_reward/std": 0.17934183776378632, "rewards/code_syntax_reward/mean": 0.017578125, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 344, "step_time": 50.94522703438997 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 511.8125, "completions/mean_terminated_length": 480.0, "completions/min_length": 471.0, "completions/min_terminated_length": 471.0, "entropy": 0.4978583948686719, "epoch": 0.3933865450399088, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.004166281782090664, "kl": 0.0029832505679223686, "learning_rate": 3.809851717913164e-06, "loss": 0.0006, "num_tokens": 110977128.0, "reward": 0.06240234524011612, "reward_std": 0.3675132989883423, "rewards/code_complexity_reward/mean": 0.02626953087747097, "rewards/code_complexity_reward/std": 0.1516973376274109, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.0146484375, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 345, "step_time": 51.08100237138569 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 511.76953125, "completions/mean_terminated_length": 453.0, "completions/min_length": 434.0, "completions/min_terminated_length": 434.0, "entropy": 0.49317307071760297, "epoch": 0.3945267958950969, "frac_reward_zero_std": 0.953125, "grad_norm": 0.0024897719267755747, "kl": 0.002895938523579389, "learning_rate": 3.8013626926201343e-06, "loss": 0.0004, "num_tokens": 111298550.0, "reward": 0.03173828125, "reward_std": 0.2598801255226135, "rewards/code_complexity_reward/mean": 0.014160155318677425, "rewards/code_complexity_reward/std": 0.11256517469882965, "rewards/code_execution_reward/mean": 0.009765625, "rewards/code_execution_reward/std": 0.09843364357948303, "rewards/code_syntax_reward/mean": 0.0078125, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 346, "step_time": 46.30949283018708 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 511.8125, "completions/mean_terminated_length": 464.0, "completions/min_length": 456.0, "completions/min_terminated_length": 456.0, "entropy": 0.49164394941180944, "epoch": 0.3956670467502851, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.0030321015510708094, "kl": 0.0030203222995623946, "learning_rate": 3.792853035223144e-06, "loss": 0.0004, "num_tokens": 111619338.0, "reward": 0.05058594048023224, "reward_std": 0.33375003933906555, "rewards/code_complexity_reward/mean": 0.02128906175494194, "rewards/code_complexity_reward/std": 0.13776202499866486, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 347, "step_time": 45.706929757259786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.7109375, "completions/mean_terminated_length": 475.0, "completions/min_length": 413.0, "completions/min_terminated_length": 413.0, "entropy": 0.4858338003978133, "epoch": 0.39680729760547323, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002935437485575676, "kl": 0.0030333481590787414, "learning_rate": 3.7843228806362635e-06, "loss": 0.0006, "num_tokens": 111939526.0, "reward": 0.0537109375, "reward_std": 0.34142541885375977, "rewards/code_complexity_reward/mean": 0.0234375, "rewards/code_complexity_reward/std": 0.14547143876552582, "rewards/code_execution_reward/mean": 0.017578125, "rewards/code_execution_reward/std": 0.13154059648513794, "rewards/code_syntax_reward/mean": 0.0126953125, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 348, "step_time": 45.7832544054836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 511.306640625, "completions/mean_terminated_length": 393.66668701171875, "completions/min_length": 342.0, "completions/min_terminated_length": 342.0, "entropy": 0.4912582952529192, "epoch": 0.3979475484606613, "frac_reward_zero_std": 0.921875, "grad_norm": 0.0031762714497745037, "kl": 0.003200290178938303, "learning_rate": 3.775772364098529e-06, "loss": 0.0008, "num_tokens": 112259515.0, "reward": 0.05654297024011612, "reward_std": 0.3653821051120758, "rewards/code_complexity_reward/mean": 0.02138671837747097, "rewards/code_complexity_reward/std": 0.13829627633094788, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 349, "step_time": 45.379896010272205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 511.291015625, "completions/mean_terminated_length": 421.25, "completions/min_length": 362.0, "completions/min_terminated_length": 362.0, "entropy": 0.4836463574320078, "epoch": 0.3990877993158495, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.0033475009258836508, "kl": 0.003143364203424426, "learning_rate": 3.7672016211717977e-06, "loss": 0.0014, "num_tokens": 112580500.0, "reward": 0.08603515475988388, "reward_std": 0.4347720444202423, "rewards/code_complexity_reward/mean": 0.03525390475988388, "rewards/code_complexity_reward/std": 0.17526143789291382, "rewards/code_execution_reward/mean": 0.03125, "rewards/code_execution_reward/std": 0.17416280508041382, "rewards/code_syntax_reward/mean": 0.01953125, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 350, "step_time": 46.3818280082196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 511.7578125, "completions/mean_terminated_length": 470.66668701171875, "completions/min_length": 453.0, "completions/min_terminated_length": 453.0, "entropy": 0.4923890600912273, "epoch": 0.40022805017103763, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.002941593760624528, "kl": 0.0031011568898975383, "learning_rate": 3.758610787738604e-06, "loss": 0.0003, "num_tokens": 112902560.0, "reward": 0.06230469048023224, "reward_std": 0.36769625544548035, "rewards/code_complexity_reward/mean": 0.02617187425494194, "rewards/code_complexity_reward/std": 0.15179485082626343, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.0146484375, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 351, "step_time": 52.27508026454598 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 511.95703125, "completions/mean_terminated_length": 490.0, "completions/min_length": 490.0, "completions/min_terminated_length": 490.0, "entropy": 0.4964366601780057, "epoch": 0.4013683010262258, "frac_reward_zero_std": 0.921875, "grad_norm": 0.0028737529646605253, "kl": 0.003167693590512499, "learning_rate": 3.7500000000000005e-06, "loss": 0.0001, "num_tokens": 113223346.0, "reward": 0.05009765923023224, "reward_std": 0.3414957821369171, "rewards/code_complexity_reward/mean": 0.01982421800494194, "rewards/code_complexity_reward/std": 0.134005606174469, "rewards/code_execution_reward/mean": 0.01953125, "rewards/code_execution_reward/std": 0.1385180652141571, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 352, "step_time": 51.302593065425754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 511.87109375, "completions/mean_terminated_length": 495.5, "completions/min_length": 465.0, "completions/min_terminated_length": 465.0, "entropy": 0.4822723586112261, "epoch": 0.40250855188141393, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.0033288230188190937, "kl": 0.00322551323188236, "learning_rate": 3.7413693944734e-06, "loss": 0.0002, "num_tokens": 113543332.0, "reward": 0.07016602158546448, "reward_std": 0.3959306478500366, "rewards/code_complexity_reward/mean": 0.02890625037252903, "rewards/code_complexity_reward/std": 0.16122201085090637, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 353, "step_time": 46.238775100558996 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.8359375, "completions/mean_terminated_length": 484.0, "completions/min_length": 437.0, "completions/min_terminated_length": 437.0, "entropy": 0.5018138568848372, "epoch": 0.40364880273660203, "frac_reward_zero_std": 0.90625, "grad_norm": 0.004107371903955936, "kl": 0.0031973922996257897, "learning_rate": 3.7327191079904096e-06, "loss": 0.0005, "num_tokens": 113864284.0, "reward": 0.06005859375, "reward_std": 0.3654761016368866, "rewards/code_complexity_reward/mean": 0.02490234375, "rewards/code_complexity_reward/std": 0.1489361822605133, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.013671875, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 354, "step_time": 52.538495604880154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 511.708984375, "completions/mean_terminated_length": 363.0, "completions/min_length": 363.0, "completions/min_terminated_length": 363.0, "entropy": 0.5029452722519636, "epoch": 0.4047890535917902, "frac_reward_zero_std": 0.890625, "grad_norm": 0.003698466345667839, "kl": 0.003217647143173963, "learning_rate": 3.7240492776946663e-06, "loss": -0.0003, "num_tokens": 114184319.0, "reward": 0.06787109375, "reward_std": 0.378492146730423, "rewards/code_complexity_reward/mean": 0.02978515438735485, "rewards/code_complexity_reward/std": 0.16138027608394623, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.0166015625, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 355, "step_time": 44.9970733942464 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.48272705078125, "epoch": 0.40592930444697833, "frac_reward_zero_std": 0.921875, "grad_norm": 0.0035221201833337545, "kl": 0.003107205033302307, "learning_rate": 3.7153600410396558e-06, "loss": 0.0, "num_tokens": 114505131.0, "reward": 0.06142578274011612, "reward_std": 0.37233245372772217, "rewards/code_complexity_reward/mean": 0.02431640587747097, "rewards/code_complexity_reward/std": 0.14551185071468353, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.013671875, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 356, "step_time": 46.204270779155195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.4876251220703125, "epoch": 0.4070695553021665, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.0033124082256108522, "kl": 0.003254726529121399, "learning_rate": 3.7066515357865384e-06, "loss": 0.0, "num_tokens": 114829831.0, "reward": 0.06894531100988388, "reward_std": 0.3826374411582947, "rewards/code_complexity_reward/mean": 0.03085937350988388, "rewards/code_complexity_reward/std": 0.16677217185497284, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.0166015625, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 357, "step_time": 46.24454983416945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 511.8671875, "completions/mean_terminated_length": 444.0, "completions/min_length": 444.0, "completions/min_terminated_length": 444.0, "entropy": 0.4865068830549717, "epoch": 0.40820980615735464, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.003331919200718403, "kl": 0.0033607705772737972, "learning_rate": 3.6979239000019622e-06, "loss": 0.0002, "num_tokens": 115151179.0, "reward": 0.06865234673023224, "reward_std": 0.3900909721851349, "rewards/code_complexity_reward/mean": 0.02763671800494194, "rewards/code_complexity_reward/std": 0.15471355617046356, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 358, "step_time": 52.442013991996646 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 511.56640625, "completions/mean_terminated_length": 438.0, "completions/min_length": 386.0, "completions/min_terminated_length": 386.0, "entropy": 0.4974302798509598, "epoch": 0.40935005701254273, "frac_reward_zero_std": 0.8828125, "grad_norm": 0.003783579682931304, "kl": 0.0034116393617296126, "learning_rate": 3.689177272055877e-06, "loss": 0.0012, "num_tokens": 115474869.0, "reward": 0.08447265625, "reward_std": 0.42818745970726013, "rewards/code_complexity_reward/mean": 0.03564453125, "rewards/code_complexity_reward/std": 0.17718161642551422, "rewards/code_execution_reward/mean": 0.029296875, "rewards/code_execution_reward/std": 0.16880230605602264, "rewards/code_syntax_reward/mean": 0.01953125, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 359, "step_time": 45.495277093723416 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.982421875, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 510.63671875, "completions/mean_terminated_length": 434.4444580078125, "completions/min_length": 366.0, "completions/min_terminated_length": 366.0, "entropy": 0.4825796843506396, "epoch": 0.4104903078677309, "frac_reward_zero_std": 0.890625, "grad_norm": 0.003768930211663246, "kl": 0.0033951314944715705, "learning_rate": 3.6804117906193367e-06, "loss": 0.0008, "num_tokens": 115795995.0, "reward": 0.10976562649011612, "reward_std": 0.4991404712200165, "rewards/code_complexity_reward/mean": 0.04335937649011612, "rewards/code_complexity_reward/std": 0.19593431055545807, "rewards/code_execution_reward/mean": 0.04296875, "rewards/code_execution_reward/std": 0.2029850035905838, "rewards/code_syntax_reward/mean": 0.0234375, "rewards/code_syntax_reward/std": 0.10578890144824982, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 360, "step_time": 46.26161209307611 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.79296875, "completions/mean_terminated_length": 476.66668701171875, "completions/min_length": 452.0, "completions/min_terminated_length": 452.0, "entropy": 0.49560797587037086, "epoch": 0.41163055872291904, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.0031859290320426226, "kl": 0.0033110630174633116, "learning_rate": 3.671627594662303e-06, "loss": 0.0004, "num_tokens": 116117537.0, "reward": 0.05117187649011612, "reward_std": 0.32650163769721985, "rewards/code_complexity_reward/mean": 0.02285156212747097, "rewards/code_complexity_reward/std": 0.14195749163627625, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0126953125, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 361, "step_time": 49.90447509661317 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.98828125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 511.296875, "completions/mean_terminated_length": 452.0, "completions/min_length": 374.0, "completions/min_terminated_length": 374.0, "entropy": 0.49566992931067944, "epoch": 0.4127708095781072, "frac_reward_zero_std": 0.8828125, "grad_norm": 0.00394066795706749, "kl": 0.0036057345605513547, "learning_rate": 3.6628248234514434e-06, "loss": 0.0011, "num_tokens": 116439429.0, "reward": 0.10039062052965164, "reward_std": 0.4563852548599243, "rewards/code_complexity_reward/mean": 0.04472656548023224, "rewards/code_complexity_reward/std": 0.19793830811977386, "rewards/code_execution_reward/mean": 0.03125, "rewards/code_execution_reward/std": 0.17416280508041382, "rewards/code_syntax_reward/mean": 0.0244140625, "rewards/code_syntax_reward/std": 0.10785966366529465, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 362, "step_time": 50.56124278809875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.587890625, "completions/mean_terminated_length": 459.25, "completions/min_length": 348.0, "completions/min_terminated_length": 348.0, "entropy": 0.5057316045276821, "epoch": 0.41391106043329534, "frac_reward_zero_std": 0.9375, "grad_norm": 0.003312233602628112, "kl": 0.0035028448910452425, "learning_rate": 3.6540036165479203e-06, "loss": 0.0012, "num_tokens": 116761726.0, "reward": 0.05292968824505806, "reward_std": 0.3476946949958801, "rewards/code_complexity_reward/mean": 0.02167968824505806, "rewards/code_complexity_reward/std": 0.14037519693374634, "rewards/code_execution_reward/mean": 0.01953125, "rewards/code_execution_reward/std": 0.1385180652141571, "rewards/code_syntax_reward/mean": 0.01171875, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 363, "step_time": 44.779079323634505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.544921875, "completions/mean_terminated_length": 453.75, "completions/min_length": 398.0, "completions/min_terminated_length": 398.0, "entropy": 0.4718871833756566, "epoch": 0.4150513112884835, "frac_reward_zero_std": 0.84375, "grad_norm": 0.004908151924610138, "kl": 0.0035326896468177438, "learning_rate": 3.6451641138051806e-06, "loss": 0.001, "num_tokens": 117084285.0, "reward": 0.11337890475988388, "reward_std": 0.49861031770706177, "rewards/code_complexity_reward/mean": 0.04501952975988388, "rewards/code_complexity_reward/std": 0.1955212950706482, "rewards/code_execution_reward/mean": 0.04296875, "rewards/code_execution_reward/std": 0.2029850035905838, "rewards/code_syntax_reward/mean": 0.025390625, "rewards/code_syntax_reward/std": 0.10988271236419678, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 364, "step_time": 52.97949921712279 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.986328125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.6796875, "completions/mean_terminated_length": 488.5714416503906, "completions/min_length": 427.0, "completions/min_terminated_length": 427.0, "entropy": 0.4972880706191063, "epoch": 0.4161915621436716, "frac_reward_zero_std": 0.890625, "grad_norm": 0.0036693362053483725, "kl": 0.003590385695133591, "learning_rate": 3.6363064553667378e-06, "loss": 0.0003, "num_tokens": 117404881.0, "reward": 0.12382812798023224, "reward_std": 0.5164501667022705, "rewards/code_complexity_reward/mean": 0.05253906175494194, "rewards/code_complexity_reward/std": 0.21482117474079132, "rewards/code_execution_reward/mean": 0.04296875, "rewards/code_execution_reward/std": 0.2029850035905838, "rewards/code_syntax_reward/mean": 0.0283203125, "rewards/code_syntax_reward/std": 0.11569035053253174, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 365, "step_time": 45.715394890867174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 511.77734375, "completions/mean_terminated_length": 483.5, "completions/min_length": 460.0, "completions/min_terminated_length": 460.0, "entropy": 0.48993656737729907, "epoch": 0.41733181299885974, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.0038829909171909094, "kl": 0.003655603955849074, "learning_rate": 3.627430781663948e-06, "loss": 0.0003, "num_tokens": 117726931.0, "reward": 0.08027344197034836, "reward_std": 0.41803932189941406, "rewards/code_complexity_reward/mean": 0.03242187574505806, "rewards/code_complexity_reward/std": 0.16618074476718903, "rewards/code_execution_reward/mean": 0.029296875, "rewards/code_execution_reward/std": 0.16880230605602264, "rewards/code_syntax_reward/mean": 0.0185546875, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 366, "step_time": 46.52245946601033 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.580078125, "completions/mean_terminated_length": 440.3333435058594, "completions/min_length": 376.0, "completions/min_terminated_length": 376.0, "entropy": 0.4862046130001545, "epoch": 0.4184720638540479, "frac_reward_zero_std": 0.8828125, "grad_norm": 0.0038262244779616594, "kl": 0.0036488957121036947, "learning_rate": 3.618537233413789e-06, "loss": 0.0009, "num_tokens": 118048408.0, "reward": 0.083984375, "reward_std": 0.4338619112968445, "rewards/code_complexity_reward/mean": 0.0341796875, "rewards/code_complexity_reward/std": 0.17445388436317444, "rewards/code_execution_reward/mean": 0.03125, "rewards/code_execution_reward/std": 0.17416280508041382, "rewards/code_syntax_reward/mean": 0.0185546875, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 367, "step_time": 46.13383572362363 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.8515625, "completions/mean_terminated_length": 486.66668701171875, "completions/min_length": 474.0, "completions/min_terminated_length": 474.0, "entropy": 0.4947885684669018, "epoch": 0.41961231470923605, "frac_reward_zero_std": 0.90625, "grad_norm": 0.0035624525044113398, "kl": 0.0037315250374376774, "learning_rate": 3.6096259516166226e-06, "loss": 0.0003, "num_tokens": 118370328.0, "reward": 0.06425781548023224, "reward_std": 0.37823858857154846, "rewards/code_complexity_reward/mean": 0.02617187425494194, "rewards/code_complexity_reward/std": 0.15143990516662598, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.0146484375, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 368, "step_time": 45.83201956562698 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 511.677734375, "completions/mean_terminated_length": 429.5, "completions/min_length": 424.0, "completions/min_terminated_length": 424.0, "entropy": 0.49866973608732224, "epoch": 0.4207525655644242, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.004012160934507847, "kl": 0.003762927841307828, "learning_rate": 3.600697077553964e-06, "loss": 0.0007, "num_tokens": 118691775.0, "reward": 0.07285156100988388, "reward_std": 0.3934583365917206, "rewards/code_complexity_reward/mean": 0.03183593600988388, "rewards/code_complexity_reward/std": 0.16732074320316315, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.017578125, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 369, "step_time": 45.550658002495766 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 511.830078125, "completions/mean_terminated_length": 468.5, "completions/min_length": 458.0, "completions/min_terminated_length": 458.0, "entropy": 0.49138691276311874, "epoch": 0.4218928164196123, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.00362313911318779, "kl": 0.0038251943042268977, "learning_rate": 3.5917507527862394e-06, "loss": 0.0004, "num_tokens": 119012620.0, "reward": 0.06279297173023224, "reward_std": 0.3699412941932678, "rewards/code_complexity_reward/mean": 0.02666015550494194, "rewards/code_complexity_reward/std": 0.15380744636058807, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.0146484375, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 370, "step_time": 45.172115268185735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.4993133544921875, "epoch": 0.42303306727480045, "frac_reward_zero_std": 0.9296875, "grad_norm": 0.0035048930440098047, "kl": 0.003855973482131958, "learning_rate": 3.5827871191505425e-06, "loss": 0.0, "num_tokens": 119333768.0, "reward": 0.0458984375, "reward_std": 0.3174210786819458, "rewards/code_complexity_reward/mean": 0.01953125, "rewards/code_complexity_reward/std": 0.13211871683597565, "rewards/code_execution_reward/mean": 0.015625, "rewards/code_execution_reward/std": 0.12414088100194931, "rewards/code_syntax_reward/mean": 0.0107421875, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 371, "step_time": 45.73260552249849 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 511.51171875, "completions/mean_terminated_length": 387.0, "completions/min_length": 361.0, "completions/min_terminated_length": 361.0, "entropy": 0.48485505720600486, "epoch": 0.4241733181299886, "frac_reward_zero_std": 0.8828125, "grad_norm": 0.003702565561980009, "kl": 0.003913741842552554, "learning_rate": 3.573806318758388e-06, "loss": 0.0013, "num_tokens": 119655538.0, "reward": 0.09433594346046448, "reward_std": 0.4476150572299957, "rewards/code_complexity_reward/mean": 0.04062499850988388, "rewards/code_complexity_reward/std": 0.1879178285598755, "rewards/code_execution_reward/mean": 0.03125, "rewards/code_execution_reward/std": 0.17416280508041382, "rewards/code_syntax_reward/mean": 0.0224609375, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 372, "step_time": 51.19568594917655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 511.693359375, "completions/mean_terminated_length": 355.0, "completions/min_length": 355.0, "completions/min_terminated_length": 355.0, "entropy": 0.4912733267992735, "epoch": 0.42531356898517675, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.0035882589872926474, "kl": 0.004056005032907706, "learning_rate": 3.5648084939934523e-06, "loss": 0.0009, "num_tokens": 119976965.0, "reward": 0.06777343899011612, "reward_std": 0.3853563666343689, "rewards/code_complexity_reward/mean": 0.02871093899011612, "rewards/code_complexity_reward/std": 0.1601915955543518, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 373, "step_time": 51.34409004636109 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 511.353515625, "completions/mean_terminated_length": 429.25, "completions/min_length": 346.0, "completions/min_terminated_length": 346.0, "entropy": 0.49183742329478264, "epoch": 0.4264538198403649, "frac_reward_zero_std": 0.890625, "grad_norm": 0.004096983931958675, "kl": 0.0039722472574794665, "learning_rate": 3.5557937875093242e-06, "loss": 0.0017, "num_tokens": 120296094.0, "reward": 0.08144531399011612, "reward_std": 0.42285194993019104, "rewards/code_complexity_reward/mean": 0.03359375149011612, "rewards/code_complexity_reward/std": 0.1715429425239563, "rewards/code_execution_reward/mean": 0.029296875, "rewards/code_execution_reward/std": 0.16880230605602264, "rewards/code_syntax_reward/mean": 0.0185546875, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 374, "step_time": 51.06279413681477 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 511.76171875, "completions/mean_terminated_length": 471.3333435058594, "completions/min_length": 460.0, "completions/min_terminated_length": 460.0, "entropy": 0.49446571012958884, "epoch": 0.427594070695553, "frac_reward_zero_std": 0.8515625, "grad_norm": 0.004645847249776125, "kl": 0.004178005103312898, "learning_rate": 3.5467623422272353e-06, "loss": 0.0005, "num_tokens": 120617276.0, "reward": 0.10615234822034836, "reward_std": 0.47883400321006775, "rewards/code_complexity_reward/mean": 0.04462890326976776, "rewards/code_complexity_reward/std": 0.19740353524684906, "rewards/code_execution_reward/mean": 0.037109375, "rewards/code_execution_reward/std": 0.18921469151973724, "rewards/code_syntax_reward/mean": 0.0244140625, "rewards/code_syntax_reward/std": 0.10785966366529465, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 375, "step_time": 46.056248736567795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 511.537109375, "completions/mean_terminated_length": 433.0, "completions/min_length": 420.0, "completions/min_terminated_length": 420.0, "entropy": 0.4852081495337188, "epoch": 0.42873432155074115, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.00376095250248909, "kl": 0.004197438476694515, "learning_rate": 3.537714301333801e-06, "loss": 0.0009, "num_tokens": 120938691.0, "reward": 0.08447265625, "reward_std": 0.41281822323799133, "rewards/code_complexity_reward/mean": 0.03955078125, "rewards/code_complexity_reward/std": 0.18709121644496918, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.021484375, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 376, "step_time": 45.47065044846386 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.5034217834472656, "epoch": 0.4298745724059293, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.003520382335409522, "kl": 0.004287853837013245, "learning_rate": 3.528649808278747e-06, "loss": 0.0, "num_tokens": 121260687.0, "reward": 0.06425781548023224, "reward_std": 0.37678712606430054, "rewards/code_complexity_reward/mean": 0.02617187425494194, "rewards/code_complexity_reward/std": 0.15105174481868744, "rewards/code_execution_reward/mean": 0.0234375, "rewards/code_execution_reward/std": 0.15143637359142303, "rewards/code_syntax_reward/mean": 0.0146484375, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 377, "step_time": 46.36215888801962 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.986328125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 511.42578125, "completions/mean_terminated_length": 470.0000305175781, "completions/min_length": 392.0, "completions/min_terminated_length": 392.0, "entropy": 0.47882730793207884, "epoch": 0.43101482326111745, "frac_reward_zero_std": 0.875, "grad_norm": 0.004219799768179655, "kl": 0.004355524550192058, "learning_rate": 3.519569006772633e-06, "loss": 0.0003, "num_tokens": 121582397.0, "reward": 0.11025390774011612, "reward_std": 0.4941510856151581, "rewards/code_complexity_reward/mean": 0.04482421651482582, "rewards/code_complexity_reward/std": 0.19820022583007812, "rewards/code_execution_reward/mean": 0.041015625, "rewards/code_execution_reward/std": 0.19852031767368317, "rewards/code_syntax_reward/mean": 0.0244140625, "rewards/code_syntax_reward/std": 0.10785966366529465, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 378, "step_time": 52.232189948670566 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.982421875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 511.44140625, "completions/mean_terminated_length": 480.22222900390625, "completions/min_length": 442.0, "completions/min_terminated_length": 442.0, "entropy": 0.48082780465483665, "epoch": 0.4321550741163056, "frac_reward_zero_std": 0.8203125, "grad_norm": 0.004948736168444157, "kl": 0.004446252380148508, "learning_rate": 3.5104720407845794e-06, "loss": 0.0013, "num_tokens": 121902523.0, "reward": 0.13671875, "reward_std": 0.5405199527740479, "rewards/code_complexity_reward/mean": 0.056640625, "rewards/code_complexity_reward/std": 0.21995562314987183, "rewards/code_execution_reward/mean": 0.048828125, "rewards/code_execution_reward/std": 0.2157193273305893, "rewards/code_syntax_reward/mean": 0.03125, "rewards/code_syntax_reward/std": 0.12114909291267395, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 379, "step_time": 45.968601278960705 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.75, "completions/mean_terminated_length": 480.0, "completions/min_length": 452.0, "completions/min_terminated_length": 452.0, "entropy": 0.49316713539883494, "epoch": 0.43329532497149376, "frac_reward_zero_std": 0.90625, "grad_norm": 0.0033314244356006384, "kl": 0.004330966272391379, "learning_rate": 3.5013590545399818e-06, "loss": 0.0006, "num_tokens": 122224051.0, "reward": 0.06953125447034836, "reward_std": 0.3931399881839752, "rewards/code_complexity_reward/mean": 0.02851562388241291, "rewards/code_complexity_reward/std": 0.1591235250234604, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 380, "step_time": 46.12850445136428 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 511.76953125, "completions/mean_terminated_length": 472.66668701171875, "completions/min_length": 435.0, "completions/min_terminated_length": 435.0, "entropy": 0.48331937240436673, "epoch": 0.43443557582668185, "frac_reward_zero_std": 0.921875, "grad_norm": 0.003214549506083131, "kl": 0.0046294073363242205, "learning_rate": 3.492230192518221e-06, "loss": 0.0005, "num_tokens": 122544813.0, "reward": 0.07148437947034836, "reward_std": 0.40261784195899963, "rewards/code_complexity_reward/mean": 0.02851562388241291, "rewards/code_complexity_reward/std": 0.1590927690267563, "rewards/code_execution_reward/mean": 0.02734375, "rewards/code_execution_reward/std": 0.16324250400066376, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 381, "step_time": 46.20273306686431 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.986328125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 510.720703125, "completions/mean_terminated_length": 418.4285888671875, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "entropy": 0.49303552228957415, "epoch": 0.43557582668187, "frac_reward_zero_std": 0.84375, "grad_norm": 0.004884173162281513, "kl": 0.004749439482111484, "learning_rate": 3.483085599450381e-06, "loss": 0.0023, "num_tokens": 122865450.0, "reward": 0.13261720538139343, "reward_std": 0.545476496219635, "rewards/code_complexity_reward/mean": 0.05156249925494194, "rewards/code_complexity_reward/std": 0.21094675362110138, "rewards/code_execution_reward/mean": 0.052734375, "rewards/code_execution_reward/std": 0.22372129559516907, "rewards/code_syntax_reward/mean": 0.0283203125, "rewards/code_syntax_reward/std": 0.11569035053253174, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 382, "step_time": 52.07351534254849 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 511.693359375, "completions/mean_terminated_length": 480.6000061035156, "completions/min_length": 468.0, "completions/min_terminated_length": 468.0, "entropy": 0.4691150705330074, "epoch": 0.43671607753705816, "frac_reward_zero_std": 0.78125, "grad_norm": 0.0061683859676122665, "kl": 0.004650669921829831, "learning_rate": 3.473925420316946e-06, "loss": 0.0007, "num_tokens": 123186237.0, "reward": 0.146484375, "reward_std": 0.5426139235496521, "rewards/code_complexity_reward/mean": 0.0634765625, "rewards/code_complexity_reward/std": 0.22869926691055298, "rewards/code_execution_reward/mean": 0.046875, "rewards/code_execution_reward/std": 0.21157780289649963, "rewards/code_syntax_reward/mean": 0.0361328125, "rewards/code_syntax_reward/std": 0.1295902281999588, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 383, "step_time": 45.90304858889431 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.41796875, "completions/mean_terminated_length": 437.5, "completions/min_length": 359.0, "completions/min_terminated_length": 359.0, "entropy": 0.4965074909850955, "epoch": 0.4378563283922463, "frac_reward_zero_std": 0.875, "grad_norm": 0.004008037503808737, "kl": 0.004589801850670483, "learning_rate": 3.464749800345507e-06, "loss": 0.0015, "num_tokens": 123508595.0, "reward": 0.07246093451976776, "reward_std": 0.38299646973609924, "rewards/code_complexity_reward/mean": 0.03242187201976776, "rewards/code_complexity_reward/std": 0.16585659980773926, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.0185546875, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 384, "step_time": 46.20693675521761 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.998046875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 511.994140625, "completions/mean_terminated_length": 509.0, "completions/min_length": 509.0, "completions/min_terminated_length": 509.0, "entropy": 0.46736594662070274, "epoch": 0.43899657924743446, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.003529242007061839, "kl": 0.004535905856755562, "learning_rate": 3.4555588850084575e-06, "loss": 0.0, "num_tokens": 123830808.0, "reward": 0.07500000298023224, "reward_std": 0.40159907937049866, "rewards/code_complexity_reward/mean": 0.03203125298023224, "rewards/code_complexity_reward/std": 0.16827481985092163, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.017578125, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 385, "step_time": 46.11921405419707 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.986328125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 511.2109375, "completions/mean_terminated_length": 454.2857360839844, "completions/min_length": 396.0, "completions/min_terminated_length": 396.0, "entropy": 0.47847710410133004, "epoch": 0.44013683010262256, "frac_reward_zero_std": 0.8203125, "grad_norm": 0.0051162405870854855, "kl": 0.004902718890662072, "learning_rate": 3.4463528200206868e-06, "loss": 0.0016, "num_tokens": 124152568.0, "reward": 0.14814454317092896, "reward_std": 0.5597748160362244, "rewards/code_complexity_reward/mean": 0.06318359076976776, "rewards/code_complexity_reward/std": 0.23372048139572144, "rewards/code_execution_reward/mean": 0.05078125, "rewards/code_execution_reward/std": 0.21976542472839355, "rewards/code_syntax_reward/mean": 0.0341796875, "rewards/code_syntax_reward/std": 0.12630419433116913, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 386, "step_time": 51.93024830054492 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 511.44921875, "completions/mean_terminated_length": 455.6000061035156, "completions/min_length": 390.0, "completions/min_terminated_length": 390.0, "entropy": 0.4818765218369663, "epoch": 0.4412770809578107, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.00403361301869154, "kl": 0.004840993842663011, "learning_rate": 3.4371317513372692e-06, "loss": 0.0013, "num_tokens": 124474910.0, "reward": 0.06406249850988388, "reward_std": 0.3666069805622101, "rewards/code_complexity_reward/mean": 0.02695312350988388, "rewards/code_complexity_reward/std": 0.1513672024011612, "rewards/code_execution_reward/mean": 0.021484375, "rewards/code_execution_reward/std": 0.14513419568538666, "rewards/code_syntax_reward/mean": 0.015625, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 387, "step_time": 51.02850620355457 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 511.552734375, "completions/mean_terminated_length": 466.20001220703125, "completions/min_length": 414.0, "completions/min_terminated_length": 414.0, "entropy": 0.4811326125636697, "epoch": 0.44241733181299886, "frac_reward_zero_std": 0.859375, "grad_norm": 0.0043686251156032085, "kl": 0.0051551924625528045, "learning_rate": 3.427895825151153e-06, "loss": 0.001, "num_tokens": 124794605.0, "reward": 0.09208984673023224, "reward_std": 0.4374147057533264, "rewards/code_complexity_reward/mean": 0.04033203050494194, "rewards/code_complexity_reward/std": 0.18653102219104767, "rewards/code_execution_reward/mean": 0.029296875, "rewards/code_execution_reward/std": 0.16880230605602264, "rewards/code_syntax_reward/mean": 0.0224609375, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 388, "step_time": 51.765123010613024 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.982421875, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 510.806640625, "completions/mean_terminated_length": 444.1111145019531, "completions/min_length": 330.0, "completions/min_terminated_length": 330.0, "entropy": 0.49281404819339514, "epoch": 0.443557582668187, "frac_reward_zero_std": 0.8515625, "grad_norm": 0.004889100790023804, "kl": 0.005151681893039495, "learning_rate": 3.4186451878908393e-06, "loss": 0.0018, "num_tokens": 125116422.0, "reward": 0.12558594346046448, "reward_std": 0.5155884027481079, "rewards/code_complexity_reward/mean": 0.05332031100988388, "rewards/code_complexity_reward/std": 0.2142632007598877, "rewards/code_execution_reward/mean": 0.04296875, "rewards/code_execution_reward/std": 0.2029850035905838, "rewards/code_syntax_reward/mean": 0.029296875, "rewards/code_syntax_reward/std": 0.11754623055458069, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 389, "step_time": 50.70056823361665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.974609375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 510.119140625, "completions/mean_terminated_length": 437.923095703125, "completions/min_length": 343.0, "completions/min_terminated_length": 343.0, "entropy": 0.482000146061182, "epoch": 0.44469783352337516, "frac_reward_zero_std": 0.8125, "grad_norm": 0.005151406861841679, "kl": 0.005152739897312131, "learning_rate": 3.4093799862180627e-06, "loss": 0.0019, "num_tokens": 125437307.0, "reward": 0.18662109971046448, "reward_std": 0.6330908536911011, "rewards/code_complexity_reward/mean": 0.07333984225988388, "rewards/code_complexity_reward/std": 0.24645665287971497, "rewards/code_execution_reward/mean": 0.072265625, "rewards/code_execution_reward/std": 0.2591804563999176, "rewards/code_syntax_reward/mean": 0.041015625, "rewards/code_syntax_reward/std": 0.13734035193920135, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 390, "step_time": 46.33625808078796 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.99609375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.935546875, "completions/mean_terminated_length": 495.5, "completions/min_length": 480.0, "completions/min_terminated_length": 480.0, "entropy": 0.4797836276702583, "epoch": 0.44583808437856326, "frac_reward_zero_std": 0.9140625, "grad_norm": 0.0034395332913845778, "kl": 0.005105297903355677, "learning_rate": 3.4001003670254656e-06, "loss": 0.0001, "num_tokens": 125761018.0, "reward": 0.06865234673023224, "reward_std": 0.39802441000938416, "rewards/code_complexity_reward/mean": 0.02666015550494194, "rewards/code_complexity_reward/std": 0.1539028435945511, "rewards/code_execution_reward/mean": 0.02734375, "rewards/code_execution_reward/std": 0.16324250400066376, "rewards/code_syntax_reward/mean": 0.0146484375, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 391, "step_time": 50.46972591429949 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.98828125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 511.7109375, "completions/mean_terminated_length": 487.3333435058594, "completions/min_length": 450.0, "completions/min_terminated_length": 450.0, "entropy": 0.4772152625955641, "epoch": 0.4469783352337514, "frac_reward_zero_std": 0.8125, "grad_norm": 0.005414540413767099, "kl": 0.005617643990262877, "learning_rate": 3.390806477434269e-06, "loss": 0.0006, "num_tokens": 126081662.0, "reward": 0.13886719942092896, "reward_std": 0.5411836504936218, "rewards/code_complexity_reward/mean": 0.05781249701976776, "rewards/code_complexity_reward/std": 0.2210708111524582, "rewards/code_execution_reward/mean": 0.048828125, "rewards/code_execution_reward/std": 0.2157193273305893, "rewards/code_syntax_reward/mean": 0.0322265625, "rewards/code_syntax_reward/std": 0.12289927154779434, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 392, "step_time": 45.647214385680854 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 511.318359375, "completions/mean_terminated_length": 442.20001220703125, "completions/min_length": 390.0, "completions/min_terminated_length": 390.0, "entropy": 0.4818262089975178, "epoch": 0.44811858608893956, "frac_reward_zero_std": 0.8125, "grad_norm": 0.005655732937157154, "kl": 0.005394386702391785, "learning_rate": 3.381498464791939e-06, "loss": 0.001, "num_tokens": 126403537.0, "reward": 0.12617187201976776, "reward_std": 0.5054097175598145, "rewards/code_complexity_reward/mean": 0.05585937574505806, "rewards/code_complexity_reward/std": 0.21754015982151031, "rewards/code_execution_reward/mean": 0.0390625, "rewards/code_execution_reward/std": 0.1939331740140915, "rewards/code_syntax_reward/mean": 0.03125, "rewards/code_syntax_reward/std": 0.12114909291267395, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 393, "step_time": 46.15808784030378 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.58984375, "completions/mean_terminated_length": 470.0, "completions/min_length": 400.0, "completions/min_terminated_length": 400.0, "entropy": 0.4792354665696621, "epoch": 0.4492588369441277, "frac_reward_zero_std": 0.875, "grad_norm": 0.004282795824110508, "kl": 0.00537480870843865, "learning_rate": 3.372176476669853e-06, "loss": 0.0006, "num_tokens": 126724959.0, "reward": 0.12001953274011612, "reward_std": 0.5159250497817993, "rewards/code_complexity_reward/mean": 0.04873046651482582, "rewards/code_complexity_reward/std": 0.20690937340259552, "rewards/code_execution_reward/mean": 0.044921875, "rewards/code_execution_reward/std": 0.20733514428138733, "rewards/code_syntax_reward/mean": 0.0263671875, "rewards/code_syntax_reward/std": 0.11186064779758453, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 394, "step_time": 49.920227566733956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 511.6484375, "completions/mean_terminated_length": 452.0, "completions/min_length": 414.0, "completions/min_terminated_length": 414.0, "entropy": 0.47687921300530434, "epoch": 0.45039908779931587, "frac_reward_zero_std": 0.8203125, "grad_norm": 0.005528281908482313, "kl": 0.005316565962857567, "learning_rate": 3.362840660860958e-06, "loss": 0.0003, "num_tokens": 127046587.0, "reward": 0.15253905951976776, "reward_std": 0.5608155727386475, "rewards/code_complexity_reward/mean": 0.06562500447034836, "rewards/code_complexity_reward/std": 0.23564457893371582, "rewards/code_execution_reward/mean": 0.05078125, "rewards/code_execution_reward/std": 0.21976542472839355, "rewards/code_syntax_reward/mean": 0.0361328125, "rewards/code_syntax_reward/std": 0.1295902281999588, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 395, "step_time": 45.50184595398605 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.982421875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 511.09765625, "completions/mean_terminated_length": 460.6666564941406, "completions/min_length": 385.0, "completions/min_terminated_length": 385.0, "entropy": 0.4703517151065171, "epoch": 0.45153933865450396, "frac_reward_zero_std": 0.8203125, "grad_norm": 0.005805360618978739, "kl": 0.005621872725896537, "learning_rate": 3.353491165377429e-06, "loss": 0.002, "num_tokens": 127366601.0, "reward": 0.13056640326976776, "reward_std": 0.5243561863899231, "rewards/code_complexity_reward/mean": 0.05341796576976776, "rewards/code_complexity_reward/std": 0.21107551455497742, "rewards/code_execution_reward/mean": 0.046875, "rewards/code_execution_reward/std": 0.21157780289649963, "rewards/code_syntax_reward/mean": 0.0302734375, "rewards/code_syntax_reward/std": 0.11936526000499725, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 396, "step_time": 46.01817754749209 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.986328125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 511.302734375, "completions/mean_terminated_length": 461.0000305175781, "completions/min_length": 421.0, "completions/min_terminated_length": 421.0, "entropy": 0.47809912264347076, "epoch": 0.4526795895096921, "frac_reward_zero_std": 0.859375, "grad_norm": 0.005559713114053011, "kl": 0.00578247075463878, "learning_rate": 3.3441281384483215e-06, "loss": 0.0014, "num_tokens": 127687620.0, "reward": 0.10771484673023224, "reward_std": 0.4779420793056488, "rewards/code_complexity_reward/mean": 0.04521484673023224, "rewards/code_complexity_reward/std": 0.19620057940483093, "rewards/code_execution_reward/mean": 0.037109375, "rewards/code_execution_reward/std": 0.18921469151973724, "rewards/code_syntax_reward/mean": 0.025390625, "rewards/code_syntax_reward/std": 0.10988271236419678, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 397, "step_time": 51.191483955830336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.986328125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 511.3984375, "completions/mean_terminated_length": 468.0000305175781, "completions/min_length": 377.0, "completions/min_terminated_length": 377.0, "entropy": 0.4777282397262752, "epoch": 0.45381984036488027, "frac_reward_zero_std": 0.890625, "grad_norm": 0.003831795882433653, "kl": 0.005586536608461756, "learning_rate": 3.3347517285172225e-06, "loss": 0.0007, "num_tokens": 128009944.0, "reward": 0.10400390625, "reward_std": 0.4785911440849304, "rewards/code_complexity_reward/mean": 0.04150390625, "rewards/code_complexity_reward/std": 0.18794669210910797, "rewards/code_execution_reward/mean": 0.0390625, "rewards/code_execution_reward/std": 0.1939331740140915, "rewards/code_syntax_reward/mean": 0.0234375, "rewards/code_syntax_reward/std": 0.10578890144824982, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 398, "step_time": 45.3690773146227 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 511.431640625, "completions/mean_terminated_length": 453.8000183105469, "completions/min_length": 411.0, "completions/min_terminated_length": 411.0, "entropy": 0.48249871330335736, "epoch": 0.4549600912200684, "frac_reward_zero_std": 0.8984375, "grad_norm": 0.003827685723081231, "kl": 0.005866037419764325, "learning_rate": 3.325362084239894e-06, "loss": 0.0009, "num_tokens": 128330809.0, "reward": 0.07871093600988388, "reward_std": 0.4106292128562927, "rewards/code_complexity_reward/mean": 0.03378906100988388, "rewards/code_complexity_reward/std": 0.17275509238243103, "rewards/code_execution_reward/mean": 0.025390625, "rewards/code_execution_reward/std": 0.15746226906776428, "rewards/code_syntax_reward/mean": 0.0185546875, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 399, "step_time": 45.501884549856186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.986328125, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 511.248046875, "completions/mean_terminated_length": 457.0000305175781, "completions/min_length": 374.0, "completions/min_terminated_length": 374.0, "entropy": 0.4745196672156453, "epoch": 0.45610034207525657, "frac_reward_zero_std": 0.8125, "grad_norm": 0.00559978187084198, "kl": 0.006157652824185789, "learning_rate": 3.31595935448192e-06, "loss": 0.0014, "num_tokens": 128652640.0, "reward": 0.166015625, "reward_std": 0.5975630283355713, "rewards/code_complexity_reward/mean": 0.06640625, "rewards/code_complexity_reward/std": 0.23584043979644775, "rewards/code_execution_reward/mean": 0.0625, "rewards/code_execution_reward/std": 0.2422981858253479, "rewards/code_syntax_reward/mean": 0.037109375, "rewards/code_syntax_reward/std": 0.1311914473772049, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 400, "step_time": 50.18423006590456 }, { "epoch": 0.45610034207525657, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.9925, "eval_completions/max_length": 512.0, "eval_completions/max_terminated_length": 18.66, "eval_completions/mean_length": 511.5625, "eval_completions/mean_terminated_length": 18.62, "eval_completions/min_length": 510.1, "eval_completions/min_terminated_length": 18.58, "eval_entropy": 0.49078130841255185, "eval_frac_reward_zero_std": 0.86, "eval_kl": 0.006211526673287154, "eval_loss": 0.00037225012783892453, "eval_num_tokens": 128652640.0, "eval_reward": 0.11175000160932541, "eval_reward_std": 0.20864929974079133, "eval_rewards/code_complexity_reward/mean": 0.049249999374151227, "eval_rewards/code_complexity_reward/std": 0.09406842887401581, "eval_rewards/code_execution_reward/mean": 0.035, "eval_rewards/code_execution_reward/std": 0.06674932360649109, "eval_rewards/code_syntax_reward/mean": 0.0275, "eval_rewards/code_syntax_reward/std": 0.0525225567817688, "eval_rewards/reasoning_present_reward_func/mean": 0.0, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.0, "eval_rewards/xmlcount_reward_func/std": 0.0, "eval_runtime": 1279.5107, "eval_samples_per_second": 0.078, "eval_steps_per_second": 0.01, "step": 400 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.98046875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.0390625, "completions/mean_terminated_length": 462.8000183105469, "completions/min_length": 323.0, "completions/min_terminated_length": 323.0, "entropy": 0.46992665994912386, "epoch": 0.4572405929304447, "frac_reward_zero_std": 0.84375, "grad_norm": 0.0055186813697218895, "kl": 0.006003365495416801, "learning_rate": 3.3065436883163453e-06, "loss": 0.0018, "num_tokens": 128975312.0, "reward": 0.13662110269069672, "reward_std": 0.5467227697372437, "rewards/code_complexity_reward/mean": 0.05361328274011612, "rewards/code_complexity_reward/std": 0.21227411925792694, "rewards/code_execution_reward/mean": 0.052734375, "rewards/code_execution_reward/std": 0.22372129559516907, "rewards/code_syntax_reward/mean": 0.0302734375, "rewards/code_syntax_reward/std": 0.11936526000499725, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 401, "step_time": 50.652469797991216 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.982421875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.09765625, "completions/mean_terminated_length": 460.6666564941406, "completions/min_length": 310.0, "completions/min_terminated_length": 310.0, "entropy": 0.4635886922478676, "epoch": 0.4583808437856328, "frac_reward_zero_std": 0.8046875, "grad_norm": 0.005050742533057928, "kl": 0.0062971859515528195, "learning_rate": 3.2971152350213106e-06, "loss": 0.0017, "num_tokens": 129297306.0, "reward": 0.16933594644069672, "reward_std": 0.5892440676689148, "rewards/code_complexity_reward/mean": 0.07265624403953552, "rewards/code_complexity_reward/std": 0.24730321764945984, "rewards/code_execution_reward/mean": 0.056640625, "rewards/code_execution_reward/std": 0.23138070106506348, "rewards/code_syntax_reward/mean": 0.0400390625, "rewards/code_syntax_reward/std": 0.1358397752046585, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 402, "step_time": 46.72604627162218 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.984375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 510.626953125, "completions/mean_terminated_length": 424.125, "completions/min_length": 303.0, "completions/min_terminated_length": 303.0, "entropy": 0.4759185118600726, "epoch": 0.45952109464082097, "frac_reward_zero_std": 0.828125, "grad_norm": 0.0048680780455470085, "kl": 0.00609477956459159, "learning_rate": 3.2876741440776853e-06, "loss": 0.0026, "num_tokens": 129619339.0, "reward": 0.15693360567092896, "reward_std": 0.5809733867645264, "rewards/code_complexity_reward/mean": 0.06318359076976776, "rewards/code_complexity_reward/std": 0.2306865006685257, "rewards/code_execution_reward/mean": 0.05859375, "rewards/code_execution_reward/std": 0.23509246110916138, "rewards/code_syntax_reward/mean": 0.03515625, "rewards/code_syntax_reward/std": 0.12796148657798767, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 403, "step_time": 51.91080915927887 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.98828125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.01171875, "completions/mean_terminated_length": 427.66668701171875, "completions/min_length": 317.0, "completions/min_terminated_length": 317.0, "entropy": 0.4620140683837235, "epoch": 0.4606613454960091, "frac_reward_zero_std": 0.8203125, "grad_norm": 0.005404964089393616, "kl": 0.006502075368189253, "learning_rate": 3.2782205651667013e-06, "loss": 0.0015, "num_tokens": 129939685.0, "reward": 0.15878906846046448, "reward_std": 0.5748741626739502, "rewards/code_complexity_reward/mean": 0.06699218600988388, "rewards/code_complexity_reward/std": 0.23734988272190094, "rewards/code_execution_reward/mean": 0.0546875, "rewards/code_execution_reward/std": 0.2275916188955307, "rewards/code_syntax_reward/mean": 0.037109375, "rewards/code_syntax_reward/std": 0.1311914473772049, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 404, "step_time": 45.59111980441958 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.96875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 510.248046875, "completions/mean_terminated_length": 455.9375, "completions/min_length": 304.0, "completions/min_terminated_length": 304.0, "entropy": 0.4779546386562288, "epoch": 0.4618015963511973, "frac_reward_zero_std": 0.796875, "grad_norm": 0.006730588153004646, "kl": 0.006412348935555201, "learning_rate": 3.2687546481675776e-06, "loss": 0.0022, "num_tokens": 130260728.0, "reward": 0.21406249701976776, "reward_std": 0.6598994135856628, "rewards/code_complexity_reward/mean": 0.09003905951976776, "rewards/code_complexity_reward/std": 0.2715620994567871, "rewards/code_execution_reward/mean": 0.07421875, "rewards/code_execution_reward/std": 0.2623828947544098, "rewards/code_syntax_reward/mean": 0.0498046875, "rewards/code_syntax_reward/std": 0.14988566935062408, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 405, "step_time": 46.286561974324286 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 511.58984375, "completions/mean_terminated_length": 470.0, "completions/min_length": 386.0, "completions/min_terminated_length": 386.0, "entropy": 0.4850928043015301, "epoch": 0.4629418472063854, "frac_reward_zero_std": 0.8359375, "grad_norm": 0.004923602100461721, "kl": 0.006509516293590423, "learning_rate": 3.259276543155142e-06, "loss": 0.0007, "num_tokens": 130582206.0, "reward": 0.13095703721046448, "reward_std": 0.5327380895614624, "rewards/code_complexity_reward/mean": 0.05283202975988388, "rewards/code_complexity_reward/std": 0.21228601038455963, "rewards/code_execution_reward/mean": 0.048828125, "rewards/code_execution_reward/std": 0.2157193273305893, "rewards/code_syntax_reward/mean": 0.029296875, "rewards/code_syntax_reward/std": 0.11754623055458069, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 406, "step_time": 45.81839035823941 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.986328125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 511.380859375, "completions/mean_terminated_length": 466.71429443359375, "completions/min_length": 396.0, "completions/min_terminated_length": 396.0, "entropy": 0.4731754148378968, "epoch": 0.4640820980615735, "frac_reward_zero_std": 0.84375, "grad_norm": 0.004704641178250313, "kl": 0.006768440187443048, "learning_rate": 3.2497864003974554e-06, "loss": 0.0009, "num_tokens": 130903997.0, "reward": 0.12832032144069672, "reward_std": 0.5188319683074951, "rewards/code_complexity_reward/mean": 0.05507812276482582, "rewards/code_complexity_reward/std": 0.21753734350204468, "rewards/code_execution_reward/mean": 0.04296875, "rewards/code_execution_reward/std": 0.2029850035905838, "rewards/code_syntax_reward/mean": 0.0302734375, "rewards/code_syntax_reward/std": 0.11936526000499725, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 407, "step_time": 51.932231777347624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.990234375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.7265625, "completions/mean_terminated_length": 484.0, "completions/min_length": 425.0, "completions/min_terminated_length": 425.0, "entropy": 0.477367153391242, "epoch": 0.4652223489167617, "frac_reward_zero_std": 0.84375, "grad_norm": 0.004900538828223944, "kl": 0.006941999243281316, "learning_rate": 3.2402843703534283e-06, "loss": 0.0005, "num_tokens": 131224453.0, "reward": 0.13154296576976776, "reward_std": 0.5296180248260498, "rewards/code_complexity_reward/mean": 0.05634765326976776, "rewards/code_complexity_reward/std": 0.2223643958568573, "rewards/code_execution_reward/mean": 0.044921875, "rewards/code_execution_reward/std": 0.20733514428138733, "rewards/code_syntax_reward/mean": 0.0302734375, "rewards/code_syntax_reward/std": 0.11936526000499725, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 408, "step_time": 51.80936315096915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.513671875, "completions/mean_terminated_length": 480.875, "completions/min_length": 428.0, "completions/min_terminated_length": 428.0, "entropy": 0.4752868344075978, "epoch": 0.4663625997719498, "frac_reward_zero_std": 0.828125, "grad_norm": 0.0052878158167004585, "kl": 0.006953540760150645, "learning_rate": 3.2307706036704328e-06, "loss": 0.0008, "num_tokens": 131545672.0, "reward": 0.14384765923023224, "reward_std": 0.5511118769645691, "rewards/code_complexity_reward/mean": 0.05986328050494194, "rewards/code_complexity_reward/std": 0.22518828511238098, "rewards/code_execution_reward/mean": 0.05078125, "rewards/code_execution_reward/std": 0.21976542472839355, "rewards/code_syntax_reward/mean": 0.033203125, "rewards/code_syntax_reward/std": 0.12461719661951065, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 409, "step_time": 48.38353524077684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.98046875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.18359375, "completions/mean_terminated_length": 470.20001220703125, "completions/min_length": 434.0, "completions/min_terminated_length": 434.0, "entropy": 0.46940926369279623, "epoch": 0.467502850627138, "frac_reward_zero_std": 0.7890625, "grad_norm": 0.006438682787120342, "kl": 0.006760244526958559, "learning_rate": 3.221245251181919e-06, "loss": 0.0015, "num_tokens": 131867366.0, "reward": 0.16357421875, "reward_std": 0.5722602009773254, "rewards/code_complexity_reward/mean": 0.07080078125, "rewards/code_complexity_reward/std": 0.24139179289340973, "rewards/code_execution_reward/mean": 0.052734375, "rewards/code_execution_reward/std": 0.22372129559516907, "rewards/code_syntax_reward/mean": 0.0400390625, "rewards/code_syntax_reward/std": 0.1358397752046585, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 410, "step_time": 45.80789540056139 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.98046875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 510.34375, "completions/mean_terminated_length": 427.20001220703125, "completions/min_length": 254.0, "completions/min_terminated_length": 254.0, "entropy": 0.46462062234058976, "epoch": 0.46864310148232613, "frac_reward_zero_std": 0.8203125, "grad_norm": 0.00497846445068717, "kl": 0.006946895104192663, "learning_rate": 3.2117084639050204e-06, "loss": 0.0026, "num_tokens": 132190350.0, "reward": 0.14736327528953552, "reward_std": 0.5570293664932251, "rewards/code_complexity_reward/mean": 0.06240234151482582, "rewards/code_complexity_reward/std": 0.23104773461818695, "rewards/code_execution_reward/mean": 0.05078125, "rewards/code_execution_reward/std": 0.21976542472839355, "rewards/code_syntax_reward/mean": 0.0341796875, "rewards/code_syntax_reward/std": 0.12630419433116913, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 411, "step_time": 45.64587211050093 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.986328125, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 510.9375, "completions/mean_terminated_length": 434.2857360839844, "completions/min_length": 391.0, "completions/min_terminated_length": 391.0, "entropy": 0.4775140997953713, "epoch": 0.4697833523375142, "frac_reward_zero_std": 0.828125, "grad_norm": 0.005292811430990696, "kl": 0.007010404937318526, "learning_rate": 3.2021603930381582e-06, "loss": 0.0021, "num_tokens": 132511078.0, "reward": 0.13461914658546448, "reward_std": 0.5324720144271851, "rewards/code_complexity_reward/mean": 0.05624999850988388, "rewards/code_complexity_reward/std": 0.21862852573394775, "rewards/code_execution_reward/mean": 0.046875, "rewards/code_execution_reward/std": 0.21157780289649963, "rewards/code_syntax_reward/mean": 0.03125, "rewards/code_syntax_reward/std": 0.12114909291267395, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 412, "step_time": 46.109389633871615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.962890625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 509.521484375, "completions/mean_terminated_length": 445.2105407714844, "completions/min_length": 306.0, "completions/min_terminated_length": 306.0, "entropy": 0.4783353665843606, "epoch": 0.4709236031927024, "frac_reward_zero_std": 0.796875, "grad_norm": 0.005917046684771776, "kl": 0.007284330065886024, "learning_rate": 3.1926011899586483e-06, "loss": 0.0026, "num_tokens": 132832285.0, "reward": 0.19365233182907104, "reward_std": 0.636391282081604, "rewards/code_complexity_reward/mean": 0.07939453423023224, "rewards/code_complexity_reward/std": 0.2565201222896576, "rewards/code_execution_reward/mean": 0.0703125, "rewards/code_execution_reward/std": 0.25592297315597534, "rewards/code_syntax_reward/mean": 0.0439453125, "rewards/code_syntax_reward/std": 0.14170633256435394, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 413, "step_time": 51.19445828627795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9765625, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 510.58203125, "completions/mean_terminated_length": 451.5, "completions/min_length": 361.0, "completions/min_terminated_length": 361.0, "entropy": 0.4731077295728028, "epoch": 0.47206385404789053, "frac_reward_zero_std": 0.8046875, "grad_norm": 0.005852977745234966, "kl": 0.007609685890201945, "learning_rate": 3.1830310062202996e-06, "loss": 0.0016, "num_tokens": 133152555.0, "reward": 0.18828125298023224, "reward_std": 0.6326252222061157, "rewards/code_complexity_reward/mean": 0.07597656548023224, "rewards/code_complexity_reward/std": 0.2516268491744995, "rewards/code_execution_reward/mean": 0.0703125, "rewards/code_execution_reward/std": 0.25592297315597534, "rewards/code_syntax_reward/mean": 0.0419921875, "rewards/code_syntax_reward/std": 0.13881781697273254, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 414, "step_time": 45.65585400070995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.994140625, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 511.640625, "completions/mean_terminated_length": 450.66668701171875, "completions/min_length": 374.0, "completions/min_terminated_length": 374.0, "entropy": 0.48085783794522285, "epoch": 0.4732041049030787, "frac_reward_zero_std": 0.828125, "grad_norm": 0.005343542899936438, "kl": 0.0074703907303046435, "learning_rate": 3.1734499935510093e-06, "loss": 0.0004, "num_tokens": 133472311.0, "reward": 0.12978515028953552, "reward_std": 0.5125685930252075, "rewards/code_complexity_reward/mean": 0.05458984151482582, "rewards/code_complexity_reward/std": 0.20961110293865204, "rewards/code_execution_reward/mean": 0.04296875, "rewards/code_execution_reward/std": 0.2029850035905838, "rewards/code_syntax_reward/mean": 0.0322265625, "rewards/code_syntax_reward/std": 0.12289927154779434, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 415, "step_time": 45.70363750029355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.966796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 509.626953125, "completions/mean_terminated_length": 440.5294189453125, "completions/min_length": 238.0, "completions/min_terminated_length": 238.0, "entropy": 0.48035072814673185, "epoch": 0.47434435575826683, "frac_reward_zero_std": 0.7265625, "grad_norm": 0.007948048412799835, "kl": 0.007796284444339108, "learning_rate": 3.1638583038503596e-06, "loss": 0.002, "num_tokens": 133792400.0, "reward": 0.2744140923023224, "reward_std": 0.7371333837509155, "rewards/code_complexity_reward/mean": 0.11328125, "rewards/code_complexity_reward/std": 0.29805201292037964, "rewards/code_execution_reward/mean": 0.09765625, "rewards/code_execution_reward/std": 0.29713961482048035, "rewards/code_syntax_reward/mean": 0.0634765625, "rewards/code_syntax_reward/std": 0.16662302613258362, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 416, "step_time": 45.821383230388165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.96875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 510.48828125, "completions/mean_terminated_length": 463.625, "completions/min_length": 329.0, "completions/min_terminated_length": 329.0, "entropy": 0.46998064406216145, "epoch": 0.475484606613455, "frac_reward_zero_std": 0.7734375, "grad_norm": 0.0058990889228880405, "kl": 0.007894173118984327, "learning_rate": 3.15425608918721e-06, "loss": 0.0024, "num_tokens": 134113182.0, "reward": 0.20478516817092896, "reward_std": 0.6566694974899292, "rewards/code_complexity_reward/mean": 0.08466796576976776, "rewards/code_complexity_reward/std": 0.2670859694480896, "rewards/code_execution_reward/mean": 0.07421875, "rewards/code_execution_reward/std": 0.2623828947544098, "rewards/code_syntax_reward/mean": 0.0458984375, "rewards/code_syntax_reward/std": 0.1445106863975525, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 417, "step_time": 45.713564831763506 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9765625, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 510.515625, "completions/mean_terminated_length": 448.66668701171875, "completions/min_length": 376.0, "completions/min_terminated_length": 376.0, "entropy": 0.47721854969859123, "epoch": 0.4766248574686431, "frac_reward_zero_std": 0.796875, "grad_norm": 0.006517315749078989, "kl": 0.007835793214326259, "learning_rate": 3.144643501797282e-06, "loss": 0.0029, "num_tokens": 134434150.0, "reward": 0.18896484375, "reward_std": 0.6235200762748718, "rewards/code_complexity_reward/mean": 0.07861328125, "rewards/code_complexity_reward/std": 0.2540602385997772, "rewards/code_execution_reward/mean": 0.06640625, "rewards/code_execution_reward/std": 0.2492343932390213, "rewards/code_syntax_reward/mean": 0.0439453125, "rewards/code_syntax_reward/std": 0.14170633256435394, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 418, "step_time": 45.22045713290572 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.986328125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 510.9296875, "completions/mean_terminated_length": 433.71429443359375, "completions/min_length": 273.0, "completions/min_terminated_length": 273.0, "entropy": 0.4666978269815445, "epoch": 0.47776510832383123, "frac_reward_zero_std": 0.8203125, "grad_norm": 0.005564184859395027, "kl": 0.008089927679975517, "learning_rate": 3.1350206940807523e-06, "loss": 0.0009, "num_tokens": 134756270.0, "reward": 0.17060548067092896, "reward_std": 0.592888593673706, "rewards/code_complexity_reward/mean": 0.07197265326976776, "rewards/code_complexity_reward/std": 0.24461042881011963, "rewards/code_execution_reward/mean": 0.05859375, "rewards/code_execution_reward/std": 0.23509246110916138, "rewards/code_syntax_reward/mean": 0.0400390625, "rewards/code_syntax_reward/std": 0.1358397752046585, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 419, "step_time": 45.44775966927409 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.978515625, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 510.744140625, "completions/mean_terminated_length": 453.54547119140625, "completions/min_length": 369.0, "completions/min_terminated_length": 369.0, "entropy": 0.46859703632071614, "epoch": 0.4789053591790194, "frac_reward_zero_std": 0.75, "grad_norm": 0.0071568978019058704, "kl": 0.008305964467581362, "learning_rate": 3.125387818599831e-06, "loss": 0.0027, "num_tokens": 135076079.0, "reward": 0.19462889432907104, "reward_std": 0.6212286949157715, "rewards/code_complexity_reward/mean": 0.08330078423023224, "rewards/code_complexity_reward/std": 0.26021113991737366, "rewards/code_execution_reward/mean": 0.064453125, "rewards/code_execution_reward/std": 0.24579854309558868, "rewards/code_syntax_reward/mean": 0.046875, "rewards/code_syntax_reward/std": 0.14588283002376556, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 420, "step_time": 46.597876532934606 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.986328125, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 510.794921875, "completions/mean_terminated_length": 423.8571472167969, "completions/min_length": 343.0, "completions/min_terminated_length": 343.0, "entropy": 0.4710284383036196, "epoch": 0.48004561003420754, "frac_reward_zero_std": 0.8125, "grad_norm": 0.005644808057695627, "kl": 0.008847179269650951, "learning_rate": 3.1157450280763464e-06, "loss": 0.0016, "num_tokens": 135396902.0, "reward": 0.15371093153953552, "reward_std": 0.5599097013473511, "rewards/code_complexity_reward/mean": 0.06777343899011612, "rewards/code_complexity_reward/std": 0.24044667184352875, "rewards/code_execution_reward/mean": 0.048828125, "rewards/code_execution_reward/std": 0.2157193273305893, "rewards/code_syntax_reward/mean": 0.037109375, "rewards/code_syntax_reward/std": 0.1311914473772049, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 421, "step_time": 51.90370053239167 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.98046875, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 510.982421875, "completions/mean_terminated_length": 459.8999938964844, "completions/min_length": 377.0, "completions/min_terminated_length": 377.0, "entropy": 0.46358602214604616, "epoch": 0.4811858608893957, "frac_reward_zero_std": 0.734375, "grad_norm": 0.006868367083370686, "kl": 0.008611810801085085, "learning_rate": 3.10609247538932e-06, "loss": 0.0017, "num_tokens": 135718621.0, "reward": 0.22626952826976776, "reward_std": 0.6750794649124146, "rewards/code_complexity_reward/mean": 0.09541015326976776, "rewards/code_complexity_reward/std": 0.27866649627685547, "rewards/code_execution_reward/mean": 0.078125, "rewards/code_execution_reward/std": 0.26863065361976624, "rewards/code_syntax_reward/mean": 0.052734375, "rewards/code_syntax_reward/std": 0.1537284255027771, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 422, "step_time": 50.60964491311461 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.966796875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 510.09765625, "completions/mean_terminated_length": 454.70587158203125, "completions/min_length": 330.0, "completions/min_terminated_length": 330.0, "entropy": 0.4715411737561226, "epoch": 0.4823261117445838, "frac_reward_zero_std": 0.8125, "grad_norm": 0.005498312879353762, "kl": 0.008714481846254785, "learning_rate": 3.096430313572547e-06, "loss": 0.002, "num_tokens": 136038423.0, "reward": 0.17822265625, "reward_std": 0.6153446435928345, "rewards/code_complexity_reward/mean": 0.07373046875, "rewards/code_complexity_reward/std": 0.25051483511924744, "rewards/code_execution_reward/mean": 0.064453125, "rewards/code_execution_reward/std": 0.24579854309558868, "rewards/code_syntax_reward/mean": 0.0400390625, "rewards/code_syntax_reward/std": 0.1358397752046585, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 423, "step_time": 44.691743914969265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.974609375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 509.666015625, "completions/mean_terminated_length": 420.0769348144531, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "entropy": 0.4667198327369988, "epoch": 0.48346636259977194, "frac_reward_zero_std": 0.8125, "grad_norm": 0.005536375567317009, "kl": 0.008563913972466253, "learning_rate": 3.0867586958121653e-06, "loss": 0.0027, "num_tokens": 136361116.0, "reward": 0.18515625596046448, "reward_std": 0.6248947978019714, "rewards/code_complexity_reward/mean": 0.07480467855930328, "rewards/code_complexity_reward/std": 0.2482028603553772, "rewards/code_execution_reward/mean": 0.068359375, "rewards/code_execution_reward/std": 0.25260838866233826, "rewards/code_syntax_reward/mean": 0.0419921875, "rewards/code_syntax_reward/std": 0.13881781697273254, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 424, "step_time": 45.06202755868435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9765625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 511.033203125, "completions/mean_terminated_length": 470.75, "completions/min_length": 418.0, "completions/min_terminated_length": 418.0, "entropy": 0.47143254708498716, "epoch": 0.4846066134549601, "frac_reward_zero_std": 0.78125, "grad_norm": 0.005739446263760328, "kl": 0.008793048073130194, "learning_rate": 3.0770777754442333e-06, "loss": 0.0017, "num_tokens": 136681205.0, "reward": 0.197998046875, "reward_std": 0.6437089443206787, "rewards/code_complexity_reward/mean": 0.08251953125, "rewards/code_complexity_reward/std": 0.26337435841560364, "rewards/code_execution_reward/mean": 0.0703125, "rewards/code_execution_reward/std": 0.25592297315597534, "rewards/code_syntax_reward/mean": 0.044921875, "rewards/code_syntax_reward/std": 0.1431187242269516, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 425, "step_time": 46.33072898630053 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.96484375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 509.693359375, "completions/mean_terminated_length": 446.3888854980469, "completions/min_length": 366.0, "completions/min_terminated_length": 366.0, "entropy": 0.47135504614561796, "epoch": 0.48574686431014824, "frac_reward_zero_std": 0.8203125, "grad_norm": 0.005984792485833168, "kl": 0.009177770589303691, "learning_rate": 3.0673877059522906e-06, "loss": 0.0019, "num_tokens": 137001464.0, "reward": 0.1962890774011612, "reward_std": 0.6290861964225769, "rewards/code_complexity_reward/mean": 0.0849609375, "rewards/code_complexity_reward/std": 0.26527392864227295, "rewards/code_execution_reward/mean": 0.064453125, "rewards/code_execution_reward/std": 0.24579854309558868, "rewards/code_syntax_reward/mean": 0.046875, "rewards/code_syntax_reward/std": 0.14588283002376556, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 426, "step_time": 53.85743281710893 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.966796875, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 510.669921875, "completions/mean_terminated_length": 471.9411926269531, "completions/min_length": 369.0, "completions/min_terminated_length": 369.0, "entropy": 0.44996650516986847, "epoch": 0.4868871151653364, "frac_reward_zero_std": 0.6953125, "grad_norm": 0.007905508391559124, "kl": 0.00943630161054898, "learning_rate": 3.057688640964934e-06, "loss": 0.0015, "num_tokens": 137319351.0, "reward": 0.241455078125, "reward_std": 0.6891147494316101, "rewards/code_complexity_reward/mean": 0.1015625, "rewards/code_complexity_reward/std": 0.282838374376297, "rewards/code_execution_reward/mean": 0.08203125, "rewards/code_execution_reward/std": 0.2746807038784027, "rewards/code_syntax_reward/mean": 0.0576171875, "rewards/code_syntax_reward/std": 0.1598084270954132, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 427, "step_time": 46.11811740230769 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.970703125, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 509.625, "completions/mean_terminated_length": 430.933349609375, "completions/min_length": 336.0, "completions/min_terminated_length": 336.0, "entropy": 0.4581920998170972, "epoch": 0.4880273660205245, "frac_reward_zero_std": 0.7265625, "grad_norm": 0.0069668907672166824, "kl": 0.009227391085005365, "learning_rate": 3.047980734253372e-06, "loss": 0.0023, "num_tokens": 137640543.0, "reward": 0.22783203423023224, "reward_std": 0.6693103313446045, "rewards/code_complexity_reward/mean": 0.09892578423023224, "rewards/code_complexity_reward/std": 0.2830222249031067, "rewards/code_execution_reward/mean": 0.07421875, "rewards/code_execution_reward/std": 0.2623828947544098, "rewards/code_syntax_reward/mean": 0.0546875, "rewards/code_syntax_reward/std": 0.15620718896389008, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 428, "step_time": 45.31984902732074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9609375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 509.34765625, "completions/mean_terminated_length": 444.1000061035156, "completions/min_length": 335.0, "completions/min_terminated_length": 335.0, "entropy": 0.47177633503451943, "epoch": 0.48916761687571264, "frac_reward_zero_std": 0.765625, "grad_norm": 0.00634920084849, "kl": 0.009407813980942592, "learning_rate": 3.038264139728997e-06, "loss": 0.0035, "num_tokens": 137960369.0, "reward": 0.23671874403953552, "reward_std": 0.6964330077171326, "rewards/code_complexity_reward/mean": 0.09707031399011612, "rewards/code_complexity_reward/std": 0.2806917726993561, "rewards/code_execution_reward/mean": 0.0859375, "rewards/code_execution_reward/std": 0.28054583072662354, "rewards/code_syntax_reward/mean": 0.0537109375, "rewards/code_syntax_reward/std": 0.15497584640979767, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 429, "step_time": 54.99858775455505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9765625, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 510.50390625, "completions/mean_terminated_length": 448.16668701171875, "completions/min_length": 374.0, "completions/min_terminated_length": 374.0, "entropy": 0.4641222362406552, "epoch": 0.4903078677309008, "frac_reward_zero_std": 0.703125, "grad_norm": 0.007431602105498314, "kl": 0.009615486036636867, "learning_rate": 3.0285390114409353e-06, "loss": 0.0025, "num_tokens": 138282991.0, "reward": 0.25810545682907104, "reward_std": 0.7148123383522034, "rewards/code_complexity_reward/mean": 0.10771484673023224, "rewards/code_complexity_reward/std": 0.29148727655410767, "rewards/code_execution_reward/mean": 0.08984375, "rewards/code_execution_reward/std": 0.2862374484539032, "rewards/code_syntax_reward/mean": 0.060546875, "rewards/code_syntax_reward/std": 0.16327762603759766, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 430, "step_time": 45.56695107743144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.97265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 510.041015625, "completions/mean_terminated_length": 440.357177734375, "completions/min_length": 306.0, "completions/min_terminated_length": 306.0, "entropy": 0.4662194517441094, "epoch": 0.49144811858608894, "frac_reward_zero_std": 0.78125, "grad_norm": 0.0065027386881411076, "kl": 0.009852838149527088, "learning_rate": 3.018805503573612e-06, "loss": 0.002, "num_tokens": 138603068.0, "reward": 0.20195311307907104, "reward_std": 0.6424376368522644, "rewards/code_complexity_reward/mean": 0.08476562798023224, "rewards/code_complexity_reward/std": 0.26433897018432617, "rewards/code_execution_reward/mean": 0.0703125, "rewards/code_execution_reward/std": 0.25592297315597534, "rewards/code_syntax_reward/mean": 0.046875, "rewards/code_syntax_reward/std": 0.14588283002376556, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 431, "step_time": 50.7315076244995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.97265625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 510.47265625, "completions/mean_terminated_length": 456.14288330078125, "completions/min_length": 363.0, "completions/min_terminated_length": 363.0, "entropy": 0.4521599542349577, "epoch": 0.4925883694412771, "frac_reward_zero_std": 0.6875, "grad_norm": 0.007921522483229637, "kl": 0.010066844333778135, "learning_rate": 3.0090637704443033e-06, "loss": 0.0023, "num_tokens": 138922846.0, "reward": 0.23955076932907104, "reward_std": 0.6690115332603455, "rewards/code_complexity_reward/mean": 0.10673828423023224, "rewards/code_complexity_reward/std": 0.28910163044929504, "rewards/code_execution_reward/mean": 0.072265625, "rewards/code_execution_reward/std": 0.2591804563999176, "rewards/code_syntax_reward/mean": 0.060546875, "rewards/code_syntax_reward/std": 0.16327762603759766, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 432, "step_time": 46.539375367574394 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.94921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 509.38671875, "completions/mean_terminated_length": 460.5384826660156, "completions/min_length": 370.0, "completions/min_terminated_length": 370.0, "entropy": 0.4658090160228312, "epoch": 0.49372862029646525, "frac_reward_zero_std": 0.703125, "grad_norm": 0.007342581637203693, "kl": 0.010192552304943092, "learning_rate": 2.9993139665006904e-06, "loss": 0.0035, "num_tokens": 139241660.0, "reward": 0.31689453125, "reward_std": 0.7881042957305908, "rewards/code_complexity_reward/mean": 0.12939453125, "rewards/code_complexity_reward/std": 0.31585657596588135, "rewards/code_execution_reward/mean": 0.115234375, "rewards/code_execution_reward/std": 0.3196168541908264, "rewards/code_syntax_reward/mean": 0.072265625, "rewards/code_syntax_reward/std": 0.17598573863506317, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 433, "step_time": 50.48422054480761 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9765625, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 510.642578125, "completions/mean_terminated_length": 454.0833435058594, "completions/min_length": 381.0, "completions/min_terminated_length": 381.0, "entropy": 0.4696646365337074, "epoch": 0.49486887115165334, "frac_reward_zero_std": 0.796875, "grad_norm": 0.006180059630423784, "kl": 0.010403607127955183, "learning_rate": 2.989556246318412e-06, "loss": 0.0021, "num_tokens": 139560449.0, "reward": 0.16943359375, "reward_std": 0.605752170085907, "rewards/code_complexity_reward/mean": 0.06591796875, "rewards/code_complexity_reward/std": 0.233780175447464, "rewards/code_execution_reward/mean": 0.06640625, "rewards/code_execution_reward/std": 0.2492343932390213, "rewards/code_syntax_reward/mean": 0.037109375, "rewards/code_syntax_reward/std": 0.1311914473772049, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 434, "step_time": 44.9902185741812 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.970703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 510.7578125, "completions/mean_terminated_length": 469.60003662109375, "completions/min_length": 386.0, "completions/min_terminated_length": 386.0, "entropy": 0.4593388712964952, "epoch": 0.4960091220068415, "frac_reward_zero_std": 0.6953125, "grad_norm": 0.008165620267391205, "kl": 0.010742391896201298, "learning_rate": 2.9797907645986124e-06, "loss": 0.0013, "num_tokens": 139880521.0, "reward": 0.26923826336860657, "reward_std": 0.7293444275856018, "rewards/code_complexity_reward/mean": 0.11103515326976776, "rewards/code_complexity_reward/std": 0.2950178384780884, "rewards/code_execution_reward/mean": 0.095703125, "rewards/code_execution_reward/std": 0.2944713830947876, "rewards/code_syntax_reward/mean": 0.0625, "rewards/code_syntax_reward/std": 0.16552117466926575, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 435, "step_time": 45.352603413164616 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.96484375, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 509.77734375, "completions/mean_terminated_length": 448.77777099609375, "completions/min_length": 328.0, "completions/min_terminated_length": 328.0, "entropy": 0.46606148732826114, "epoch": 0.49714937286202965, "frac_reward_zero_std": 0.75, "grad_norm": 0.006501306779682636, "kl": 0.010629401280311868, "learning_rate": 2.9700176761654875e-06, "loss": 0.0032, "num_tokens": 140201731.0, "reward": 0.22929687798023224, "reward_std": 0.6691051721572876, "rewards/code_complexity_reward/mean": 0.09941406548023224, "rewards/code_complexity_reward/std": 0.2821667194366455, "rewards/code_execution_reward/mean": 0.07421875, "rewards/code_execution_reward/std": 0.2623828947544098, "rewards/code_syntax_reward/mean": 0.0556640625, "rewards/code_syntax_reward/std": 0.15742282569408417, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 436, "step_time": 56.913297331891954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.95703125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 508.568359375, "completions/mean_terminated_length": 432.1363830566406, "completions/min_length": 302.0, "completions/min_terminated_length": 302.0, "entropy": 0.4588457099162042, "epoch": 0.4982896237172178, "frac_reward_zero_std": 0.7578125, "grad_norm": 0.006683026440441608, "kl": 0.010958804312394932, "learning_rate": 2.960237135963834e-06, "loss": 0.004, "num_tokens": 140522782.0, "reward": 0.260498046875, "reward_std": 0.7123479843139648, "rewards/code_complexity_reward/mean": 0.11015625298023224, "rewards/code_complexity_reward/std": 0.29562097787857056, "rewards/code_execution_reward/mean": 0.087890625, "rewards/code_execution_reward/std": 0.2834126651287079, "rewards/code_syntax_reward/mean": 0.0615234375, "rewards/code_syntax_reward/std": 0.16440613567829132, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 437, "step_time": 45.34481343533844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9453125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 507.8046875, "completions/mean_terminated_length": 435.2857360839844, "completions/min_length": 286.0, "completions/min_terminated_length": 286.0, "entropy": 0.45485378755256534, "epoch": 0.49942987457240595, "frac_reward_zero_std": 0.6796875, "grad_norm": 0.008114245720207691, "kl": 0.011828887043520808, "learning_rate": 2.9504492990565885e-06, "loss": 0.0054, "num_tokens": 140841430.0, "reward": 0.33906251192092896, "reward_std": 0.8148196339607239, "rewards/code_complexity_reward/mean": 0.13789062201976776, "rewards/code_complexity_reward/std": 0.3262636661529541, "rewards/code_execution_reward/mean": 0.125, "rewards/code_execution_reward/std": 0.3310423493385315, "rewards/code_syntax_reward/mean": 0.076171875, "rewards/code_syntax_reward/std": 0.17985260486602783, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 438, "step_time": 46.46872444357723 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.984375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 511.212890625, "completions/mean_terminated_length": 461.625, "completions/min_length": 364.0, "completions/min_terminated_length": 364.0, "entropy": 0.46369583485648036, "epoch": 0.500570125427594, "frac_reward_zero_std": 0.859375, "grad_norm": 0.005078698042780161, "kl": 0.010731192494858988, "learning_rate": 2.9406543206223735e-06, "loss": 0.0009, "num_tokens": 141161811.0, "reward": 0.11582031846046448, "reward_std": 0.49377110600471497, "rewards/code_complexity_reward/mean": 0.04941406100988388, "rewards/code_complexity_reward/std": 0.2060951143503189, "rewards/code_execution_reward/mean": 0.0390625, "rewards/code_execution_reward/std": 0.1939331740140915, "rewards/code_syntax_reward/mean": 0.02734375, "rewards/code_syntax_reward/std": 0.11379580944776535, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 439, "step_time": 46.26826854329556 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.970703125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 510.451171875, "completions/mean_terminated_length": 459.13336181640625, "completions/min_length": 379.0, "completions/min_terminated_length": 379.0, "entropy": 0.4677485814318061, "epoch": 0.5017103762827823, "frac_reward_zero_std": 0.7734375, "grad_norm": 0.006579430773854256, "kl": 0.011257904770900495, "learning_rate": 2.930852355953034e-06, "loss": 0.0013, "num_tokens": 141481086.0, "reward": 0.20986327528953552, "reward_std": 0.654111921787262, "rewards/code_complexity_reward/mean": 0.08876953274011612, "rewards/code_complexity_reward/std": 0.2705467641353607, "rewards/code_execution_reward/mean": 0.072265625, "rewards/code_execution_reward/std": 0.2591804563999176, "rewards/code_syntax_reward/mean": 0.048828125, "rewards/code_syntax_reward/std": 0.14856980741024017, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 440, "step_time": 45.792276251129806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.96875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 509.955078125, "completions/mean_terminated_length": 446.5625, "completions/min_length": 357.0, "completions/min_terminated_length": 357.0, "entropy": 0.4578982777893543, "epoch": 0.5028506271379704, "frac_reward_zero_std": 0.7265625, "grad_norm": 0.007734412327408791, "kl": 0.0116451604408212, "learning_rate": 2.9210435604511756e-06, "loss": 0.0029, "num_tokens": 141800855.0, "reward": 0.2964843809604645, "reward_std": 0.7499558329582214, "rewards/code_complexity_reward/mean": 0.12753906846046448, "rewards/code_complexity_reward/std": 0.31415310502052307, "rewards/code_execution_reward/mean": 0.09765625, "rewards/code_execution_reward/std": 0.29713961482048035, "rewards/code_syntax_reward/mean": 0.0712890625, "rewards/code_syntax_reward/std": 0.17499202489852905, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 441, "step_time": 52.385598086752 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.970703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 509.220703125, "completions/mean_terminated_length": 417.13336181640625, "completions/min_length": 272.0, "completions/min_terminated_length": 272.0, "entropy": 0.47211851458996534, "epoch": 0.5039908779931584, "frac_reward_zero_std": 0.78125, "grad_norm": 0.006186331156641245, "kl": 0.010963527849526145, "learning_rate": 2.9112280896277017e-06, "loss": 0.0045, "num_tokens": 142121680.0, "reward": 0.20546874403953552, "reward_std": 0.632161021232605, "rewards/code_complexity_reward/mean": 0.08828124403953552, "rewards/code_complexity_reward/std": 0.264444500207901, "rewards/code_execution_reward/mean": 0.06640625, "rewards/code_execution_reward/std": 0.2492343932390213, "rewards/code_syntax_reward/mean": 0.05078125, "rewards/code_syntax_reward/std": 0.15118376910686493, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 442, "step_time": 52.54983762931079 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.96875, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 509.958984375, "completions/mean_terminated_length": 446.6875, "completions/min_length": 386.0, "completions/min_terminated_length": 386.0, "entropy": 0.4572644680738449, "epoch": 0.5051311288483467, "frac_reward_zero_std": 0.765625, "grad_norm": 0.006985691841691732, "kl": 0.01157831508317031, "learning_rate": 2.90140609909935e-06, "loss": 0.0034, "num_tokens": 142443123.0, "reward": 0.22109374403953552, "reward_std": 0.6683905720710754, "rewards/code_complexity_reward/mean": 0.09316405653953552, "rewards/code_complexity_reward/std": 0.27527278661727905, "rewards/code_execution_reward/mean": 0.076171875, "rewards/code_execution_reward/std": 0.26553234457969666, "rewards/code_syntax_reward/mean": 0.0517578125, "rewards/code_syntax_reward/std": 0.15246453881263733, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 443, "step_time": 45.701365349814296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.943359375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 508.0078125, "completions/mean_terminated_length": 441.5172424316406, "completions/min_length": 284.0, "completions/min_terminated_length": 284.0, "entropy": 0.4531589704565704, "epoch": 0.5062713797035348, "frac_reward_zero_std": 0.71875, "grad_norm": 0.006958519108593464, "kl": 0.012190999725135043, "learning_rate": 2.8915777445862185e-06, "loss": 0.004, "num_tokens": 142762155.0, "reward": 0.3501953184604645, "reward_std": 0.8233029842376709, "rewards/code_complexity_reward/mean": 0.14218750596046448, "rewards/code_complexity_reward/std": 0.3294386565685272, "rewards/code_execution_reward/mean": 0.12890625, "rewards/code_execution_reward/std": 0.33542385697364807, "rewards/code_syntax_reward/mean": 0.0791015625, "rewards/code_syntax_reward/std": 0.18264412879943848, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 444, "step_time": 55.22824283130467 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.970703125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 510.6015625, "completions/mean_terminated_length": 464.2666931152344, "completions/min_length": 407.0, "completions/min_terminated_length": 407.0, "entropy": 0.4649019665084779, "epoch": 0.507411630558723, "frac_reward_zero_std": 0.6796875, "grad_norm": 0.0077860294841229916, "kl": 0.012031487145577557, "learning_rate": 2.8817431819093065e-06, "loss": 0.0019, "num_tokens": 143082063.0, "reward": 0.2793945372104645, "reward_std": 0.728682816028595, "rewards/code_complexity_reward/mean": 0.12021484225988388, "rewards/code_complexity_reward/std": 0.30584099888801575, "rewards/code_execution_reward/mean": 0.091796875, "rewards/code_execution_reward/std": 0.289021372795105, "rewards/code_syntax_reward/mean": 0.0673828125, "rewards/code_syntax_reward/std": 0.1709035038948059, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 445, "step_time": 47.391963358968496 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.97265625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 510.599609375, "completions/mean_terminated_length": 460.7857360839844, "completions/min_length": 355.0, "completions/min_terminated_length": 355.0, "entropy": 0.4755849512293935, "epoch": 0.508551881413911, "frac_reward_zero_std": 0.734375, "grad_norm": 0.006894511170685291, "kl": 0.011605919760768302, "learning_rate": 2.8719025669880357e-06, "loss": 0.0021, "num_tokens": 143403278.0, "reward": 0.22998046875, "reward_std": 0.6712028980255127, "rewards/code_complexity_reward/mean": 0.10009765625, "rewards/code_complexity_reward/std": 0.2838183045387268, "rewards/code_execution_reward/mean": 0.07421875, "rewards/code_execution_reward/std": 0.2623828947544098, "rewards/code_syntax_reward/mean": 0.0556640625, "rewards/code_syntax_reward/std": 0.15742282569408417, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 446, "step_time": 45.05633646808565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.96484375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 509.10546875, "completions/mean_terminated_length": 429.6666564941406, "completions/min_length": 333.0, "completions/min_terminated_length": 333.0, "entropy": 0.4541864702478051, "epoch": 0.5096921322690992, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.009094372391700745, "kl": 0.012192431691801175, "learning_rate": 2.8620560558377825e-06, "loss": 0.0028, "num_tokens": 143723796.0, "reward": 0.3017578125, "reward_std": 0.7683864235877991, "rewards/code_complexity_reward/mean": 0.123046875, "rewards/code_complexity_reward/std": 0.30783331394195557, "rewards/code_execution_reward/mean": 0.109375, "rewards/code_execution_reward/std": 0.31241437792778015, "rewards/code_syntax_reward/mean": 0.0693359375, "rewards/code_syntax_reward/std": 0.17297089099884033, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 447, "step_time": 45.60986705403775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.95703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 509.01171875, "completions/mean_terminated_length": 442.4545593261719, "completions/min_length": 349.0, "completions/min_terminated_length": 349.0, "entropy": 0.461833190638572, "epoch": 0.5108323831242874, "frac_reward_zero_std": 0.71875, "grad_norm": 0.007416355423629284, "kl": 0.012401593805407174, "learning_rate": 2.8522038045674026e-06, "loss": 0.0045, "num_tokens": 144043234.0, "reward": 0.3089843988418579, "reward_std": 0.7837559580802917, "rewards/code_complexity_reward/mean": 0.12441406399011612, "rewards/code_complexity_reward/std": 0.3115513324737549, "rewards/code_execution_reward/mean": 0.115234375, "rewards/code_execution_reward/std": 0.3196168541908264, "rewards/code_syntax_reward/mean": 0.0693359375, "rewards/code_syntax_reward/std": 0.17297089099884033, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 448, "step_time": 45.898409670218825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.951171875, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 507.94140625, "completions/mean_terminated_length": 428.8800048828125, "completions/min_length": 328.0, "completions/min_terminated_length": 328.0, "entropy": 0.46665124129503965, "epoch": 0.5119726339794755, "frac_reward_zero_std": 0.671875, "grad_norm": 0.008050276897847652, "kl": 0.012697945727268234, "learning_rate": 2.8423459693767586e-06, "loss": 0.0066, "num_tokens": 144362648.0, "reward": 0.27294921875, "reward_std": 0.7158573865890503, "rewards/code_complexity_reward/mean": 0.11767578125, "rewards/code_complexity_reward/std": 0.299763560295105, "rewards/code_execution_reward/mean": 0.087890625, "rewards/code_execution_reward/std": 0.2834126651287079, "rewards/code_syntax_reward/mean": 0.0673828125, "rewards/code_syntax_reward/std": 0.1709035038948059, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 449, "step_time": 45.717749807052314 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.939453125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 507.74609375, "completions/mean_terminated_length": 441.7419128417969, "completions/min_length": 328.0, "completions/min_terminated_length": 328.0, "entropy": 0.4591799471527338, "epoch": 0.5131128848346637, "frac_reward_zero_std": 0.75, "grad_norm": 0.006420123856514692, "kl": 0.01259754026250448, "learning_rate": 2.8324827065542405e-06, "loss": 0.0023, "num_tokens": 144683830.0, "reward": 0.31025391817092896, "reward_std": 0.7695955038070679, "rewards/code_complexity_reward/mean": 0.12763671576976776, "rewards/code_complexity_reward/std": 0.3096505105495453, "rewards/code_execution_reward/mean": 0.109375, "rewards/code_execution_reward/std": 0.31241437792778015, "rewards/code_syntax_reward/mean": 0.0732421875, "rewards/code_syntax_reward/std": 0.17696848511695862, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 450, "step_time": 51.51287073362619 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9765625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 510.373046875, "completions/mean_terminated_length": 442.5833435058594, "completions/min_length": 360.0, "completions/min_terminated_length": 360.0, "entropy": 0.46033009653910995, "epoch": 0.5142531356898518, "frac_reward_zero_std": 0.7109375, "grad_norm": 0.007669390179216862, "kl": 0.012852178188040853, "learning_rate": 2.822614172474289e-06, "loss": 0.0028, "num_tokens": 145003329.0, "reward": 0.24404297769069672, "reward_std": 0.6898755431175232, "rewards/code_complexity_reward/mean": 0.10146484524011612, "rewards/code_complexity_reward/std": 0.2800314426422119, "rewards/code_execution_reward/mean": 0.083984375, "rewards/code_execution_reward/std": 0.2776356339454651, "rewards/code_syntax_reward/mean": 0.05859375, "rewards/code_syntax_reward/std": 0.16097907721996307, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 451, "step_time": 45.476037707179785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.958984375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 509.298828125, "completions/mean_terminated_length": 446.1428527832031, "completions/min_length": 286.0, "completions/min_terminated_length": 286.0, "entropy": 0.4718529563397169, "epoch": 0.5153933865450399, "frac_reward_zero_std": 0.703125, "grad_norm": 0.010044282302260399, "kl": 0.013061965568340383, "learning_rate": 2.8127405235949173e-06, "loss": 0.0038, "num_tokens": 145321866.0, "reward": 0.31621092557907104, "reward_std": 0.7691335678100586, "rewards/code_complexity_reward/mean": 0.13261717557907104, "rewards/code_complexity_reward/std": 0.3146155774593353, "rewards/code_execution_reward/mean": 0.107421875, "rewards/code_execution_reward/std": 0.30995169281959534, "rewards/code_syntax_reward/mean": 0.076171875, "rewards/code_syntax_reward/std": 0.17985260486602783, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 452, "step_time": 44.893405159935355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.958984375, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 508.513671875, "completions/mean_terminated_length": 427.0, "completions/min_length": 247.0, "completions/min_terminated_length": 247.0, "entropy": 0.47247170377522707, "epoch": 0.5165336374002281, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.008096635341644287, "kl": 0.012569215468829498, "learning_rate": 2.80286191645523e-06, "loss": 0.003, "num_tokens": 145644469.0, "reward": 0.3370116949081421, "reward_std": 0.7925177216529846, "rewards/code_complexity_reward/mean": 0.14462891221046448, "rewards/code_complexity_reward/std": 0.33072513341903687, "rewards/code_execution_reward/mean": 0.111328125, "rewards/code_execution_reward/std": 0.31484565138816833, "rewards/code_syntax_reward/mean": 0.0810546875, "rewards/code_syntax_reward/std": 0.1844557821750641, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 453, "step_time": 46.165715385228395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.943359375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 507.732421875, "completions/mean_terminated_length": 436.6551818847656, "completions/min_length": 280.0, "completions/min_terminated_length": 280.0, "entropy": 0.44281477062031627, "epoch": 0.5176738882554162, "frac_reward_zero_std": 0.6796875, "grad_norm": 0.009166921488940716, "kl": 0.01351771027839277, "learning_rate": 2.792978507672941e-06, "loss": 0.0032, "num_tokens": 145965204.0, "reward": 0.32861328125, "reward_std": 0.7832122445106506, "rewards/code_complexity_reward/mean": 0.14013671875, "rewards/code_complexity_reward/std": 0.3247539699077606, "rewards/code_execution_reward/mean": 0.109375, "rewards/code_execution_reward/std": 0.31241437792778015, "rewards/code_syntax_reward/mean": 0.0791015625, "rewards/code_syntax_reward/std": 0.18264412879943848, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 454, "step_time": 52.73018193989992 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.939453125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 508.3671875, "completions/mean_terminated_length": 452.0, "completions/min_length": 357.0, "completions/min_terminated_length": 357.0, "entropy": 0.4537067203782499, "epoch": 0.5188141391106044, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.009330032393336296, "kl": 0.013909556582802907, "learning_rate": 2.7830904539418884e-06, "loss": 0.0036, "num_tokens": 146284892.0, "reward": 0.433837890625, "reward_std": 0.8745861649513245, "rewards/code_complexity_reward/mean": 0.18359375, "rewards/code_complexity_reward/std": 0.3609001338481903, "rewards/code_execution_reward/mean": 0.146484375, "rewards/code_execution_reward/std": 0.35393697023391724, "rewards/code_syntax_reward/mean": 0.103515625, "rewards/code_syntax_reward/std": 0.20278719067573547, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 455, "step_time": 45.15896955411881 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.958984375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 509.685546875, "completions/mean_terminated_length": 455.5714416503906, "completions/min_length": 358.0, "completions/min_terminated_length": 358.0, "entropy": 0.4718150547705591, "epoch": 0.5199543899657925, "frac_reward_zero_std": 0.7109375, "grad_norm": 0.008284919895231724, "kl": 0.013915267438278534, "learning_rate": 2.7731979120295564e-06, "loss": 0.0027, "num_tokens": 146603319.0, "reward": 0.25849610567092896, "reward_std": 0.6980761289596558, "rewards/code_complexity_reward/mean": 0.11201171576976776, "rewards/code_complexity_reward/std": 0.29261499643325806, "rewards/code_execution_reward/mean": 0.08203125, "rewards/code_execution_reward/std": 0.2746807038784027, "rewards/code_syntax_reward/mean": 0.064453125, "rewards/code_syntax_reward/std": 0.16771192848682404, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 456, "step_time": 46.03305127751082 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9609375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 509.486328125, "completions/mean_terminated_length": 447.6499938964844, "completions/min_length": 349.0, "completions/min_terminated_length": 349.0, "entropy": 0.448079907335341, "epoch": 0.5210946408209807, "frac_reward_zero_std": 0.640625, "grad_norm": 0.012058096937835217, "kl": 0.014175882635754533, "learning_rate": 2.763301038774583e-06, "loss": 0.0015, "num_tokens": 146923652.0, "reward": 0.30517578125, "reward_std": 0.7438315153121948, "rewards/code_complexity_reward/mean": 0.13232421875, "rewards/code_complexity_reward/std": 0.3123283386230469, "rewards/code_execution_reward/mean": 0.095703125, "rewards/code_execution_reward/std": 0.2944713830947876, "rewards/code_syntax_reward/mean": 0.0771484375, "rewards/code_syntax_reward/std": 0.18079319596290588, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 457, "step_time": 52.14820401929319 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.951171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 508.7265625, "completions/mean_terminated_length": 444.9599914550781, "completions/min_length": 340.0, "completions/min_terminated_length": 340.0, "entropy": 0.4619690221734345, "epoch": 0.5222348916761688, "frac_reward_zero_std": 0.6796875, "grad_norm": 0.008125435560941696, "kl": 0.013548452683608048, "learning_rate": 2.7533999910842766e-06, "loss": 0.0039, "num_tokens": 147245352.0, "reward": 0.30986329913139343, "reward_std": 0.77265864610672, "rewards/code_complexity_reward/mean": 0.13017578423023224, "rewards/code_complexity_reward/std": 0.3180828094482422, "rewards/code_execution_reward/mean": 0.107421875, "rewards/code_execution_reward/std": 0.30995169281959534, "rewards/code_syntax_reward/mean": 0.072265625, "rewards/code_syntax_reward/std": 0.17598573863506317, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 458, "step_time": 45.57806204166263 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.943359375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 508.13671875, "completions/mean_terminated_length": 443.7930908203125, "completions/min_length": 266.0, "completions/min_terminated_length": 266.0, "entropy": 0.4498466379009187, "epoch": 0.5233751425313569, "frac_reward_zero_std": 0.625, "grad_norm": 0.009083162061870098, "kl": 0.014867047910229303, "learning_rate": 2.743494925932129e-06, "loss": 0.0056, "num_tokens": 147563814.0, "reward": 0.3573242425918579, "reward_std": 0.8227204084396362, "rewards/code_complexity_reward/mean": 0.14736329019069672, "rewards/code_complexity_reward/std": 0.3317331075668335, "rewards/code_execution_reward/mean": 0.126953125, "rewards/code_execution_reward/std": 0.33324605226516724, "rewards/code_syntax_reward/mean": 0.0830078125, "rewards/code_syntax_reward/std": 0.18622928857803345, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 459, "step_time": 46.184586523100734 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.939453125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 508.609375, "completions/mean_terminated_length": 456.0, "completions/min_length": 283.0, "completions/min_terminated_length": 283.0, "entropy": 0.46414555329829454, "epoch": 0.5245153933865451, "frac_reward_zero_std": 0.6640625, "grad_norm": 0.008905015885829926, "kl": 0.014441972627537325, "learning_rate": 2.7335860003553257e-06, "loss": 0.0037, "num_tokens": 147884686.0, "reward": 0.3153320252895355, "reward_std": 0.7684484720230103, "rewards/code_complexity_reward/mean": 0.13564452528953552, "rewards/code_complexity_reward/std": 0.3211159110069275, "rewards/code_execution_reward/mean": 0.103515625, "rewards/code_execution_reward/std": 0.30492907762527466, "rewards/code_syntax_reward/mean": 0.076171875, "rewards/code_syntax_reward/std": 0.17985260486602783, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 460, "step_time": 50.181749490089715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.94921875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 508.986328125, "completions/mean_terminated_length": 452.65386962890625, "completions/min_length": 258.0, "completions/min_terminated_length": 258.0, "entropy": 0.4475742466747761, "epoch": 0.5256556442417332, "frac_reward_zero_std": 0.6796875, "grad_norm": 0.007061809301376343, "kl": 0.014764166175154969, "learning_rate": 2.723673371452254e-06, "loss": 0.002, "num_tokens": 148202803.0, "reward": 0.3746093809604645, "reward_std": 0.8393279910087585, "rewards/code_complexity_reward/mean": 0.15000000596046448, "rewards/code_complexity_reward/std": 0.3310866951942444, "rewards/code_execution_reward/mean": 0.138671875, "rewards/code_execution_reward/std": 0.34594178199768066, "rewards/code_syntax_reward/mean": 0.0859375, "rewards/code_syntax_reward/std": 0.18882036209106445, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 461, "step_time": 52.665643780492246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.94921875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 507.80078125, "completions/mean_terminated_length": 429.3077087402344, "completions/min_length": 295.0, "completions/min_terminated_length": 295.0, "entropy": 0.451062242500484, "epoch": 0.5267958950969214, "frac_reward_zero_std": 0.625, "grad_norm": 0.008589810691773891, "kl": 0.015545004222076386, "learning_rate": 2.713757196380017e-06, "loss": 0.0024, "num_tokens": 148522397.0, "reward": 0.3857421875, "reward_std": 0.8318994045257568, "rewards/code_complexity_reward/mean": 0.1650390625, "rewards/code_complexity_reward/std": 0.3454461991786957, "rewards/code_execution_reward/mean": 0.126953125, "rewards/code_execution_reward/std": 0.33324605226516724, "rewards/code_syntax_reward/mean": 0.09375, "rewards/code_syntax_reward/std": 0.19534705579280853, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 462, "step_time": 46.01741787977517 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.935546875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 507.580078125, "completions/mean_terminated_length": 443.42425537109375, "completions/min_length": 332.0, "completions/min_terminated_length": 332.0, "entropy": 0.4546721428632736, "epoch": 0.5279361459521095, "frac_reward_zero_std": 0.671875, "grad_norm": 0.008516171015799046, "kl": 0.015841831947909668, "learning_rate": 2.70383763235194e-06, "loss": 0.0043, "num_tokens": 148841266.0, "reward": 0.3501953184604645, "reward_std": 0.8157991766929626, "rewards/code_complexity_reward/mean": 0.14218750596046448, "rewards/code_complexity_reward/std": 0.3246968984603882, "rewards/code_execution_reward/mean": 0.126953125, "rewards/code_execution_reward/std": 0.33324605226516724, "rewards/code_syntax_reward/mean": 0.0810546875, "rewards/code_syntax_reward/std": 0.1844557821750641, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 463, "step_time": 45.87797906342894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.939453125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 507.7734375, "completions/mean_terminated_length": 442.19354248046875, "completions/min_length": 308.0, "completions/min_terminated_length": 308.0, "entropy": 0.4536883942782879, "epoch": 0.5290763968072976, "frac_reward_zero_std": 0.59375, "grad_norm": 0.009585030376911163, "kl": 0.015685021819081157, "learning_rate": 2.693914836635076e-06, "loss": 0.0045, "num_tokens": 149158774.0, "reward": 0.3319336175918579, "reward_std": 0.767471432685852, "rewards/code_complexity_reward/mean": 0.14638671278953552, "rewards/code_complexity_reward/std": 0.3274039924144745, "rewards/code_execution_reward/mean": 0.1015625, "rewards/code_execution_reward/std": 0.30236753821372986, "rewards/code_syntax_reward/mean": 0.083984375, "rewards/code_syntax_reward/std": 0.1871020793914795, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 464, "step_time": 52.01663722284138 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.931640625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 507.02734375, "completions/mean_terminated_length": 439.25714111328125, "completions/min_length": 324.0, "completions/min_terminated_length": 324.0, "entropy": 0.4566339156590402, "epoch": 0.5302166476624858, "frac_reward_zero_std": 0.6796875, "grad_norm": 0.009238924831151962, "kl": 0.01576633438526187, "learning_rate": 2.6839889665477144e-06, "loss": 0.0027, "num_tokens": 149479008.0, "reward": 0.3759765625, "reward_std": 0.8369167447090149, "rewards/code_complexity_reward/mean": 0.1572265625, "rewards/code_complexity_reward/std": 0.3421992361545563, "rewards/code_execution_reward/mean": 0.130859375, "rewards/code_execution_reward/std": 0.33757632970809937, "rewards/code_syntax_reward/mean": 0.087890625, "rewards/code_syntax_reward/std": 0.1905031055212021, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 465, "step_time": 48.462163398973644 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.94921875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 508.544921875, "completions/mean_terminated_length": 443.9615478515625, "completions/min_length": 316.0, "completions/min_terminated_length": 316.0, "entropy": 0.4556561401113868, "epoch": 0.5313568985176739, "frac_reward_zero_std": 0.640625, "grad_norm": 0.009325365535914898, "kl": 0.01602209228440188, "learning_rate": 2.6740601794568866e-06, "loss": 0.0052, "num_tokens": 149798495.0, "reward": 0.3030761778354645, "reward_std": 0.7441399097442627, "rewards/code_complexity_reward/mean": 0.12802734971046448, "rewards/code_complexity_reward/std": 0.3062790632247925, "rewards/code_execution_reward/mean": 0.099609375, "rewards/code_execution_reward/std": 0.29977133870124817, "rewards/code_syntax_reward/mean": 0.0751953125, "rewards/code_syntax_reward/std": 0.17890173196792603, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 466, "step_time": 52.260262820869684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.94921875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 508.892578125, "completions/mean_terminated_length": 450.8077087402344, "completions/min_length": 214.0, "completions/min_terminated_length": 214.0, "entropy": 0.44783236319199204, "epoch": 0.5324971493728621, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.010139953345060349, "kl": 0.01634662935975939, "learning_rate": 2.664128632775871e-06, "loss": 0.0021, "num_tokens": 150117584.0, "reward": 0.3643554747104645, "reward_std": 0.8044360876083374, "rewards/code_complexity_reward/mean": 0.15830078721046448, "rewards/code_complexity_reward/std": 0.33955544233322144, "rewards/code_execution_reward/mean": 0.115234375, "rewards/code_execution_reward/std": 0.3196168541908264, "rewards/code_syntax_reward/mean": 0.0908203125, "rewards/code_syntax_reward/std": 0.19296257197856903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 467, "step_time": 45.37678369600326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.919921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 505.65234375, "completions/mean_terminated_length": 432.731689453125, "completions/min_length": 282.0, "completions/min_terminated_length": 282.0, "entropy": 0.4387646038085222, "epoch": 0.5336374002280502, "frac_reward_zero_std": 0.65625, "grad_norm": 0.008446444757282734, "kl": 0.01676671589666512, "learning_rate": 2.6541944839616957e-06, "loss": 0.0039, "num_tokens": 150433994.0, "reward": 0.38837888836860657, "reward_std": 0.8360438942909241, "rewards/code_complexity_reward/mean": 0.16572265326976776, "rewards/code_complexity_reward/std": 0.34717458486557007, "rewards/code_execution_reward/mean": 0.12890625, "rewards/code_execution_reward/std": 0.33542385697364807, "rewards/code_syntax_reward/mean": 0.09375, "rewards/code_syntax_reward/std": 0.19534705579280853, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 468, "step_time": 52.12633262388408 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.939453125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 508.318359375, "completions/mean_terminated_length": 451.19354248046875, "completions/min_length": 318.0, "completions/min_terminated_length": 318.0, "entropy": 0.4502713750116527, "epoch": 0.5347776510832383, "frac_reward_zero_std": 0.5625, "grad_norm": 0.010266815312206745, "kl": 0.017378125703544356, "learning_rate": 2.644257890512646e-06, "loss": 0.0042, "num_tokens": 150752737.0, "reward": 0.4193359613418579, "reward_std": 0.8627863526344299, "rewards/code_complexity_reward/mean": 0.17617188394069672, "rewards/code_complexity_reward/std": 0.3532884120941162, "rewards/code_execution_reward/mean": 0.142578125, "rewards/code_execution_reward/std": 0.3499840497970581, "rewards/code_syntax_reward/mean": 0.1005859375, "rewards/code_syntax_reward/std": 0.2006341516971588, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 469, "step_time": 45.657094988040626 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 508.6796875, "completions/mean_terminated_length": 427.0, "completions/min_length": 320.0, "completions/min_terminated_length": 320.0, "entropy": 0.4528074311092496, "epoch": 0.5359179019384265, "frac_reward_zero_std": 0.6953125, "grad_norm": 0.007949813269078732, "kl": 0.016713584554963745, "learning_rate": 2.634319009965762e-06, "loss": 0.0035, "num_tokens": 151073609.0, "reward": 0.2821289002895355, "reward_std": 0.7268326282501221, "rewards/code_complexity_reward/mean": 0.12099609524011612, "rewards/code_complexity_reward/std": 0.30331432819366455, "rewards/code_execution_reward/mean": 0.091796875, "rewards/code_execution_reward/std": 0.289021372795105, "rewards/code_syntax_reward/mean": 0.0693359375, "rewards/code_syntax_reward/std": 0.17297089099884033, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 470, "step_time": 45.55007756780833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.955078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 509.0234375, "completions/mean_terminated_length": 445.7391357421875, "completions/min_length": 346.0, "completions/min_terminated_length": 346.0, "entropy": 0.44226268166676164, "epoch": 0.5370581527936146, "frac_reward_zero_std": 0.640625, "grad_norm": 0.009775055572390556, "kl": 0.01692591565370094, "learning_rate": 2.6243779998943496e-06, "loss": 0.0033, "num_tokens": 151394957.0, "reward": 0.36357420682907104, "reward_std": 0.8099106550216675, "rewards/code_complexity_reward/mean": 0.15361326932907104, "rewards/code_complexity_reward/std": 0.33307793736457825, "rewards/code_execution_reward/mean": 0.12109375, "rewards/code_execution_reward/std": 0.3265552520751953, "rewards/code_syntax_reward/mean": 0.0888671875, "rewards/code_syntax_reward/std": 0.1913314312696457, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 471, "step_time": 50.33548277709633 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.947265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 508.509765625, "completions/mean_terminated_length": 445.8148193359375, "completions/min_length": 374.0, "completions/min_terminated_length": 374.0, "entropy": 0.46144469920545816, "epoch": 0.5381984036488028, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.009142403490841389, "kl": 0.016577332688029855, "learning_rate": 2.614435017905469e-06, "loss": 0.0039, "num_tokens": 151715082.0, "reward": 0.3516601622104645, "reward_std": 0.8003987669944763, "rewards/code_complexity_reward/mean": 0.15048828721046448, "rewards/code_complexity_reward/std": 0.3323326110839844, "rewards/code_execution_reward/mean": 0.115234375, "rewards/code_execution_reward/std": 0.3196168541908264, "rewards/code_syntax_reward/mean": 0.0859375, "rewards/code_syntax_reward/std": 0.18882036209106445, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 472, "step_time": 50.196260027587414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9453125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 508.345703125, "completions/mean_terminated_length": 445.1785888671875, "completions/min_length": 351.0, "completions/min_terminated_length": 351.0, "entropy": 0.45247003622353077, "epoch": 0.5393386545039909, "frac_reward_zero_std": 0.609375, "grad_norm": 0.010621017776429653, "kl": 0.017380268065608107, "learning_rate": 2.6044902216374497e-06, "loss": 0.0045, "num_tokens": 152034391.0, "reward": 0.3902343809604645, "reward_std": 0.8455954194068909, "rewards/code_complexity_reward/mean": 0.16171875596046448, "rewards/code_complexity_reward/std": 0.3425462543964386, "rewards/code_execution_reward/mean": 0.13671875, "rewards/code_execution_reward/std": 0.3438861668109894, "rewards/code_syntax_reward/mean": 0.091796875, "rewards/code_syntax_reward/std": 0.1937655806541443, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 473, "step_time": 45.746439452283084 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.93359375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 507.80078125, "completions/mean_terminated_length": 448.76470947265625, "completions/min_length": 326.0, "completions/min_terminated_length": 326.0, "entropy": 0.452417460270226, "epoch": 0.540478905359179, "frac_reward_zero_std": 0.640625, "grad_norm": 0.011542434804141521, "kl": 0.01820647502609063, "learning_rate": 2.5945437687573816e-06, "loss": 0.0048, "num_tokens": 152354421.0, "reward": 0.38056641817092896, "reward_std": 0.825414776802063, "rewards/code_complexity_reward/mean": 0.16474609076976776, "rewards/code_complexity_reward/std": 0.34693560004234314, "rewards/code_execution_reward/mean": 0.123046875, "rewards/code_execution_reward/std": 0.32881227135658264, "rewards/code_syntax_reward/mean": 0.0927734375, "rewards/code_syntax_reward/std": 0.19456037878990173, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 474, "step_time": 46.618425857275724 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.958984375, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 509.69921875, "completions/mean_terminated_length": 455.90478515625, "completions/min_length": 347.0, "completions/min_terminated_length": 347.0, "entropy": 0.46505724359303713, "epoch": 0.5416191562143672, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.008463387377560139, "kl": 0.017999411473283544, "learning_rate": 2.584595816958621e-06, "loss": 0.0029, "num_tokens": 152675187.0, "reward": 0.31591796875, "reward_std": 0.7313001751899719, "rewards/code_complexity_reward/mean": 0.14501953125, "rewards/code_complexity_reward/std": 0.323310911655426, "rewards/code_execution_reward/mean": 0.0859375, "rewards/code_execution_reward/std": 0.28054583072662354, "rewards/code_syntax_reward/mean": 0.0849609375, "rewards/code_syntax_reward/std": 0.1879657357931137, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 475, "step_time": 45.07798963692039 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.923828125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 507.02734375, "completions/mean_terminated_length": 446.71795654296875, "completions/min_length": 331.0, "completions/min_terminated_length": 331.0, "entropy": 0.46170778665691614, "epoch": 0.5427594070695553, "frac_reward_zero_std": 0.6875, "grad_norm": 0.007722075562924147, "kl": 0.018198048332124017, "learning_rate": 2.574646523958288e-06, "loss": 0.0056, "num_tokens": 152994065.0, "reward": 0.35107421875, "reward_std": 0.7966685891151428, "rewards/code_complexity_reward/mean": 0.15185546875, "rewards/code_complexity_reward/std": 0.3350693881511688, "rewards/code_execution_reward/mean": 0.11328125, "rewards/code_execution_reward/std": 0.3172462284564972, "rewards/code_syntax_reward/mean": 0.0859375, "rewards/code_syntax_reward/std": 0.18882036209106445, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 476, "step_time": 45.203736883588135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.951171875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 508.630859375, "completions/mean_terminated_length": 443.0, "completions/min_length": 308.0, "completions/min_terminated_length": 308.0, "entropy": 0.4437158624641597, "epoch": 0.5438996579247435, "frac_reward_zero_std": 0.59375, "grad_norm": 0.011000928469002247, "kl": 0.01863697798398789, "learning_rate": 2.564696047494765e-06, "loss": 0.0025, "num_tokens": 153313152.0, "reward": 0.3829101324081421, "reward_std": 0.8375810384750366, "rewards/code_complexity_reward/mean": 0.15927734971046448, "rewards/code_complexity_reward/std": 0.3397029936313629, "rewards/code_execution_reward/mean": 0.1328125, "rewards/code_execution_reward/std": 0.33970388770103455, "rewards/code_syntax_reward/mean": 0.0908203125, "rewards/code_syntax_reward/std": 0.19296257197856903, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 477, "step_time": 45.23554949555546 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.95703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 509.17578125, "completions/mean_terminated_length": 446.2727355957031, "completions/min_length": 331.0, "completions/min_terminated_length": 331.0, "entropy": 0.4526534201577306, "epoch": 0.5450399087799316, "frac_reward_zero_std": 0.625, "grad_norm": 0.01046576164662838, "kl": 0.01902952796081081, "learning_rate": 2.5547445453252e-06, "loss": 0.0037, "num_tokens": 153634550.0, "reward": 0.33867189288139343, "reward_std": 0.7859695553779602, "rewards/code_complexity_reward/mean": 0.14238281548023224, "rewards/code_complexity_reward/std": 0.32164356112480164, "rewards/code_execution_reward/mean": 0.11328125, "rewards/code_execution_reward/std": 0.3172462284564972, "rewards/code_syntax_reward/mean": 0.0830078125, "rewards/code_syntax_reward/std": 0.18622928857803345, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 478, "step_time": 45.42669451329857 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.888671875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 502.00390625, "completions/mean_terminated_length": 422.2105407714844, "completions/min_length": 254.0, "completions/min_terminated_length": 254.0, "entropy": 0.43391528027132154, "epoch": 0.5461801596351197, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.01174970343708992, "kl": 0.019183136624633335, "learning_rate": 2.5447921752230003e-06, "loss": 0.0075, "num_tokens": 153951004.0, "reward": 0.6356445550918579, "reward_std": 1.0149874687194824, "rewards/code_complexity_reward/mean": 0.2586914002895355, "rewards/code_complexity_reward/std": 0.4039647579193115, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.146484375, "rewards/code_syntax_reward/std": 0.227784663438797, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 479, "step_time": 45.841188528575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.91796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 506.05859375, "completions/mean_terminated_length": 439.5714416503906, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "entropy": 0.44236340653151274, "epoch": 0.5473204104903079, "frac_reward_zero_std": 0.578125, "grad_norm": 0.011157412081956863, "kl": 0.018764639215078205, "learning_rate": 2.5348390949753343e-06, "loss": 0.0067, "num_tokens": 154270190.0, "reward": 0.43193361163139343, "reward_std": 0.8598120808601379, "rewards/code_complexity_reward/mean": 0.18583983182907104, "rewards/code_complexity_reward/std": 0.3573654294013977, "rewards/code_execution_reward/mean": 0.138671875, "rewards/code_execution_reward/std": 0.34594178199768066, "rewards/code_syntax_reward/mean": 0.107421875, "rewards/code_syntax_reward/std": 0.20555779337882996, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 480, "step_time": 46.19288328103721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.939453125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 507.501953125, "completions/mean_terminated_length": 437.70965576171875, "completions/min_length": 272.0, "completions/min_terminated_length": 272.0, "entropy": 0.4468354871496558, "epoch": 0.548460661345496, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.010441267862915993, "kl": 0.019584804016631097, "learning_rate": 2.5248854623806297e-06, "loss": 0.0051, "num_tokens": 154589863.0, "reward": 0.4146484136581421, "reward_std": 0.8612046837806702, "rewards/code_complexity_reward/mean": 0.17148436605930328, "rewards/code_complexity_reward/std": 0.3478158712387085, "rewards/code_execution_reward/mean": 0.14453125, "rewards/code_execution_reward/std": 0.35197147727012634, "rewards/code_syntax_reward/mean": 0.0986328125, "rewards/code_syntax_reward/std": 0.1991618573665619, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 481, "step_time": 45.740308043546975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.916015625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 506.357421875, "completions/mean_terminated_length": 444.81396484375, "completions/min_length": 288.0, "completions/min_terminated_length": 288.0, "entropy": 0.4351339163258672, "epoch": 0.5496009122006842, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.01065710000693798, "kl": 0.02010486756626051, "learning_rate": 2.514931435246071e-06, "loss": 0.0036, "num_tokens": 154907370.0, "reward": 0.474365234375, "reward_std": 0.8878079056739807, "rewards/code_complexity_reward/mean": 0.20751953125, "rewards/code_complexity_reward/std": 0.37567561864852905, "rewards/code_execution_reward/mean": 0.1484375, "rewards/code_execution_reward/std": 0.35588082671165466, "rewards/code_syntax_reward/mean": 0.1181640625, "rewards/code_syntax_reward/std": 0.21262075006961823, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 482, "step_time": 53.535142820328474 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.919921875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 505.73828125, "completions/mean_terminated_length": 433.80487060546875, "completions/min_length": 336.0, "completions/min_terminated_length": 336.0, "entropy": 0.4454667204990983, "epoch": 0.5507411630558723, "frac_reward_zero_std": 0.59375, "grad_norm": 0.008982187137007713, "kl": 0.019718001189175993, "learning_rate": 2.504977171385098e-06, "loss": 0.006, "num_tokens": 155225560.0, "reward": 0.4603515863418579, "reward_std": 0.8982480764389038, "rewards/code_complexity_reward/mean": 0.19082030653953552, "rewards/code_complexity_reward/std": 0.36334457993507385, "rewards/code_execution_reward/mean": 0.16015625, "rewards/code_execution_reward/std": 0.3671095669269562, "rewards/code_syntax_reward/mean": 0.109375, "rewards/code_syntax_reward/std": 0.20690147578716278, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 483, "step_time": 46.01011315919459 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.955078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 509.744140625, "completions/mean_terminated_length": 461.7826232910156, "completions/min_length": 343.0, "completions/min_terminated_length": 343.0, "entropy": 0.4409339828416705, "epoch": 0.5518814139110604, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.010690787807106972, "kl": 0.020447962131584063, "learning_rate": 2.4950228286149028e-06, "loss": 0.0029, "num_tokens": 155546533.0, "reward": 0.39824217557907104, "reward_std": 0.8306193351745605, "rewards/code_complexity_reward/mean": 0.16777342557907104, "rewards/code_complexity_reward/std": 0.33934226632118225, "rewards/code_execution_reward/mean": 0.130859375, "rewards/code_execution_reward/std": 0.33757632970809937, "rewards/code_syntax_reward/mean": 0.099609375, "rewards/code_syntax_reward/std": 0.19990174472332, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 484, "step_time": 54.67362951952964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.90234375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 503.916015625, "completions/mean_terminated_length": 429.2200012207031, "completions/min_length": 316.0, "completions/min_terminated_length": 316.0, "entropy": 0.42477474408224225, "epoch": 0.5530216647662486, "frac_reward_zero_std": 0.5, "grad_norm": 0.011858484707772732, "kl": 0.02075686978059821, "learning_rate": 2.48506856475393e-06, "loss": 0.0087, "num_tokens": 155863214.0, "reward": 0.530566394329071, "reward_std": 0.941684365272522, "rewards/code_complexity_reward/mean": 0.22392578423023224, "rewards/code_complexity_reward/std": 0.38601890206336975, "rewards/code_execution_reward/mean": 0.1796875, "rewards/code_execution_reward/std": 0.38430243730545044, "rewards/code_syntax_reward/mean": 0.126953125, "rewards/code_syntax_reward/std": 0.21783512830734253, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 485, "step_time": 44.940040577203035 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.916015625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 507.326171875, "completions/mean_terminated_length": 456.3488464355469, "completions/min_length": 276.0, "completions/min_terminated_length": 276.0, "entropy": 0.44264097791165113, "epoch": 0.5541619156214367, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.009920249693095684, "kl": 0.020375119973323308, "learning_rate": 2.475114537619371e-06, "loss": 0.0035, "num_tokens": 156182661.0, "reward": 0.5013672113418579, "reward_std": 0.9143924713134766, "rewards/code_complexity_reward/mean": 0.21621093153953552, "rewards/code_complexity_reward/std": 0.3808540403842926, "rewards/code_execution_reward/mean": 0.162109375, "rewards/code_execution_reward/std": 0.3689115643501282, "rewards/code_syntax_reward/mean": 0.123046875, "rewards/code_syntax_reward/std": 0.21557752788066864, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 486, "step_time": 45.80288370791823 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.923828125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 505.7734375, "completions/mean_terminated_length": 430.25640869140625, "completions/min_length": 283.0, "completions/min_terminated_length": 283.0, "entropy": 0.44525508442893624, "epoch": 0.5553021664766249, "frac_reward_zero_std": 0.578125, "grad_norm": 0.00922943465411663, "kl": 0.02050341881113127, "learning_rate": 2.4651609050246674e-06, "loss": 0.0054, "num_tokens": 156502521.0, "reward": 0.4932129383087158, "reward_std": 0.9037073850631714, "rewards/code_complexity_reward/mean": 0.21464844048023224, "rewards/code_complexity_reward/std": 0.3798241913318634, "rewards/code_execution_reward/mean": 0.15625, "rewards/code_execution_reward/std": 0.36344730854034424, "rewards/code_syntax_reward/mean": 0.1220703125, "rewards/code_syntax_reward/std": 0.21499831974506378, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 487, "step_time": 45.82211856730282 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.91796875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 505.92578125, "completions/mean_terminated_length": 437.952392578125, "completions/min_length": 277.0, "completions/min_terminated_length": 277.0, "entropy": 0.43904267996549606, "epoch": 0.556442417331813, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.011199485510587692, "kl": 0.021600174193736166, "learning_rate": 2.4552078247770005e-06, "loss": 0.0046, "num_tokens": 156819251.0, "reward": 0.4657226800918579, "reward_std": 0.8801640868186951, "rewards/code_complexity_reward/mean": 0.19912110269069672, "rewards/code_complexity_reward/std": 0.364300012588501, "rewards/code_execution_reward/mean": 0.150390625, "rewards/code_execution_reward/std": 0.35780346393585205, "rewards/code_syntax_reward/mean": 0.1162109375, "rewards/code_syntax_reward/std": 0.21139481663703918, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 488, "step_time": 45.95646474137902 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.91015625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 504.3046875, "completions/mean_terminated_length": 426.34783935546875, "completions/min_length": 304.0, "completions/min_terminated_length": 304.0, "entropy": 0.44514686753973365, "epoch": 0.5575826681870011, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.011016963049769402, "kl": 0.021240481379209086, "learning_rate": 2.4452554546748008e-06, "loss": 0.0057, "num_tokens": 157137119.0, "reward": 0.5140625238418579, "reward_std": 0.934612512588501, "rewards/code_complexity_reward/mean": 0.21230468153953552, "rewards/code_complexity_reward/std": 0.37597227096557617, "rewards/code_execution_reward/mean": 0.1796875, "rewards/code_execution_reward/std": 0.38430243730545044, "rewards/code_syntax_reward/mean": 0.1220703125, "rewards/code_syntax_reward/std": 0.21499831974506378, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 489, "step_time": 45.65301496721804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.92578125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 505.951171875, "completions/mean_terminated_length": 430.5, "completions/min_length": 289.0, "completions/min_terminated_length": 289.0, "entropy": 0.4390854174271226, "epoch": 0.5587229190421893, "frac_reward_zero_std": 0.5625, "grad_norm": 0.010412262752652168, "kl": 0.021047023008577526, "learning_rate": 2.4353039525052354e-06, "loss": 0.0037, "num_tokens": 157456486.0, "reward": 0.515625, "reward_std": 0.9269924759864807, "rewards/code_complexity_reward/mean": 0.21484375, "rewards/code_complexity_reward/std": 0.3751116991043091, "rewards/code_execution_reward/mean": 0.17578125, "rewards/code_execution_reward/std": 0.3810062110424042, "rewards/code_syntax_reward/mean": 0.125, "rewards/code_syntax_reward/std": 0.21671809256076813, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 490, "step_time": 45.44733292516321 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.90625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 505.248046875, "completions/mean_terminated_length": 439.97918701171875, "completions/min_length": 299.0, "completions/min_terminated_length": 299.0, "entropy": 0.45242282655090094, "epoch": 0.5598631698973774, "frac_reward_zero_std": 0.609375, "grad_norm": 0.010004536248743534, "kl": 0.021719570097047836, "learning_rate": 2.425353476041713e-06, "loss": 0.0058, "num_tokens": 157774789.0, "reward": 0.44951173663139343, "reward_std": 0.8701253533363342, "rewards/code_complexity_reward/mean": 0.19853514432907104, "rewards/code_complexity_reward/std": 0.370807021856308, "rewards/code_execution_reward/mean": 0.138671875, "rewards/code_execution_reward/std": 0.34594178199768066, "rewards/code_syntax_reward/mean": 0.1123046875, "rewards/code_syntax_reward/std": 0.20886647701263428, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 491, "step_time": 46.09171526879072 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.927734375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 507.546875, "completions/mean_terminated_length": 450.3783874511719, "completions/min_length": 306.0, "completions/min_terminated_length": 306.0, "entropy": 0.4417063067667186, "epoch": 0.5610034207525656, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.01095584500581026, "kl": 0.021532452490646392, "learning_rate": 2.4154041830413803e-06, "loss": 0.0036, "num_tokens": 158094513.0, "reward": 0.47978514432907104, "reward_std": 0.8735270500183105, "rewards/code_complexity_reward/mean": 0.21123045682907104, "rewards/code_complexity_reward/std": 0.3702036440372467, "rewards/code_execution_reward/mean": 0.14453125, "rewards/code_execution_reward/std": 0.35197147727012634, "rewards/code_syntax_reward/mean": 0.1240234375, "rewards/code_syntax_reward/std": 0.21615077555179596, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 492, "step_time": 45.44064914807677 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.912109375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 506.376953125, "completions/mean_terminated_length": 448.0222473144531, "completions/min_length": 263.0, "completions/min_terminated_length": 263.0, "entropy": 0.433879507239908, "epoch": 0.5621436716077537, "frac_reward_zero_std": 0.5, "grad_norm": 0.010974327102303505, "kl": 0.0224092879507225, "learning_rate": 2.4054562312426193e-06, "loss": 0.0054, "num_tokens": 158411686.0, "reward": 0.5447266101837158, "reward_std": 0.9332001209259033, "rewards/code_complexity_reward/mean": 0.23222656548023224, "rewards/code_complexity_reward/std": 0.38444119691848755, "rewards/code_execution_reward/mean": 0.177734375, "rewards/code_execution_reward/std": 0.3826628625392914, "rewards/code_syntax_reward/mean": 0.134765625, "rewards/code_syntax_reward/std": 0.22207511961460114, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 493, "step_time": 51.14283033274114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.919921875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 507.162109375, "completions/mean_terminated_length": 451.5853576660156, "completions/min_length": 366.0, "completions/min_terminated_length": 366.0, "entropy": 0.4322985843755305, "epoch": 0.5632839224629419, "frac_reward_zero_std": 0.578125, "grad_norm": 0.010369502007961273, "kl": 0.0226724149833899, "learning_rate": 2.395509778362552e-06, "loss": 0.0034, "num_tokens": 158732193.0, "reward": 0.44892579317092896, "reward_std": 0.8699954748153687, "rewards/code_complexity_reward/mean": 0.19404296576976776, "rewards/code_complexity_reward/std": 0.36285269260406494, "rewards/code_execution_reward/mean": 0.142578125, "rewards/code_execution_reward/std": 0.3499840497970581, "rewards/code_syntax_reward/mean": 0.1123046875, "rewards/code_syntax_reward/std": 0.20886647701263428, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 494, "step_time": 45.357996008358896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.92578125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 506.373046875, "completions/mean_terminated_length": 436.1842041015625, "completions/min_length": 286.0, "completions/min_terminated_length": 286.0, "entropy": 0.42972225695848465, "epoch": 0.56442417331813, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.010849324986338615, "kl": 0.02271989904693328, "learning_rate": 2.3855649820945313e-06, "loss": 0.0048, "num_tokens": 159050324.0, "reward": 0.505664050579071, "reward_std": 0.9139906167984009, "rewards/code_complexity_reward/mean": 0.21171873807907104, "rewards/code_complexity_reward/std": 0.37115854024887085, "rewards/code_execution_reward/mean": 0.169921875, "rewards/code_execution_reward/std": 0.3759314715862274, "rewards/code_syntax_reward/mean": 0.1240234375, "rewards/code_syntax_reward/std": 0.21615077555179596, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 495, "step_time": 46.04650471918285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8984375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 505.416015625, "completions/mean_terminated_length": 447.173095703125, "completions/min_length": 301.0, "completions/min_terminated_length": 301.0, "entropy": 0.44270266499370337, "epoch": 0.5655644241733181, "frac_reward_zero_std": 0.609375, "grad_norm": 0.009648434817790985, "kl": 0.021812985942233354, "learning_rate": 2.375622000105651e-06, "loss": 0.0007, "num_tokens": 159369785.0, "reward": 0.5376952886581421, "reward_std": 0.9506229758262634, "rewards/code_complexity_reward/mean": 0.22324219346046448, "rewards/code_complexity_reward/std": 0.38487401604652405, "rewards/code_execution_reward/mean": 0.1875, "rewards/code_execution_reward/std": 0.39069411158561707, "rewards/code_syntax_reward/mean": 0.126953125, "rewards/code_syntax_reward/std": 0.21783512830734253, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 496, "step_time": 51.576270886696875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.900390625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 506.09765625, "completions/mean_terminated_length": 452.7451171875, "completions/min_length": 319.0, "completions/min_terminated_length": 319.0, "entropy": 0.4261136408895254, "epoch": 0.5667046750285063, "frac_reward_zero_std": 0.5, "grad_norm": 0.010974771343171597, "kl": 0.02246061890036799, "learning_rate": 2.3656809900342383e-06, "loss": 0.006, "num_tokens": 159688203.0, "reward": 0.6046386957168579, "reward_std": 0.9896500706672668, "rewards/code_complexity_reward/mean": 0.24697265028953552, "rewards/code_complexity_reward/std": 0.39384403824806213, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.142578125, "rewards/code_syntax_reward/std": 0.22596518695354462, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 497, "step_time": 45.20157121587545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 502.064453125, "completions/mean_terminated_length": 432.515625, "completions/min_length": 292.0, "completions/min_terminated_length": 292.0, "entropy": 0.4240594655275345, "epoch": 0.5678449258836944, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.012475437484681606, "kl": 0.02395991433877498, "learning_rate": 2.355742109487355e-06, "loss": 0.007, "num_tokens": 160003548.0, "reward": 0.6843750476837158, "reward_std": 1.021040916442871, "rewards/code_complexity_reward/mean": 0.28496092557907104, "rewards/code_complexity_reward/std": 0.4121662676334381, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.1630859375, "rewards/code_syntax_reward/std": 0.23463475704193115, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 498, "step_time": 45.819024585187435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.912109375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 505.990234375, "completions/mean_terminated_length": 443.6222229003906, "completions/min_length": 340.0, "completions/min_terminated_length": 340.0, "entropy": 0.43382127955555916, "epoch": 0.5689851767388826, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.01048598438501358, "kl": 0.02428730091196485, "learning_rate": 2.3458055160383055e-06, "loss": 0.0049, "num_tokens": 160321619.0, "reward": 0.5235351920127869, "reward_std": 0.9321019053459167, "rewards/code_complexity_reward/mean": 0.21884767711162567, "rewards/code_complexity_reward/std": 0.37777185440063477, "rewards/code_execution_reward/mean": 0.177734375, "rewards/code_execution_reward/std": 0.3826628625392914, "rewards/code_syntax_reward/mean": 0.126953125, "rewards/code_syntax_reward/std": 0.21783512830734253, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 499, "step_time": 45.4949767999351 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.880859375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 503.515625, "completions/mean_terminated_length": 440.786865234375, "completions/min_length": 290.0, "completions/min_terminated_length": 290.0, "entropy": 0.42992085916921496, "epoch": 0.5701254275940707, "frac_reward_zero_std": 0.484375, "grad_norm": 0.01238776370882988, "kl": 0.024364841199712828, "learning_rate": 2.33587136722413e-06, "loss": 0.007, "num_tokens": 160639043.0, "reward": 0.6232910752296448, "reward_std": 0.9867447018623352, "rewards/code_complexity_reward/mean": 0.26171875, "rewards/code_complexity_reward/std": 0.4014764428138733, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.150390625, "rewards/code_syntax_reward/std": 0.22952312231063843, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 500, "step_time": 44.96433504857123 }, { "epoch": 0.5701254275940707, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.9275, "eval_completions/max_length": 512.0, "eval_completions/max_terminated_length": 166.44, "eval_completions/mean_length": 507.39, "eval_completions/mean_terminated_length": 162.38333374023438, "eval_completions/min_length": 486.14, "eval_completions/min_terminated_length": 158.46, "eval_entropy": 0.44661502957344057, "eval_frac_reward_zero_std": 0.48, "eval_kl": 0.024821587949991227, "eval_loss": 0.00714588537812233, "eval_num_tokens": 160639043.0, "eval_reward": 0.4865000069141388, "eval_reward_std": 0.7111800849437714, "eval_rewards/code_complexity_reward/mean": 0.20899999842047692, "eval_rewards/code_complexity_reward/std": 0.3048744648694992, "eval_rewards/code_execution_reward/mean": 0.155, "eval_rewards/code_execution_reward/std": 0.2483835107088089, "eval_rewards/code_syntax_reward/mean": 0.1225, "eval_rewards/code_syntax_reward/std": 0.18062446832656862, "eval_rewards/reasoning_present_reward_func/mean": 0.0, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.0, "eval_rewards/xmlcount_reward_func/std": 0.0, "eval_runtime": 1249.4501, "eval_samples_per_second": 0.08, "eval_steps_per_second": 0.01, "step": 500 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 503.732421875, "completions/mean_terminated_length": 430.5961608886719, "completions/min_length": 241.0, "completions/min_terminated_length": 241.0, "entropy": 0.44744650973007083, "epoch": 0.5712656784492588, "frac_reward_zero_std": 0.578125, "grad_norm": 0.009415715001523495, "kl": 0.025177775969495997, "learning_rate": 2.325939820543114e-06, "loss": 0.0052, "num_tokens": 160957866.0, "reward": 0.49658203125, "reward_std": 0.9061288833618164, "rewards/code_complexity_reward/mean": 0.21337890625, "rewards/code_complexity_reward/std": 0.3763062357902527, "rewards/code_execution_reward/mean": 0.16015625, "rewards/code_execution_reward/std": 0.3671095669269562, "rewards/code_syntax_reward/mean": 0.123046875, "rewards/code_syntax_reward/std": 0.21557752788066864, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 501, "step_time": 45.0336176212877 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.912109375, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 505.98828125, "completions/mean_terminated_length": 443.6000061035156, "completions/min_length": 289.0, "completions/min_terminated_length": 289.0, "entropy": 0.428849330637604, "epoch": 0.572405929304447, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.011687872931361198, "kl": 0.024521583720343187, "learning_rate": 2.3160110334522864e-06, "loss": 0.0047, "num_tokens": 161275368.0, "reward": 0.5042968988418579, "reward_std": 0.9197182655334473, "rewards/code_complexity_reward/mean": 0.21035155653953552, "rewards/code_complexity_reward/std": 0.3728824853897095, "rewards/code_execution_reward/mean": 0.171875, "rewards/code_execution_reward/std": 0.3776407241821289, "rewards/code_syntax_reward/mean": 0.1220703125, "rewards/code_syntax_reward/std": 0.21499831974506378, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 502, "step_time": 45.234064505435526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9140625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 505.44921875, "completions/mean_terminated_length": 435.7727355957031, "completions/min_length": 345.0, "completions/min_terminated_length": 345.0, "entropy": 0.44329947559162974, "epoch": 0.5735461801596351, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.011789603158831596, "kl": 0.02447249778197147, "learning_rate": 2.3060851633649246e-06, "loss": 0.0037, "num_tokens": 161593842.0, "reward": 0.5130859613418579, "reward_std": 0.893417239189148, "rewards/code_complexity_reward/mean": 0.22988280653953552, "rewards/code_complexity_reward/std": 0.38465625047683716, "rewards/code_execution_reward/mean": 0.150390625, "rewards/code_execution_reward/std": 0.35780346393585205, "rewards/code_syntax_reward/mean": 0.1328125, "rewards/code_syntax_reward/std": 0.22104869782924652, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 503, "step_time": 45.56222769245505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.892578125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 504.056640625, "completions/mean_terminated_length": 438.0545349121094, "completions/min_length": 282.0, "completions/min_terminated_length": 282.0, "entropy": 0.4335947157815099, "epoch": 0.5746864310148233, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.011525403708219528, "kl": 0.0253441609966103, "learning_rate": 2.296162367648061e-06, "loss": 0.0033, "num_tokens": 161912619.0, "reward": 0.6187499761581421, "reward_std": 0.9724382758140564, "rewards/code_complexity_reward/mean": 0.2642577886581421, "rewards/code_complexity_reward/std": 0.40067434310913086, "rewards/code_execution_reward/mean": 0.201171875, "rewards/code_execution_reward/std": 0.4012683033943176, "rewards/code_syntax_reward/mean": 0.1533203125, "rewards/code_syntax_reward/std": 0.2307749092578888, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 504, "step_time": 46.358739640563726 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.857421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 499.337890625, "completions/mean_terminated_length": 423.1917724609375, "completions/min_length": 175.0, "completions/min_terminated_length": 175.0, "entropy": 0.4332830458879471, "epoch": 0.5758266818700114, "frac_reward_zero_std": 0.546875, "grad_norm": 0.010040322318673134, "kl": 0.02581785959773697, "learning_rate": 2.2862428036199834e-06, "loss": 0.0055, "num_tokens": 162227828.0, "reward": 0.6496093273162842, "reward_std": 1.0016424655914307, "rewards/code_complexity_reward/mean": 0.27656251192092896, "rewards/code_complexity_reward/std": 0.41209396719932556, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.15625, "rewards/code_syntax_reward/std": 0.23198285698890686, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 505, "step_time": 51.452862865291536 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.931640625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 507.673828125, "completions/mean_terminated_length": 448.71429443359375, "completions/min_length": 324.0, "completions/min_terminated_length": 324.0, "entropy": 0.42866856837645173, "epoch": 0.5769669327251995, "frac_reward_zero_std": 0.46875, "grad_norm": 0.012234801426529884, "kl": 0.02478788373991847, "learning_rate": 2.2763266285477476e-06, "loss": 0.0043, "num_tokens": 162546453.0, "reward": 0.4999023377895355, "reward_std": 0.9068045616149902, "rewards/code_complexity_reward/mean": 0.21181640028953552, "rewards/code_complexity_reward/std": 0.37241214513778687, "rewards/code_execution_reward/mean": 0.1640625, "rewards/code_execution_reward/std": 0.37069445848464966, "rewards/code_syntax_reward/mean": 0.1240234375, "rewards/code_syntax_reward/std": 0.21615077555179596, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 506, "step_time": 50.52902011293918 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.88671875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 501.998046875, "completions/mean_terminated_length": 423.7069091796875, "completions/min_length": 261.0, "completions/min_terminated_length": 261.0, "entropy": 0.422852732706815, "epoch": 0.5781071835803877, "frac_reward_zero_std": 0.4375, "grad_norm": 0.011903177946805954, "kl": 0.02689050225308165, "learning_rate": 2.2664139996446756e-06, "loss": 0.0084, "num_tokens": 162863148.0, "reward": 0.6583007574081421, "reward_std": 0.9952796697616577, "rewards/code_complexity_reward/mean": 0.2803710997104645, "rewards/code_complexity_reward/std": 0.40957289934158325, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.1611328125, "rewards/code_syntax_reward/std": 0.23390056192874908, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 507, "step_time": 50.72891462687403 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.888671875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 504.19921875, "completions/mean_terminated_length": 441.9298400878906, "completions/min_length": 291.0, "completions/min_terminated_length": 291.0, "entropy": 0.4317715698853135, "epoch": 0.5792474344355758, "frac_reward_zero_std": 0.453125, "grad_norm": 0.012801013886928558, "kl": 0.02543744698050432, "learning_rate": 2.256505074067872e-06, "loss": 0.0055, "num_tokens": 163182266.0, "reward": 0.6197265386581421, "reward_std": 0.9542425274848938, "rewards/code_complexity_reward/mean": 0.2681640386581421, "rewards/code_complexity_reward/std": 0.39746278524398804, "rewards/code_execution_reward/mean": 0.193359375, "rewards/code_execution_reward/std": 0.39531853795051575, "rewards/code_syntax_reward/mean": 0.158203125, "rewards/code_syntax_reward/std": 0.23276415467262268, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 508, "step_time": 45.45836135651916 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.873046875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 502.5859375, "completions/mean_terminated_length": 437.8461608886719, "completions/min_length": 291.0, "completions/min_terminated_length": 291.0, "entropy": 0.43445646297186613, "epoch": 0.580387685290764, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.012521771714091301, "kl": 0.02550555521156639, "learning_rate": 2.246600008915724e-06, "loss": 0.0044, "num_tokens": 163497390.0, "reward": 0.61376953125, "reward_std": 0.9607517123222351, "rewards/code_complexity_reward/mean": 0.26416015625, "rewards/code_complexity_reward/std": 0.3980747163295746, "rewards/code_execution_reward/mean": 0.1953125, "rewards/code_execution_reward/std": 0.3968288004398346, "rewards/code_syntax_reward/mean": 0.154296875, "rewards/code_syntax_reward/std": 0.23118239641189575, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 509, "step_time": 45.871860740706325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.88671875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 504.37109375, "completions/mean_terminated_length": 444.6551818847656, "completions/min_length": 299.0, "completions/min_terminated_length": 299.0, "entropy": 0.4227089728228748, "epoch": 0.5815279361459521, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.012768099084496498, "kl": 0.02603916815132834, "learning_rate": 2.236698961225417e-06, "loss": 0.0081, "num_tokens": 163815808.0, "reward": 0.6167969107627869, "reward_std": 0.965602695941925, "rewards/code_complexity_reward/mean": 0.2671874761581421, "rewards/code_complexity_reward/std": 0.4032251536846161, "rewards/code_execution_reward/mean": 0.1953125, "rewards/code_execution_reward/std": 0.3968288004398346, "rewards/code_syntax_reward/mean": 0.154296875, "rewards/code_syntax_reward/std": 0.23118239641189575, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 510, "step_time": 46.10985985118896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9140625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 506.41015625, "completions/mean_terminated_length": 446.9545593261719, "completions/min_length": 349.0, "completions/min_terminated_length": 349.0, "entropy": 0.4326815605163574, "epoch": 0.5826681870011402, "frac_reward_zero_std": 0.453125, "grad_norm": 0.011142152361571789, "kl": 0.0265097442897968, "learning_rate": 2.226802087970444e-06, "loss": 0.0055, "num_tokens": 164135010.0, "reward": 0.58740234375, "reward_std": 0.9362416863441467, "rewards/code_complexity_reward/mean": 0.26025390625, "rewards/code_complexity_reward/std": 0.39837896823883057, "rewards/code_execution_reward/mean": 0.17578125, "rewards/code_execution_reward/std": 0.3810062110424042, "rewards/code_syntax_reward/mean": 0.1513671875, "rewards/code_syntax_reward/std": 0.22994530200958252, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 511, "step_time": 49.97749787848443 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.888671875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 503.3515625, "completions/mean_terminated_length": 434.3157958984375, "completions/min_length": 220.0, "completions/min_terminated_length": 220.0, "entropy": 0.43456385051831603, "epoch": 0.5838084378563284, "frac_reward_zero_std": 0.421875, "grad_norm": 0.013628631830215454, "kl": 0.0261387656792067, "learning_rate": 2.2169095460581116e-06, "loss": 0.0064, "num_tokens": 164453046.0, "reward": 0.5769531726837158, "reward_std": 0.8952998518943787, "rewards/code_complexity_reward/mean": 0.27128908038139343, "rewards/code_complexity_reward/std": 0.40106987953186035, "rewards/code_execution_reward/mean": 0.146484375, "rewards/code_execution_reward/std": 0.35393697023391724, "rewards/code_syntax_reward/mean": 0.1591796875, "rewards/code_syntax_reward/std": 0.23314768075942993, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 512, "step_time": 45.40715577173978 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 505.890625, "completions/mean_terminated_length": 451.8461608886719, "completions/min_length": 362.0, "completions/min_terminated_length": 362.0, "entropy": 0.42053159791976213, "epoch": 0.5849486887115165, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.011841727420687675, "kl": 0.02703081027721055, "learning_rate": 2.2070214923270604e-06, "loss": 0.0016, "num_tokens": 164770822.0, "reward": 0.623730480670929, "reward_std": 0.9998083114624023, "rewards/code_complexity_reward/mean": 0.25458985567092896, "rewards/code_complexity_reward/std": 0.398030161857605, "rewards/code_execution_reward/mean": 0.22265625, "rewards/code_execution_reward/std": 0.41643625497817993, "rewards/code_syntax_reward/mean": 0.146484375, "rewards/code_syntax_reward/std": 0.227784663438797, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 513, "step_time": 45.62266534939408 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.890625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 502.138671875, "completions/mean_terminated_length": 421.83929443359375, "completions/min_length": 273.0, "completions/min_terminated_length": 273.0, "entropy": 0.42803090438246727, "epoch": 0.5860889395667047, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.011155565269291401, "kl": 0.028289320529438555, "learning_rate": 2.197138083544771e-06, "loss": 0.0059, "num_tokens": 165085577.0, "reward": 0.611132800579071, "reward_std": 0.9679669141769409, "rewards/code_complexity_reward/mean": 0.26054686307907104, "rewards/code_complexity_reward/std": 0.3983897268772125, "rewards/code_execution_reward/mean": 0.19921875, "rewards/code_execution_reward/std": 0.39980348944664, "rewards/code_syntax_reward/mean": 0.1513671875, "rewards/code_syntax_reward/std": 0.22994530200958252, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 514, "step_time": 45.878487566486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.88671875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 503.279296875, "completions/mean_terminated_length": 435.0172424316406, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.41720019560307264, "epoch": 0.5872291904218928, "frac_reward_zero_std": 0.4375, "grad_norm": 0.0127190463244915, "kl": 0.027696148958057165, "learning_rate": 2.1872594764050835e-06, "loss": 0.0058, "num_tokens": 165402420.0, "reward": 0.6750977039337158, "reward_std": 1.0008851289749146, "rewards/code_complexity_reward/mean": 0.28056639432907104, "rewards/code_complexity_reward/std": 0.40180400013923645, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.166015625, "rewards/code_syntax_reward/std": 0.23570136725902557, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 515, "step_time": 60.388084484264255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.869140625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 502.400390625, "completions/mean_terminated_length": 438.64178466796875, "completions/min_length": 313.0, "completions/min_terminated_length": 313.0, "entropy": 0.4381196345202625, "epoch": 0.5883694412770809, "frac_reward_zero_std": 0.46875, "grad_norm": 0.011461460031569004, "kl": 0.025944092514691874, "learning_rate": 2.177385827525712e-06, "loss": 0.0019, "num_tokens": 165719953.0, "reward": 0.6136718988418579, "reward_std": 0.9641309976577759, "rewards/code_complexity_reward/mean": 0.2679687440395355, "rewards/code_complexity_reward/std": 0.40459612011909485, "rewards/code_execution_reward/mean": 0.19140625, "rewards/code_execution_reward/std": 0.3937928080558777, "rewards/code_syntax_reward/mean": 0.154296875, "rewards/code_syntax_reward/std": 0.23118239641189575, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 516, "step_time": 46.160646656528115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.880859375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 502.4921875, "completions/mean_terminated_length": 432.1966857910156, "completions/min_length": 221.0, "completions/min_terminated_length": 221.0, "entropy": 0.42478609178215265, "epoch": 0.5895096921322691, "frac_reward_zero_std": 0.390625, "grad_norm": 0.012582962401211262, "kl": 0.02828111802227795, "learning_rate": 2.16751729344576e-06, "loss": 0.0049, "num_tokens": 166035045.0, "reward": 0.630566418170929, "reward_std": 0.9675752520561218, "rewards/code_complexity_reward/mean": 0.27216795086860657, "rewards/code_complexity_reward/std": 0.4019300043582916, "rewards/code_execution_reward/mean": 0.19921875, "rewards/code_execution_reward/std": 0.39980348944664, "rewards/code_syntax_reward/mean": 0.1591796875, "rewards/code_syntax_reward/std": 0.23314768075942993, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 517, "step_time": 54.36546165961772 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.849609375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 498.923828125, "completions/mean_terminated_length": 425.05194091796875, "completions/min_length": 251.0, "completions/min_terminated_length": 251.0, "entropy": 0.4206453408114612, "epoch": 0.5906499429874572, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.0126181086525321, "kl": 0.02816488570533693, "learning_rate": 2.1576540306232418e-06, "loss": 0.0086, "num_tokens": 166348446.0, "reward": 0.6553223133087158, "reward_std": 0.9777951836585999, "rewards/code_complexity_reward/mean": 0.28984373807907104, "rewards/code_complexity_reward/std": 0.4152362048625946, "rewards/code_execution_reward/mean": 0.19921875, "rewards/code_execution_reward/std": 0.39980348944664, "rewards/code_syntax_reward/mean": 0.166015625, "rewards/code_syntax_reward/std": 0.23570136725902557, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 518, "step_time": 45.079562787897885 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 498.470703125, "completions/mean_terminated_length": 425.4125061035156, "completions/min_length": 294.0, "completions/min_terminated_length": 294.0, "entropy": 0.4269517147913575, "epoch": 0.5917901938426454, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.012929542921483517, "kl": 0.02924390500993468, "learning_rate": 2.147796195432597e-06, "loss": 0.0047, "num_tokens": 166662767.0, "reward": 0.6885741949081421, "reward_std": 0.9847373962402344, "rewards/code_complexity_reward/mean": 0.3057616949081421, "rewards/code_complexity_reward/std": 0.4198138117790222, "rewards/code_execution_reward/mean": 0.20703125, "rewards/code_execution_reward/std": 0.40557438135147095, "rewards/code_syntax_reward/mean": 0.17578125, "rewards/code_syntax_reward/std": 0.2389625608921051, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 519, "step_time": 51.18101944308728 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.890625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 502.8671875, "completions/mean_terminated_length": 428.5000305175781, "completions/min_length": 279.0, "completions/min_terminated_length": 279.0, "entropy": 0.431795054115355, "epoch": 0.5929304446978335, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.011474580504000187, "kl": 0.028476029430748895, "learning_rate": 2.1379439441622183e-06, "loss": 0.007, "num_tokens": 166979715.0, "reward": 0.5987304449081421, "reward_std": 0.9735158681869507, "rewards/code_complexity_reward/mean": 0.2520507872104645, "rewards/code_complexity_reward/std": 0.39691901206970215, "rewards/code_execution_reward/mean": 0.201171875, "rewards/code_execution_reward/std": 0.4012683033943176, "rewards/code_syntax_reward/mean": 0.1455078125, "rewards/code_syntax_reward/std": 0.22733746469020844, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 520, "step_time": 46.35220223199576 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.916015625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 504.689453125, "completions/mean_terminated_length": 424.9534912109375, "completions/min_length": 277.0, "completions/min_terminated_length": 277.0, "entropy": 0.42455752193927765, "epoch": 0.5940706955530216, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.01022822130471468, "kl": 0.02749238593969494, "learning_rate": 2.1280974330119647e-06, "loss": 0.0066, "num_tokens": 167298632.0, "reward": 0.550488293170929, "reward_std": 0.9342288374900818, "rewards/code_complexity_reward/mean": 0.23505859076976776, "rewards/code_complexity_reward/std": 0.384784460067749, "rewards/code_execution_reward/mean": 0.177734375, "rewards/code_execution_reward/std": 0.3826628625392914, "rewards/code_syntax_reward/mean": 0.1376953125, "rewards/code_syntax_reward/std": 0.22357389330863953, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 521, "step_time": 59.127631878480315 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84765625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 500.0234375, "completions/mean_terminated_length": 433.3846130371094, "completions/min_length": 289.0, "completions/min_terminated_length": 289.0, "entropy": 0.4144504372961819, "epoch": 0.5952109464082098, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.013438702560961246, "kl": 0.030663086159620434, "learning_rate": 2.1182568180906947e-06, "loss": 0.009, "num_tokens": 167612712.0, "reward": 0.7227538824081421, "reward_std": 1.0121642351150513, "rewards/code_complexity_reward/mean": 0.3106445372104645, "rewards/code_complexity_reward/std": 0.4178139269351959, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.1796875, "rewards/code_syntax_reward/std": 0.24014326930046082, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 522, "step_time": 47.049015450291336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.869140625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 500.92578125, "completions/mean_terminated_length": 427.3731384277344, "completions/min_length": 250.0, "completions/min_terminated_length": 250.0, "entropy": 0.41472879238426685, "epoch": 0.5963511972633979, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.012379328720271587, "kl": 0.029598495515529066, "learning_rate": 2.108422255413782e-06, "loss": 0.0089, "num_tokens": 167927966.0, "reward": 0.6349608898162842, "reward_std": 0.9545638561248779, "rewards/code_complexity_reward/mean": 0.28242188692092896, "rewards/code_complexity_reward/std": 0.4063026010990143, "rewards/code_execution_reward/mean": 0.1875, "rewards/code_execution_reward/std": 0.39069411158561707, "rewards/code_syntax_reward/mean": 0.1650390625, "rewards/code_syntax_reward/std": 0.23535043001174927, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 523, "step_time": 51.08203738555312 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.87890625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 503.46875, "completions/mean_terminated_length": 441.5483703613281, "completions/min_length": 295.0, "completions/min_terminated_length": 295.0, "entropy": 0.42958433367311954, "epoch": 0.5974914481185861, "frac_reward_zero_std": 0.4375, "grad_norm": 0.012469903565943241, "kl": 0.030194835417205468, "learning_rate": 2.0985939009006506e-06, "loss": 0.003, "num_tokens": 168244706.0, "reward": 0.651074230670929, "reward_std": 0.9566721320152283, "rewards/code_complexity_reward/mean": 0.28662109375, "rewards/code_complexity_reward/std": 0.40292680263519287, "rewards/code_execution_reward/mean": 0.193359375, "rewards/code_execution_reward/std": 0.39531853795051575, "rewards/code_syntax_reward/mean": 0.169921875, "rewards/code_syntax_reward/std": 0.2370595932006836, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 524, "step_time": 55.54540235828608 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.900390625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 503.630859375, "completions/mean_terminated_length": 427.98040771484375, "completions/min_length": 223.0, "completions/min_terminated_length": 223.0, "entropy": 0.4322591037489474, "epoch": 0.5986316989737742, "frac_reward_zero_std": 0.453125, "grad_norm": 0.011657224036753178, "kl": 0.0293323143851012, "learning_rate": 2.0887719103722987e-06, "loss": 0.0056, "num_tokens": 168561357.0, "reward": 0.552734375, "reward_std": 0.901288628578186, "rewards/code_complexity_reward/mean": 0.2509765625, "rewards/code_complexity_reward/std": 0.39099329710006714, "rewards/code_execution_reward/mean": 0.154296875, "rewards/code_execution_reward/std": 0.36158639192581177, "rewards/code_syntax_reward/mean": 0.1474609375, "rewards/code_syntax_reward/std": 0.22822681069374084, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 525, "step_time": 45.61505167558789 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.87890625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 502.001953125, "completions/mean_terminated_length": 429.43548583984375, "completions/min_length": 249.0, "completions/min_terminated_length": 249.0, "entropy": 0.43030738877132535, "epoch": 0.5997719498289624, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.013489986769855022, "kl": 0.031169565016170964, "learning_rate": 2.0789564395488252e-06, "loss": 0.0058, "num_tokens": 168879286.0, "reward": 0.638134777545929, "reward_std": 0.9748837947845459, "rewards/code_complexity_reward/mean": 0.27265626192092896, "rewards/code_complexity_reward/std": 0.40206703543663025, "rewards/code_execution_reward/mean": 0.205078125, "rewards/code_execution_reward/std": 0.4041535556316376, "rewards/code_syntax_reward/mean": 0.16015625, "rewards/code_syntax_reward/std": 0.23352646827697754, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 526, "step_time": 60.61659603379667 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.849609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 500.21875, "completions/mean_terminated_length": 433.6623229980469, "completions/min_length": 242.0, "completions/min_terminated_length": 242.0, "entropy": 0.42824889393523335, "epoch": 0.6009122006841505, "frac_reward_zero_std": 0.46875, "grad_norm": 0.012596562504768372, "kl": 0.030410292907617986, "learning_rate": 2.0691476440469674e-06, "loss": 0.0057, "num_tokens": 169194530.0, "reward": 0.658203125, "reward_std": 0.9895941615104675, "rewards/code_complexity_reward/mean": 0.2802734375, "rewards/code_complexity_reward/std": 0.40608328580856323, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.1630859375, "rewards/code_syntax_reward/std": 0.23463475704193115, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 527, "step_time": 55.03361554071307 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.85546875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 500.8203125, "completions/mean_terminated_length": 434.6486511230469, "completions/min_length": 262.0, "completions/min_terminated_length": 262.0, "entropy": 0.4324890370480716, "epoch": 0.6020524515393386, "frac_reward_zero_std": 0.484375, "grad_norm": 0.01229864452034235, "kl": 0.031256136688170955, "learning_rate": 2.059345679377627e-06, "loss": 0.0072, "num_tokens": 169511894.0, "reward": 0.6341797113418579, "reward_std": 0.9685534238815308, "rewards/code_complexity_reward/mean": 0.2728515863418579, "rewards/code_complexity_reward/std": 0.4009959101676941, "rewards/code_execution_reward/mean": 0.201171875, "rewards/code_execution_reward/std": 0.4012683033943176, "rewards/code_syntax_reward/mean": 0.16015625, "rewards/code_syntax_reward/std": 0.23352646827697754, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 528, "step_time": 57.02378184814006 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.853515625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 498.3984375, "completions/mean_terminated_length": 419.14666748046875, "completions/min_length": 245.0, "completions/min_terminated_length": 245.0, "entropy": 0.4339269855991006, "epoch": 0.6031927023945268, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.014228380285203457, "kl": 0.029714831442106515, "learning_rate": 2.0495507009434127e-06, "loss": 0.0103, "num_tokens": 169827754.0, "reward": 0.7176758050918579, "reward_std": 1.006186842918396, "rewards/code_complexity_reward/mean": 0.3114257752895355, "rewards/code_complexity_reward/std": 0.41907933354377747, "rewards/code_execution_reward/mean": 0.2265625, "rewards/code_execution_reward/std": 0.4190165400505066, "rewards/code_syntax_reward/mean": 0.1796875, "rewards/code_syntax_reward/std": 0.24014326930046082, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 529, "step_time": 54.86728748586029 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.90234375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 505.228515625, "completions/mean_terminated_length": 442.6600036621094, "completions/min_length": 326.0, "completions/min_terminated_length": 326.0, "entropy": 0.4262539395131171, "epoch": 0.6043329532497149, "frac_reward_zero_std": 0.421875, "grad_norm": 0.012209487147629261, "kl": 0.03126579275703989, "learning_rate": 2.0397628640361674e-06, "loss": 0.0057, "num_tokens": 170146063.0, "reward": 0.60888671875, "reward_std": 0.962762713432312, "rewards/code_complexity_reward/mean": 0.26123046875, "rewards/code_complexity_reward/std": 0.3980945646762848, "rewards/code_execution_reward/mean": 0.1953125, "rewards/code_execution_reward/std": 0.3968288004398346, "rewards/code_syntax_reward/mean": 0.15234375, "rewards/code_syntax_reward/std": 0.23036254942417145, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 530, "step_time": 54.43981505185366 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8515625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 500.669921875, "completions/mean_terminated_length": 435.6710510253906, "completions/min_length": 298.0, "completions/min_terminated_length": 298.0, "entropy": 0.41956845996901393, "epoch": 0.6054732041049031, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.013035438023507595, "kl": 0.03208161462680437, "learning_rate": 2.0299823238345125e-06, "loss": 0.006, "num_tokens": 170463714.0, "reward": 0.7061035633087158, "reward_std": 1.0000944137573242, "rewards/code_complexity_reward/mean": 0.30351561307907104, "rewards/code_complexity_reward/std": 0.4125920832157135, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.177734375, "rewards/code_syntax_reward/std": 0.23956161737442017, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 531, "step_time": 51.73188875056803 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.87890625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 502.607421875, "completions/mean_terminated_length": 434.43548583984375, "completions/min_length": 280.0, "completions/min_terminated_length": 280.0, "entropy": 0.4238056279718876, "epoch": 0.6066134549600912, "frac_reward_zero_std": 0.34375, "grad_norm": 0.014115573838353157, "kl": 0.030908804532373324, "learning_rate": 2.0202092354013885e-06, "loss": 0.0049, "num_tokens": 170781633.0, "reward": 0.703906238079071, "reward_std": 0.9865060448646545, "rewards/code_complexity_reward/mean": 0.3082031011581421, "rewards/code_complexity_reward/std": 0.4118013083934784, "rewards/code_execution_reward/mean": 0.212890625, "rewards/code_execution_reward/std": 0.409751296043396, "rewards/code_syntax_reward/mean": 0.181640625, "rewards/code_syntax_reward/std": 0.2407076209783554, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 532, "step_time": 45.39113878738135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.80859375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 496.8671875, "completions/mean_terminated_length": 432.93878173828125, "completions/min_length": 272.0, "completions/min_terminated_length": 272.0, "entropy": 0.4156722682528198, "epoch": 0.6077537058152793, "frac_reward_zero_std": 0.359375, "grad_norm": 0.012284796684980392, "kl": 0.033654346043476835, "learning_rate": 2.0104437536815884e-06, "loss": 0.0076, "num_tokens": 171095093.0, "reward": 0.8067383170127869, "reward_std": 1.0239206552505493, "rewards/code_complexity_reward/mean": 0.3555663824081421, "rewards/code_complexity_reward/std": 0.4299948811531067, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.205078125, "rewards/code_syntax_reward/std": 0.24617145955562592, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 533, "step_time": 54.37236913945526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.859375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 500.69921875, "completions/mean_terminated_length": 431.6388854980469, "completions/min_length": 288.0, "completions/min_terminated_length": 288.0, "entropy": 0.4249996575526893, "epoch": 0.6088939566704675, "frac_reward_zero_std": 0.453125, "grad_norm": 0.011213117279112339, "kl": 0.03188471490284428, "learning_rate": 2.0006860334993105e-06, "loss": 0.0091, "num_tokens": 171411635.0, "reward": 0.6670898199081421, "reward_std": 0.9798887372016907, "rewards/code_complexity_reward/mean": 0.2911132574081421, "rewards/code_complexity_reward/std": 0.41073429584503174, "rewards/code_execution_reward/mean": 0.20703125, "rewards/code_execution_reward/std": 0.40557438135147095, "rewards/code_syntax_reward/mean": 0.1689453125, "rewards/code_syntax_reward/std": 0.2367268204689026, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 534, "step_time": 46.47598792798817 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8515625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 499.09765625, "completions/mean_terminated_length": 425.0789489746094, "completions/min_length": 286.0, "completions/min_terminated_length": 286.0, "entropy": 0.42160375276580453, "epoch": 0.6100342075256556, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.013458729721605778, "kl": 0.03244117327267304, "learning_rate": 1.990936229555697e-06, "loss": 0.0083, "num_tokens": 171725845.0, "reward": 0.6915038824081421, "reward_std": 0.9915450215339661, "rewards/code_complexity_reward/mean": 0.3018554747104645, "rewards/code_complexity_reward/std": 0.4153859317302704, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.1748046875, "rewards/code_syntax_reward/std": 0.23865646123886108, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 535, "step_time": 52.375270588323474 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.884765625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 502.9453125, "completions/mean_terminated_length": 433.4237365722656, "completions/min_length": 298.0, "completions/min_terminated_length": 298.0, "entropy": 0.41808445239439607, "epoch": 0.6111744583808438, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.012454884126782417, "kl": 0.032733062631450593, "learning_rate": 1.981194496426389e-06, "loss": 0.0079, "num_tokens": 172042073.0, "reward": 0.5824218988418579, "reward_std": 0.9398409128189087, "rewards/code_complexity_reward/mean": 0.2523437440395355, "rewards/code_complexity_reward/std": 0.3915501534938812, "rewards/code_execution_reward/mean": 0.181640625, "rewards/code_execution_reward/std": 0.38592514395713806, "rewards/code_syntax_reward/mean": 0.1484375, "rewards/code_syntax_reward/std": 0.22866390645503998, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 536, "step_time": 45.814681502990425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.837890625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 499.1171875, "completions/mean_terminated_length": 432.53009033203125, "completions/min_length": 273.0, "completions/min_terminated_length": 273.0, "entropy": 0.4151640455238521, "epoch": 0.6123147092360319, "frac_reward_zero_std": 0.390625, "grad_norm": 0.013134843669831753, "kl": 0.0333511084318161, "learning_rate": 1.971460988559065e-06, "loss": 0.0077, "num_tokens": 172356249.0, "reward": 0.804492175579071, "reward_std": 1.0363545417785645, "rewards/code_complexity_reward/mean": 0.33964842557907104, "rewards/code_complexity_reward/std": 0.4211510121822357, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.19921875, "rewards/code_syntax_reward/std": 0.2450276017189026, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 537, "step_time": 52.810523516498506 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.845703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 498.880859375, "completions/mean_terminated_length": 426.9747009277344, "completions/min_length": 219.0, "completions/min_terminated_length": 219.0, "entropy": 0.41437674080953, "epoch": 0.61345496009122, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.013414798304438591, "kl": 0.03197716848808341, "learning_rate": 1.9617358602710034e-06, "loss": 0.0045, "num_tokens": 172673008.0, "reward": 0.7095702886581421, "reward_std": 0.9896361827850342, "rewards/code_complexity_reward/mean": 0.3111328184604645, "rewards/code_complexity_reward/std": 0.4158939719200134, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.181640625, "rewards/code_syntax_reward/std": 0.2407076209783554, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 538, "step_time": 45.721727680414915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.861328125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 497.74609375, "completions/mean_terminated_length": 409.2112731933594, "completions/min_length": 263.0, "completions/min_terminated_length": 263.0, "entropy": 0.41351956874132156, "epoch": 0.6145952109464082, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.013722891919314861, "kl": 0.0339804045506753, "learning_rate": 1.9520192657466286e-06, "loss": 0.0053, "num_tokens": 172987498.0, "reward": 0.687304675579071, "reward_std": 1.0017231702804565, "rewards/code_complexity_reward/mean": 0.29667967557907104, "rewards/code_complexity_reward/std": 0.41653531789779663, "rewards/code_execution_reward/mean": 0.220703125, "rewards/code_execution_reward/std": 0.4151262938976288, "rewards/code_syntax_reward/mean": 0.169921875, "rewards/code_syntax_reward/std": 0.2370595932006836, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 539, "step_time": 50.70924585964531 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.859375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 502.357421875, "completions/mean_terminated_length": 443.4305725097656, "completions/min_length": 288.0, "completions/min_terminated_length": 288.0, "entropy": 0.40738551365211606, "epoch": 0.6157354618015963, "frac_reward_zero_std": 0.34375, "grad_norm": 0.013917393982410431, "kl": 0.0334961810731329, "learning_rate": 1.9423113590350666e-06, "loss": 0.0103, "num_tokens": 173305925.0, "reward": 0.7430176138877869, "reward_std": 0.987723708152771, "rewards/code_complexity_reward/mean": 0.3221679925918579, "rewards/code_complexity_reward/std": 0.41161203384399414, "rewards/code_execution_reward/mean": 0.2265625, "rewards/code_execution_reward/std": 0.4190165400505066, "rewards/code_syntax_reward/mean": 0.193359375, "rewards/code_syntax_reward/std": 0.24373729526996613, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 540, "step_time": 57.578066187910736 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.82421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 496.169921875, "completions/mean_terminated_length": 421.9444580078125, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.41654082015156746, "epoch": 0.6168757126567845, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.012787071987986565, "kl": 0.03429181064711884, "learning_rate": 1.9326122940477102e-06, "loss": 0.0051, "num_tokens": 173619028.0, "reward": 0.76708984375, "reward_std": 1.0147039890289307, "rewards/code_complexity_reward/mean": 0.33447265625, "rewards/code_complexity_reward/std": 0.42323434352874756, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.1943359375, "rewards/code_syntax_reward/std": 0.2439626157283783, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 541, "step_time": 45.503630777820945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.806640625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 495.421875, "completions/mean_terminated_length": 426.26263427734375, "completions/min_length": 241.0, "completions/min_terminated_length": 241.0, "entropy": 0.406856341753155, "epoch": 0.6180159635119726, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.014517105184495449, "kl": 0.03466846639639698, "learning_rate": 1.9229222245557675e-06, "loss": 0.0095, "num_tokens": 173931440.0, "reward": 0.8558593988418579, "reward_std": 1.0618748664855957, "rewards/code_complexity_reward/mean": 0.3578124940395355, "rewards/code_complexity_reward/std": 0.42679059505462646, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.208984375, "rewards/code_syntax_reward/std": 0.24685366451740265, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 542, "step_time": 47.93885596655309 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.86328125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 501.671875, "completions/mean_terminated_length": 436.4571533203125, "completions/min_length": 306.0, "completions/min_terminated_length": 306.0, "entropy": 0.41928902780637145, "epoch": 0.6191562143671607, "frac_reward_zero_std": 0.3359375, "grad_norm": 0.013665441423654556, "kl": 0.03313735325355083, "learning_rate": 1.9132413041878356e-06, "loss": 0.0076, "num_tokens": 174247504.0, "reward": 0.722363293170929, "reward_std": 0.9760802388191223, "rewards/code_complexity_reward/mean": 0.32587888836860657, "rewards/code_complexity_reward/std": 0.41824719309806824, "rewards/code_execution_reward/mean": 0.205078125, "rewards/code_execution_reward/std": 0.4041535556316376, "rewards/code_syntax_reward/mean": 0.19140625, "rewards/code_syntax_reward/std": 0.2432742565870285, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 543, "step_time": 46.686847103759646 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.802734375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 497.111328125, "completions/mean_terminated_length": 436.5247497558594, "completions/min_length": 278.0, "completions/min_terminated_length": 278.0, "entropy": 0.4206189061515033, "epoch": 0.6202964652223489, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.015276198275387287, "kl": 0.035296945861773565, "learning_rate": 1.903569686427454e-06, "loss": 0.0072, "num_tokens": 174560289.0, "reward": 0.8465820550918579, "reward_std": 1.0404127836227417, "rewards/code_complexity_reward/mean": 0.3661132752895355, "rewards/code_complexity_reward/std": 0.4286911189556122, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.212890625, "rewards/code_syntax_reward/std": 0.24747224152088165, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 544, "step_time": 45.241843819618225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.86328125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 500.328125, "completions/mean_terminated_length": 426.6285705566406, "completions/min_length": 256.0, "completions/min_terminated_length": 256.0, "entropy": 0.4254331411793828, "epoch": 0.621436716077537, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.014385128393769264, "kl": 0.034154367400333285, "learning_rate": 1.8939075246106809e-06, "loss": 0.0046, "num_tokens": 174876529.0, "reward": 0.679882824420929, "reward_std": 0.9654985666275024, "rewards/code_complexity_reward/mean": 0.30195313692092896, "rewards/code_complexity_reward/std": 0.40876656770706177, "rewards/code_execution_reward/mean": 0.19921875, "rewards/code_execution_reward/std": 0.39980348944664, "rewards/code_syntax_reward/mean": 0.1787109375, "rewards/code_syntax_reward/std": 0.2398546040058136, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 545, "step_time": 45.28518220037222 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 497.404296875, "completions/mean_terminated_length": 427.0795593261719, "completions/min_length": 265.0, "completions/min_terminated_length": 265.0, "entropy": 0.4205785715021193, "epoch": 0.6225769669327252, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.011881617829203606, "kl": 0.035137498256517574, "learning_rate": 1.8842549719236544e-06, "loss": 0.0074, "num_tokens": 175191508.0, "reward": 0.751416027545929, "reward_std": 1.006239891052246, "rewards/code_complexity_reward/mean": 0.32421875, "rewards/code_complexity_reward/std": 0.4166065752506256, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.19140625, "rewards/code_syntax_reward/std": 0.2432742565870285, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.001220703125, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 546, "step_time": 52.56929983198643 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8671875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 502.2109375, "completions/mean_terminated_length": 438.29412841796875, "completions/min_length": 234.0, "completions/min_terminated_length": 234.0, "entropy": 0.4198590540327132, "epoch": 0.6237172177879133, "frac_reward_zero_std": 0.359375, "grad_norm": 0.011641229502856731, "kl": 0.03326747953542508, "learning_rate": 1.874612181400169e-06, "loss": 0.004, "num_tokens": 175511924.0, "reward": 0.7606444954872131, "reward_std": 0.9902109503746033, "rewards/code_complexity_reward/mean": 0.3358398377895355, "rewards/code_complexity_reward/std": 0.4150539040565491, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.2001953125, "rewards/code_syntax_reward/std": 0.2452283650636673, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 547, "step_time": 45.36724080052227 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84765625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 499.2734375, "completions/mean_terminated_length": 428.4615478515625, "completions/min_length": 297.0, "completions/min_terminated_length": 297.0, "entropy": 0.40784508269280195, "epoch": 0.6248574686431014, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.013475478626787663, "kl": 0.03510148261557333, "learning_rate": 1.8649793059192484e-06, "loss": 0.0062, "num_tokens": 175826432.0, "reward": 0.8329101800918579, "reward_std": 1.0112265348434448, "rewards/code_complexity_reward/mean": 0.3680664300918579, "rewards/code_complexity_reward/std": 0.4241265654563904, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.216796875, "rewards/code_syntax_reward/std": 0.24802762269973755, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 548, "step_time": 45.397717237472534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.80859375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 495.3515625, "completions/mean_terminated_length": 425.0203857421875, "completions/min_length": 229.0, "completions/min_terminated_length": 229.0, "entropy": 0.4089445173740387, "epoch": 0.6259977194982896, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.014746963046491146, "kl": 0.035774739721091464, "learning_rate": 1.8553564982027183e-06, "loss": 0.0095, "num_tokens": 176137176.0, "reward": 0.911914050579071, "reward_std": 1.0644792318344116, "rewards/code_complexity_reward/mean": 0.38457033038139343, "rewards/code_complexity_reward/std": 0.4297477900981903, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.224609375, "rewards/code_syntax_reward/std": 0.2489505261182785, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 549, "step_time": 54.01109248865396 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.826171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 497.953125, "completions/mean_terminated_length": 431.1910095214844, "completions/min_length": 243.0, "completions/min_terminated_length": 243.0, "entropy": 0.4122110051102936, "epoch": 0.6271379703534777, "frac_reward_zero_std": 0.375, "grad_norm": 0.013732296414673328, "kl": 0.034553961479105055, "learning_rate": 1.8457439108127914e-06, "loss": 0.0078, "num_tokens": 176452464.0, "reward": 0.8595702648162842, "reward_std": 1.0327543020248413, "rewards/code_complexity_reward/mean": 0.37324219942092896, "rewards/code_complexity_reward/std": 0.4275113642215729, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.21875, "rewards/code_syntax_reward/std": 0.24828176200389862, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 550, "step_time": 45.061343654990196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.830078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 496.966796875, "completions/mean_terminated_length": 423.52874755859375, "completions/min_length": 296.0, "completions/min_terminated_length": 296.0, "entropy": 0.4138361350633204, "epoch": 0.6282782212086659, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.015602034516632557, "kl": 0.03477684978861362, "learning_rate": 1.8361416961496412e-06, "loss": 0.0051, "num_tokens": 176766959.0, "reward": 0.8000000715255737, "reward_std": 1.0253347158432007, "rewards/code_complexity_reward/mean": 0.34394529461860657, "rewards/code_complexity_reward/std": 0.42149028182029724, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.2021484375, "rewards/code_syntax_reward/std": 0.24561770260334015, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 551, "step_time": 45.80607631150633 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.857421875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 499.75, "completions/mean_terminated_length": 426.0821838378906, "completions/min_length": 253.0, "completions/min_terminated_length": 253.0, "entropy": 0.41572407027706504, "epoch": 0.629418472063854, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.01277999673038721, "kl": 0.03690066290437244, "learning_rate": 1.8265500064489915e-06, "loss": 0.0046, "num_tokens": 177082419.0, "reward": 0.6717773675918579, "reward_std": 0.9802598357200623, "rewards/code_complexity_reward/mean": 0.2840820252895355, "rewards/code_complexity_reward/std": 0.3982907831668854, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.1708984375, "rewards/code_syntax_reward/std": 0.23738788068294525, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 552, "step_time": 52.020502069965005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.810546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 495.7265625, "completions/mean_terminated_length": 426.10308837890625, "completions/min_length": 225.0, "completions/min_terminated_length": 225.0, "entropy": 0.4206100390292704, "epoch": 0.6305587229190421, "frac_reward_zero_std": 0.359375, "grad_norm": 0.013673624955117702, "kl": 0.03657584253232926, "learning_rate": 1.816968993779701e-06, "loss": 0.008, "num_tokens": 177393827.0, "reward": 0.818554699420929, "reward_std": 1.025977373123169, "rewards/code_complexity_reward/mean": 0.35761719942092896, "rewards/code_complexity_reward/std": 0.42654183506965637, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.208984375, "rewards/code_syntax_reward/std": 0.24685366451740265, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 553, "step_time": 55.378645645454526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.826171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 497.69921875, "completions/mean_terminated_length": 429.7303466796875, "completions/min_length": 278.0, "completions/min_terminated_length": 278.0, "entropy": 0.41902238922193646, "epoch": 0.6316989737742303, "frac_reward_zero_std": 0.34375, "grad_norm": 0.012777332216501236, "kl": 0.03511088024242781, "learning_rate": 1.8073988100413515e-06, "loss": 0.0063, "num_tokens": 177707581.0, "reward": 0.7992187738418579, "reward_std": 1.008491039276123, "rewards/code_complexity_reward/mean": 0.3519531190395355, "rewards/code_complexity_reward/std": 0.42276936769485474, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.20703125, "rewards/code_syntax_reward/std": 0.24652054905891418, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 554, "step_time": 44.918695977889 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 496.01171875, "completions/mean_terminated_length": 426.72918701171875, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.4066266482695937, "epoch": 0.6328392246294184, "frac_reward_zero_std": 0.390625, "grad_norm": 0.013109978288412094, "kl": 0.03869361663237214, "learning_rate": 1.7978396069618426e-06, "loss": 0.0093, "num_tokens": 178020099.0, "reward": 0.9091796875, "reward_std": 1.0525310039520264, "rewards/code_complexity_reward/mean": 0.38671875, "rewards/code_complexity_reward/std": 0.42775455117225647, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.2275390625, "rewards/code_syntax_reward/std": 0.2492324709892273, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 555, "step_time": 45.7637859582901 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84765625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 500.87890625, "completions/mean_terminated_length": 439.0, "completions/min_length": 240.0, "completions/min_terminated_length": 240.0, "entropy": 0.408763546962291, "epoch": 0.6339794754846066, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.012821494601666927, "kl": 0.03595470430445857, "learning_rate": 1.7882915360949802e-06, "loss": 0.0056, "num_tokens": 178335677.0, "reward": 0.729736328125, "reward_std": 1.0103384256362915, "rewards/code_complexity_reward/mean": 0.310546875, "rewards/code_complexity_reward/std": 0.4147785007953644, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.1826171875, "rewards/code_syntax_reward/std": 0.24098336696624756, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 556, "step_time": 60.55027406942099 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84765625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 499.15625, "completions/mean_terminated_length": 427.69232177734375, "completions/min_length": 270.0, "completions/min_terminated_length": 270.0, "entropy": 0.41786631383001804, "epoch": 0.6351197263397947, "frac_reward_zero_std": 0.328125, "grad_norm": 0.01453600637614727, "kl": 0.03665581994573586, "learning_rate": 1.7787547488180816e-06, "loss": 0.0118, "num_tokens": 178650221.0, "reward": 0.8133789300918579, "reward_std": 1.0174368619918823, "rewards/code_complexity_reward/mean": 0.3534179925918579, "rewards/code_complexity_reward/std": 0.4228236973285675, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.2080078125, "rewards/code_syntax_reward/std": 0.246689110994339, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 557, "step_time": 54.36704859137535 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.82421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 496.0, "completions/mean_terminated_length": 420.977783203125, "completions/min_length": 197.0, "completions/min_terminated_length": 197.0, "entropy": 0.40936345141381025, "epoch": 0.636259977194983, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.013235640712082386, "kl": 0.0374250732420478, "learning_rate": 1.7692293963295679e-06, "loss": 0.0053, "num_tokens": 178961285.0, "reward": 0.8802734017372131, "reward_std": 1.0319132804870605, "rewards/code_complexity_reward/mean": 0.3822265863418579, "rewards/code_complexity_reward/std": 0.428151398897171, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.224609375, "rewards/code_syntax_reward/std": 0.2489505261182785, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 558, "step_time": 45.28424678836018 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.841796875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 498.119140625, "completions/mean_terminated_length": 424.25927734375, "completions/min_length": 243.0, "completions/min_terminated_length": 243.0, "entropy": 0.41758942883461714, "epoch": 0.637400228050171, "frac_reward_zero_std": 0.359375, "grad_norm": 0.01437465287744999, "kl": 0.03765917691634968, "learning_rate": 1.7597156296465734e-06, "loss": 0.0102, "num_tokens": 179277338.0, "reward": 0.727783203125, "reward_std": 0.9641016721725464, "rewards/code_complexity_reward/mean": 0.330078125, "rewards/code_complexity_reward/std": 0.41548261046409607, "rewards/code_execution_reward/mean": 0.201171875, "rewards/code_execution_reward/std": 0.4012683033943176, "rewards/code_syntax_reward/mean": 0.1962890625, "rewards/code_syntax_reward/std": 0.24440090358257294, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 559, "step_time": 45.317033015191555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.798828125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 493.880859375, "completions/mean_terminated_length": 421.9320373535156, "completions/min_length": 244.0, "completions/min_terminated_length": 244.0, "entropy": 0.40587982814759016, "epoch": 0.6385404789053591, "frac_reward_zero_std": 0.328125, "grad_norm": 0.012786989100277424, "kl": 0.03600735793588683, "learning_rate": 1.7502135996025454e-06, "loss": 0.0029, "num_tokens": 179589173.0, "reward": 0.832958996295929, "reward_std": 1.020894169807434, "rewards/code_complexity_reward/mean": 0.36884766817092896, "rewards/code_complexity_reward/std": 0.43081095814704895, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.2138671875, "rewards/code_syntax_reward/std": 0.2476169914007187, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 560, "step_time": 45.239651705138385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.841796875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 498.466796875, "completions/mean_terminated_length": 426.456787109375, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.4026783867739141, "epoch": 0.6396807297605474, "frac_reward_zero_std": 0.328125, "grad_norm": 0.013646327890455723, "kl": 0.03840630466584116, "learning_rate": 1.7407234568448583e-06, "loss": 0.0082, "num_tokens": 179904640.0, "reward": 0.8270508050918579, "reward_std": 1.0100889205932617, "rewards/code_complexity_reward/mean": 0.3641601502895355, "rewards/code_complexity_reward/std": 0.42067375779151917, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.216796875, "rewards/code_syntax_reward/std": 0.24802762269973755, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 561, "step_time": 53.11595815420151 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 493.955078125, "completions/mean_terminated_length": 432.35345458984375, "completions/min_length": 254.0, "completions/min_terminated_length": 254.0, "entropy": 0.41118399519473314, "epoch": 0.6408209806157354, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.01238078810274601, "kl": 0.036997046845499426, "learning_rate": 1.7312453518324232e-06, "loss": 0.011, "num_tokens": 180217621.0, "reward": 0.893847644329071, "reward_std": 1.0578550100326538, "rewards/code_complexity_reward/mean": 0.38115233182907104, "rewards/code_complexity_reward/std": 0.4314577877521515, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.2216796875, "rewards/code_syntax_reward/std": 0.24863366782665253, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 562, "step_time": 45.490707661025226 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84765625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 497.69140625, "completions/mean_terminated_length": 418.0769348144531, "completions/min_length": 223.0, "completions/min_terminated_length": 223.0, "entropy": 0.4088031002320349, "epoch": 0.6419612314709237, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.014419618993997574, "kl": 0.03755671076942235, "learning_rate": 1.7217794348332989e-06, "loss": 0.0065, "num_tokens": 180531563.0, "reward": 0.8477539420127869, "reward_std": 1.0051237344741821, "rewards/code_complexity_reward/mean": 0.3760742247104645, "rewards/code_complexity_reward/std": 0.4233969748020172, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.2236328125, "rewards/code_syntax_reward/std": 0.2488487958908081, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 563, "step_time": 52.888416537083685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.818359375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 494.57421875, "completions/mean_terminated_length": 416.06451416015625, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "entropy": 0.4237537607550621, "epoch": 0.6431014823261118, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.014225257560610771, "kl": 0.037267218431225047, "learning_rate": 1.712325855922316e-06, "loss": 0.0055, "num_tokens": 180844453.0, "reward": 0.8434571027755737, "reward_std": 1.052878499031067, "rewards/code_complexity_reward/mean": 0.35615235567092896, "rewards/code_complexity_reward/std": 0.4260892868041992, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.2080078125, "rewards/code_syntax_reward/std": 0.246689110994339, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 564, "step_time": 55.62414052337408 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.826171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 498.001953125, "completions/mean_terminated_length": 431.471923828125, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 0.41064254753291607, "epoch": 0.6442417331812998, "frac_reward_zero_std": 0.34375, "grad_norm": 0.013139236718416214, "kl": 0.03846743481699377, "learning_rate": 1.7028847649786907e-06, "loss": 0.0059, "num_tokens": 181159290.0, "reward": 0.76513671875, "reward_std": 0.9819009900093079, "rewards/code_complexity_reward/mean": 0.34033203125, "rewards/code_complexity_reward/std": 0.4135892391204834, "rewards/code_execution_reward/mean": 0.220703125, "rewards/code_execution_reward/std": 0.4151262938976288, "rewards/code_syntax_reward/mean": 0.2041015625, "rewards/code_syntax_reward/std": 0.24599088728427887, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 565, "step_time": 45.9740204680711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.720703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 487.111328125, "completions/mean_terminated_length": 422.8880920410156, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "entropy": 0.3981690681539476, "epoch": 0.645381984036488, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.013121861964464188, "kl": 0.03834986235597171, "learning_rate": 1.6934563116836555e-06, "loss": 0.0038, "num_tokens": 181468863.0, "reward": 1.0522949695587158, "reward_std": 1.0652700662612915, "rewards/code_complexity_reward/mean": 0.45537108182907104, "rewards/code_complexity_reward/std": 0.4373638331890106, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.2626953125, "rewards/code_syntax_reward/std": 0.24992163479328156, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 566, "step_time": 52.44509554840624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.826171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 497.57421875, "completions/mean_terminated_length": 429.01123046875, "completions/min_length": 182.0, "completions/min_terminated_length": 182.0, "entropy": 0.4041154272854328, "epoch": 0.6465222348916762, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.01485420297831297, "kl": 0.03840870116255246, "learning_rate": 1.6840406455180801e-06, "loss": 0.0058, "num_tokens": 181782117.0, "reward": 0.7850097417831421, "reward_std": 1.0013401508331299, "rewards/code_complexity_reward/mean": 0.3414062559604645, "rewards/code_complexity_reward/std": 0.4147292673587799, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.205078125, "rewards/code_syntax_reward/std": 0.24617145955562592, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 567, "step_time": 45.57947148755193 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.830078125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 495.875, "completions/mean_terminated_length": 417.10345458984375, "completions/min_length": 232.0, "completions/min_terminated_length": 232.0, "entropy": 0.41912114061415195, "epoch": 0.6476624857468644, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.014589724130928516, "kl": 0.03876133752055466, "learning_rate": 1.6746379157601062e-06, "loss": 0.0094, "num_tokens": 182094469.0, "reward": 0.8564453721046448, "reward_std": 1.0273116827011108, "rewards/code_complexity_reward/mean": 0.373046875, "rewards/code_complexity_reward/std": 0.4269379675388336, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.2197265625, "rewards/code_syntax_reward/std": 0.24840296804904938, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 568, "step_time": 45.84136444423348 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.794921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 492.478515625, "completions/mean_terminated_length": 416.8095397949219, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.4058277551084757, "epoch": 0.6488027366020525, "frac_reward_zero_std": 0.359375, "grad_norm": 0.013959970325231552, "kl": 0.038590863114222884, "learning_rate": 1.6652482714827783e-06, "loss": 0.0072, "num_tokens": 182406722.0, "reward": 0.9053710699081421, "reward_std": 1.0209667682647705, "rewards/code_complexity_reward/mean": 0.4046875238418579, "rewards/code_complexity_reward/std": 0.43245023488998413, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.236328125, "rewards/code_syntax_reward/std": 0.24987001717090607, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 569, "step_time": 45.95006196387112 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.791015625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 495.736328125, "completions/mean_terminated_length": 434.17755126953125, "completions/min_length": 267.0, "completions/min_terminated_length": 267.0, "entropy": 0.4117262093350291, "epoch": 0.6499429874572406, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.013068152591586113, "kl": 0.039999817905481905, "learning_rate": 1.6558718615516787e-06, "loss": 0.0086, "num_tokens": 182720103.0, "reward": 0.9452148079872131, "reward_std": 1.069489598274231, "rewards/code_complexity_reward/mean": 0.4032226502895355, "rewards/code_complexity_reward/std": 0.43518713116645813, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.2333984375, "rewards/code_syntax_reward/std": 0.2496921271085739, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 570, "step_time": 45.845667785964906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.833984375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 498.3203125, "completions/mean_terminated_length": 429.6000061035156, "completions/min_length": 229.0, "completions/min_terminated_length": 229.0, "entropy": 0.4139084639027715, "epoch": 0.6510832383124288, "frac_reward_zero_std": 0.390625, "grad_norm": 0.012609140016138554, "kl": 0.0358898616686929, "learning_rate": 1.6465088346225719e-06, "loss": 0.0067, "num_tokens": 183035819.0, "reward": 0.8103514909744263, "reward_std": 1.0237016677856445, "rewards/code_complexity_reward/mean": 0.34941405057907104, "rewards/code_complexity_reward/std": 0.42272719740867615, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.205078125, "rewards/code_syntax_reward/std": 0.24617145955562592, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 571, "step_time": 54.214433315210044 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.833984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 500.244140625, "completions/mean_terminated_length": 441.188232421875, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "entropy": 0.4167235726490617, "epoch": 0.6522234891676169, "frac_reward_zero_std": 0.3125, "grad_norm": 0.0140765979886055, "kl": 0.0388961915159598, "learning_rate": 1.6371593391390427e-06, "loss": 0.0058, "num_tokens": 183352984.0, "reward": 0.7889648675918579, "reward_std": 0.9666302800178528, "rewards/code_complexity_reward/mean": 0.3631835877895355, "rewards/code_complexity_reward/std": 0.4196343421936035, "rewards/code_execution_reward/mean": 0.208984375, "rewards/code_execution_reward/std": 0.40698084235191345, "rewards/code_syntax_reward/mean": 0.216796875, "rewards/code_syntax_reward/std": 0.24802762269973755, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 572, "step_time": 50.606528147123754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8046875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 497.716796875, "completions/mean_terminated_length": 438.8699951171875, "completions/min_length": 271.0, "completions/min_terminated_length": 271.0, "entropy": 0.4117675172165036, "epoch": 0.6533637400228051, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.017272410914301872, "kl": 0.039289353822823614, "learning_rate": 1.6278235233301482e-06, "loss": 0.0052, "num_tokens": 183667607.0, "reward": 0.8207031488418579, "reward_std": 1.004433035850525, "rewards/code_complexity_reward/mean": 0.3685547113418579, "rewards/code_complexity_reward/std": 0.42730605602264404, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.2158203125, "rewards/code_syntax_reward/std": 0.24789467453956604, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 573, "step_time": 50.641407035291195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.783203125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 492.19921875, "completions/mean_terminated_length": 420.66668701171875, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.4162778714671731, "epoch": 0.6545039908779932, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.012577803805470467, "kl": 0.03927760769147426, "learning_rate": 1.6185015352080614e-06, "loss": 0.0085, "num_tokens": 183978973.0, "reward": 0.863964855670929, "reward_std": 1.021493673324585, "rewards/code_complexity_reward/mean": 0.38154298067092896, "rewards/code_complexity_reward/std": 0.4272816777229309, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.224609375, "rewards/code_syntax_reward/std": 0.2489505261182785, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 574, "step_time": 53.19447868131101 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 498.4140625, "completions/mean_terminated_length": 425.0500183105469, "completions/min_length": 265.0, "completions/min_terminated_length": 265.0, "entropy": 0.4122682767920196, "epoch": 0.6556442417331813, "frac_reward_zero_std": 0.28125, "grad_norm": 0.0174997728317976, "kl": 0.03924601306789555, "learning_rate": 1.6091935225657312e-06, "loss": 0.007, "num_tokens": 184294753.0, "reward": 0.847900390625, "reward_std": 1.002097487449646, "rewards/code_complexity_reward/mean": 0.37890625, "rewards/code_complexity_reward/std": 0.4238883852958679, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.224609375, "rewards/code_syntax_reward/std": 0.2489505261182785, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 575, "step_time": 62.873462612740695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 491.787109375, "completions/mean_terminated_length": 422.78448486328125, "completions/min_length": 249.0, "completions/min_terminated_length": 249.0, "entropy": 0.41105279279872775, "epoch": 0.6567844925883695, "frac_reward_zero_std": 0.296875, "grad_norm": 0.014109865762293339, "kl": 0.04243181459605694, "learning_rate": 1.599899632974535e-06, "loss": 0.0056, "num_tokens": 184605356.0, "reward": 0.8646484613418579, "reward_std": 1.0156121253967285, "rewards/code_complexity_reward/mean": 0.38496094942092896, "rewards/code_complexity_reward/std": 0.4281306862831116, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.2265625, "rewards/code_syntax_reward/std": 0.2491423636674881, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 576, "step_time": 44.89443180523813 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7578125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 490.3125, "completions/mean_terminated_length": 422.45159912109375, "completions/min_length": 289.0, "completions/min_terminated_length": 289.0, "entropy": 0.3992698108777404, "epoch": 0.6579247434435576, "frac_reward_zero_std": 0.3125, "grad_norm": 0.01620016060769558, "kl": 0.04067509472952224, "learning_rate": 1.5906200137819378e-06, "loss": 0.0096, "num_tokens": 184916124.0, "reward": 1.024511694908142, "reward_std": 1.0924396514892578, "rewards/code_complexity_reward/mean": 0.4234374761581421, "rewards/code_complexity_reward/std": 0.4332866072654724, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.2470703125, "rewards/code_syntax_reward/std": 0.25022733211517334, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 577, "step_time": 46.40479526668787 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.81640625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 496.80078125, "completions/mean_terminated_length": 429.2127380371094, "completions/min_length": 243.0, "completions/min_terminated_length": 243.0, "entropy": 0.4097813395783305, "epoch": 0.6590649942987458, "frac_reward_zero_std": 0.28125, "grad_norm": 0.014115443453192711, "kl": 0.04021596253733151, "learning_rate": 1.581354812109162e-06, "loss": 0.0061, "num_tokens": 185230718.0, "reward": 0.874218761920929, "reward_std": 0.9986681342124939, "rewards/code_complexity_reward/mean": 0.40058591961860657, "rewards/code_complexity_reward/std": 0.4296492338180542, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.2353515625, "rewards/code_syntax_reward/std": 0.24981455504894257, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 578, "step_time": 46.3839874798432 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.79296875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 494.98828125, "completions/mean_terminated_length": 429.8302001953125, "completions/min_length": 270.0, "completions/min_terminated_length": 270.0, "entropy": 0.41255154460668564, "epoch": 0.6602052451539339, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.015811795368790627, "kl": 0.04028087854385376, "learning_rate": 1.572104174848848e-06, "loss": 0.0099, "num_tokens": 185544900.0, "reward": 0.8694336414337158, "reward_std": 1.0239826440811157, "rewards/code_complexity_reward/mean": 0.38017576932907104, "rewards/code_complexity_reward/std": 0.4241517186164856, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.2255859375, "rewards/code_syntax_reward/std": 0.2490483820438385, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 579, "step_time": 45.128369422629476 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.76171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 490.67578125, "completions/mean_terminated_length": 422.5081787109375, "completions/min_length": 233.0, "completions/min_terminated_length": 233.0, "entropy": 0.4075985150411725, "epoch": 0.661345496009122, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.01442309282720089, "kl": 0.0404826098238118, "learning_rate": 1.562868248662732e-06, "loss": 0.0149, "num_tokens": 185855706.0, "reward": 0.9017578363418579, "reward_std": 1.0308104753494263, "rewards/code_complexity_reward/mean": 0.3998046815395355, "rewards/code_complexity_reward/std": 0.4331538677215576, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.232421875, "rewards/code_syntax_reward/std": 0.24962514638900757, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 580, "step_time": 46.81171205919236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.80078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 494.76953125, "completions/mean_terminated_length": 425.50982666015625, "completions/min_length": 234.0, "completions/min_terminated_length": 234.0, "entropy": 0.4047235334292054, "epoch": 0.6624857468643102, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.014538241550326347, "kl": 0.04021197755355388, "learning_rate": 1.553647179979314e-06, "loss": 0.0091, "num_tokens": 186167712.0, "reward": 0.9150390625, "reward_std": 1.0246180295944214, "rewards/code_complexity_reward/mean": 0.4052734375, "rewards/code_complexity_reward/std": 0.42951470613479614, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.23828125, "rewards/code_syntax_reward/std": 0.24996942281723022, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 581, "step_time": 45.32169197779149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.767578125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 488.3203125, "completions/mean_terminated_length": 410.11767578125, "completions/min_length": 229.0, "completions/min_terminated_length": 229.0, "entropy": 0.40407051937654614, "epoch": 0.6636259977194983, "frac_reward_zero_std": 0.265625, "grad_norm": 0.01619892194867134, "kl": 0.041171774006215855, "learning_rate": 1.5444411149915427e-06, "loss": 0.0071, "num_tokens": 186477496.0, "reward": 0.9716796875, "reward_std": 1.0235623121261597, "rewards/code_complexity_reward/mean": 0.4296875, "rewards/code_complexity_reward/std": 0.4268249571323395, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.2548828125, "rewards/code_syntax_reward/std": 0.25019675493240356, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 582, "step_time": 53.25785032752901 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.806640625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 495.9375, "completions/mean_terminated_length": 428.9292907714844, "completions/min_length": 281.0, "completions/min_terminated_length": 281.0, "entropy": 0.39919489761814475, "epoch": 0.6647662485746865, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.01472484041005373, "kl": 0.04250697066891007, "learning_rate": 1.5352501996544935e-06, "loss": 0.006, "num_tokens": 186791008.0, "reward": 0.8489745855331421, "reward_std": 1.024881362915039, "rewards/code_complexity_reward/mean": 0.3682617247104645, "rewards/code_complexity_reward/std": 0.42355257272720337, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.21875, "rewards/code_syntax_reward/std": 0.24828176200389862, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 583, "step_time": 44.9792915917933 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.78125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 493.126953125, "completions/mean_terminated_length": 425.7232360839844, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.3999244663864374, "epoch": 0.6659064994298746, "frac_reward_zero_std": 0.3125, "grad_norm": 0.01247837021946907, "kl": 0.04099129093810916, "learning_rate": 1.5260745796830545e-06, "loss": 0.0061, "num_tokens": 187103321.0, "reward": 0.9712890386581421, "reward_std": 1.0538188219070435, "rewards/code_complexity_reward/mean": 0.4214843511581421, "rewards/code_complexity_reward/std": 0.4347855746746063, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.2451171875, "rewards/code_syntax_reward/std": 0.25019675493240356, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 584, "step_time": 52.33270827308297 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.849609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 499.869140625, "completions/mean_terminated_length": 431.3376770019531, "completions/min_length": 251.0, "completions/min_terminated_length": 251.0, "entropy": 0.4198006330989301, "epoch": 0.6670467502850627, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.013182491064071655, "kl": 0.04143300908617675, "learning_rate": 1.51691440054962e-06, "loss": 0.006, "num_tokens": 187418554.0, "reward": 0.7500976324081421, "reward_std": 0.9808314442634583, "rewards/code_complexity_reward/mean": 0.3350585997104645, "rewards/code_complexity_reward/std": 0.41731882095336914, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.1982421875, "rewards/code_syntax_reward/std": 0.24482278525829315, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 585, "step_time": 45.909609135240316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.744140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 487.29296875, "completions/mean_terminated_length": 415.43511962890625, "completions/min_length": 238.0, "completions/min_terminated_length": 238.0, "entropy": 0.3922712220810354, "epoch": 0.6681870011402509, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.015131920576095581, "kl": 0.04225369531195611, "learning_rate": 1.5077698074817793e-06, "loss": 0.0128, "num_tokens": 187728488.0, "reward": 1.0227539539337158, "reward_std": 1.0544123649597168, "rewards/code_complexity_reward/mean": 0.44658201932907104, "rewards/code_complexity_reward/std": 0.43529802560806274, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.259765625, "rewards/code_syntax_reward/std": 0.2500534951686859, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 586, "step_time": 54.38036857359111 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.80078125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 493.78515625, "completions/mean_terminated_length": 420.5686340332031, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 0.40736847557127476, "epoch": 0.669327251995439, "frac_reward_zero_std": 0.3125, "grad_norm": 0.013736825436353683, "kl": 0.04308845757623203, "learning_rate": 1.49864094546002e-06, "loss": 0.0109, "num_tokens": 188039790.0, "reward": 0.903369128704071, "reward_std": 1.0216419696807861, "rewards/code_complexity_reward/mean": 0.3975585699081421, "rewards/code_complexity_reward/std": 0.426471084356308, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.2353515625, "rewards/code_syntax_reward/std": 0.24981455504894257, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 587, "step_time": 51.73085052333772 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.775390625, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 490.8671875, "completions/mean_terminated_length": 417.91302490234375, "completions/min_length": 176.0, "completions/min_terminated_length": 176.0, "entropy": 0.4006691235117614, "epoch": 0.6704675028506272, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.013944470323622227, "kl": 0.04469159149448387, "learning_rate": 1.489527959215421e-06, "loss": 0.0125, "num_tokens": 188349654.0, "reward": 1.01123046875, "reward_std": 1.0429879426956177, "rewards/code_complexity_reward/mean": 0.43017578125, "rewards/code_complexity_reward/std": 0.42036300897598267, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.2587890625, "rewards/code_syntax_reward/std": 0.2500897943973541, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 588, "step_time": 45.897017183713615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.77734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 493.078125, "completions/mean_terminated_length": 427.0175476074219, "completions/min_length": 251.0, "completions/min_terminated_length": 251.0, "entropy": 0.3967894301749766, "epoch": 0.6716077537058153, "frac_reward_zero_std": 0.3125, "grad_norm": 0.014690576121211052, "kl": 0.04396358731901273, "learning_rate": 1.4804309932273669e-06, "loss": 0.0092, "num_tokens": 188660910.0, "reward": 1.0648926496505737, "reward_std": 1.0761237144470215, "rewards/code_complexity_reward/mean": 0.45039063692092896, "rewards/code_complexity_reward/std": 0.4337519407272339, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.2626953125, "rewards/code_syntax_reward/std": 0.24992163479328156, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 589, "step_time": 45.35509510524571 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.798828125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 493.361328125, "completions/mean_terminated_length": 419.3495178222656, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.39664435712620616, "epoch": 0.6727480045610034, "frac_reward_zero_std": 0.25, "grad_norm": 0.01396817434579134, "kl": 0.04223097086651251, "learning_rate": 1.471350191721254e-06, "loss": 0.0093, "num_tokens": 188973343.0, "reward": 0.9254882335662842, "reward_std": 0.992077112197876, "rewards/code_complexity_reward/mean": 0.42060548067092896, "rewards/code_complexity_reward/std": 0.42222991585731506, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.2529296875, "rewards/code_syntax_reward/std": 0.25022733211517334, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 590, "step_time": 44.9082193672657 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.80859375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 493.75, "completions/mean_terminated_length": 416.6530456542969, "completions/min_length": 224.0, "completions/min_terminated_length": 224.0, "entropy": 0.400173450820148, "epoch": 0.6738882554161916, "frac_reward_zero_std": 0.3359375, "grad_norm": 0.01341539528220892, "kl": 0.042249276128131896, "learning_rate": 1.462285698666199e-06, "loss": 0.0094, "num_tokens": 189285291.0, "reward": 0.9292969107627869, "reward_std": 1.0409666299819946, "rewards/code_complexity_reward/mean": 0.3990234434604645, "rewards/code_complexity_reward/std": 0.42503485083580017, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.2373046875, "rewards/code_syntax_reward/std": 0.24992163479328156, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 591, "step_time": 45.51510190684348 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7890625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 492.16796875, "completions/mean_terminated_length": 417.9814758300781, "completions/min_length": 244.0, "completions/min_terminated_length": 244.0, "entropy": 0.4027615925297141, "epoch": 0.6750285062713797, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.014152070507407188, "kl": 0.04256869503296912, "learning_rate": 1.4532376577727662e-06, "loss": 0.0087, "num_tokens": 189596845.0, "reward": 0.8726562857627869, "reward_std": 0.9790188074111938, "rewards/code_complexity_reward/mean": 0.4019531011581421, "rewards/code_complexity_reward/std": 0.42283883690834045, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.240234375, "rewards/code_syntax_reward/std": 0.2500534951686859, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 592, "step_time": 52.149195219390094 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.84375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 499.333984375, "completions/mean_terminated_length": 430.9375, "completions/min_length": 305.0, "completions/min_terminated_length": 305.0, "entropy": 0.40663180220872164, "epoch": 0.6761687571265679, "frac_reward_zero_std": 0.28125, "grad_norm": 0.014117092825472355, "kl": 0.04389766251551919, "learning_rate": 1.4442062124906764e-06, "loss": 0.0075, "num_tokens": 189913200.0, "reward": 0.8675781488418579, "reward_std": 1.0183366537094116, "rewards/code_complexity_reward/mean": 0.3827148377895355, "rewards/code_complexity_reward/std": 0.4261155128479004, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.2265625, "rewards/code_syntax_reward/std": 0.2491423636674881, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 593, "step_time": 46.0195144508034 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.779296875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 491.857421875, "completions/mean_terminated_length": 420.7344970703125, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.40399725176393986, "epoch": 0.677309007981756, "frac_reward_zero_std": 0.3125, "grad_norm": 0.014539504423737526, "kl": 0.044032819918356836, "learning_rate": 1.4351915060065488e-06, "loss": 0.0078, "num_tokens": 190223303.0, "reward": 0.937207043170929, "reward_std": 1.0183912515640259, "rewards/code_complexity_reward/mean": 0.41572266817092896, "rewards/code_complexity_reward/std": 0.42855724692344666, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.24609375, "rewards/code_syntax_reward/std": 0.25021395087242126, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 594, "step_time": 54.00353926513344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.826171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 497.033203125, "completions/mean_terminated_length": 425.8988952636719, "completions/min_length": 211.0, "completions/min_terminated_length": 211.0, "entropy": 0.41733548790216446, "epoch": 0.6784492588369442, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.014330682344734669, "kl": 0.04137621313566342, "learning_rate": 1.4261936812416124e-06, "loss": 0.0074, "num_tokens": 190536184.0, "reward": 0.8948730230331421, "reward_std": 1.0172746181488037, "rewards/code_complexity_reward/mean": 0.3926757872104645, "rewards/code_complexity_reward/std": 0.42266982793807983, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.234375, "rewards/code_syntax_reward/std": 0.24975526332855225, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 595, "step_time": 54.73210580833256 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 489.888671875, "completions/mean_terminated_length": 414.4051818847656, "completions/min_length": 174.0, "completions/min_terminated_length": 174.0, "entropy": 0.40132548613473773, "epoch": 0.6795895096921323, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.013245238922536373, "kl": 0.042630395997548476, "learning_rate": 1.4172128808494572e-06, "loss": 0.0101, "num_tokens": 190845907.0, "reward": 0.9608398675918579, "reward_std": 1.0339325666427612, "rewards/code_complexity_reward/mean": 0.4227539300918579, "rewards/code_complexity_reward/std": 0.4301445186138153, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.2490234375, "rewards/code_syntax_reward/std": 0.25024259090423584, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 596, "step_time": 51.60815981309861 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.744140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.84765625, "completions/mean_terminated_length": 413.6946716308594, "completions/min_length": 265.0, "completions/min_terminated_length": 265.0, "entropy": 0.3934655203483999, "epoch": 0.6807297605473204, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.013743579387664795, "kl": 0.04505608370527625, "learning_rate": 1.408249247213762e-06, "loss": 0.01, "num_tokens": 191154001.0, "reward": 1.01318359375, "reward_std": 1.032015323638916, "rewards/code_complexity_reward/mean": 0.45166015625, "rewards/code_complexity_reward/std": 0.43204808235168457, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.2646484375, "rewards/code_syntax_reward/std": 0.24981455504894257, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 597, "step_time": 51.73572745360434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.79296875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 493.23828125, "completions/mean_terminated_length": 421.3773498535156, "completions/min_length": 174.0, "completions/min_terminated_length": 174.0, "entropy": 0.39990955777466297, "epoch": 0.6818700114025086, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.016800301149487495, "kl": 0.04274851694935933, "learning_rate": 1.3993029224460364e-06, "loss": 0.0076, "num_tokens": 191466659.0, "reward": 0.911181628704071, "reward_std": 1.010155439376831, "rewards/code_complexity_reward/mean": 0.40507811307907104, "rewards/code_complexity_reward/std": 0.4242912828922272, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.2421875, "rewards/code_syntax_reward/std": 0.2501222789287567, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 598, "step_time": 48.981991245411336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.73828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 488.74609375, "completions/mean_terminated_length": 423.14923095703125, "completions/min_length": 253.0, "completions/min_terminated_length": 253.0, "entropy": 0.394634694326669, "epoch": 0.6830102622576967, "frac_reward_zero_std": 0.265625, "grad_norm": 0.013205907307565212, "kl": 0.04437840823084116, "learning_rate": 1.390374048383379e-06, "loss": 0.0085, "num_tokens": 191776185.0, "reward": 1.0286132097244263, "reward_std": 1.0077569484710693, "rewards/code_complexity_reward/mean": 0.46611326932907104, "rewards/code_complexity_reward/std": 0.42405545711517334, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.27734375, "rewards/code_syntax_reward/std": 0.24874316155910492, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 599, "step_time": 45.35283666849136 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.765625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 491.484375, "completions/mean_terminated_length": 424.4666748046875, "completions/min_length": 239.0, "completions/min_terminated_length": 239.0, "entropy": 0.40945024136453867, "epoch": 0.6841505131128849, "frac_reward_zero_std": 0.3125, "grad_norm": 0.013795594684779644, "kl": 0.04316240717889741, "learning_rate": 1.3814627665862112e-06, "loss": 0.008, "num_tokens": 192087433.0, "reward": 0.9365234375, "reward_std": 1.0562942028045654, "rewards/code_complexity_reward/mean": 0.3974609375, "rewards/code_complexity_reward/std": 0.42691195011138916, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.234375, "rewards/code_syntax_reward/std": 0.24975526332855225, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 600, "step_time": 50.788126074709 }, { "epoch": 0.6841505131128849, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.8175, "eval_completions/max_length": 511.28, "eval_completions/max_terminated_length": 262.78, "eval_completions/mean_length": 495.4675, "eval_completions/mean_terminated_length": 242.78166748046874, "eval_completions/min_length": 448.8, "eval_completions/min_terminated_length": 223.52, "eval_entropy": 0.4187179785966873, "eval_frac_reward_zero_std": 0.3, "eval_kl": 0.04373398713767528, "eval_loss": 0.009043463505804539, "eval_num_tokens": 192087433.0, "eval_reward": 0.8178749999403954, "eval_reward_std": 0.8029952943325043, "eval_rewards/code_complexity_reward/mean": 0.36975000008940695, "eval_rewards/code_complexity_reward/std": 0.35423240907490255, "eval_rewards/code_execution_reward/mean": 0.225, "eval_rewards/code_execution_reward/std": 0.287596241235733, "eval_rewards/code_syntax_reward/mean": 0.2225, "eval_rewards/code_syntax_reward/std": 0.21474837481975556, "eval_rewards/reasoning_present_reward_func/mean": 0.0, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.000625, "eval_rewards/xmlcount_reward_func/std": 0.001767766922712326, "eval_runtime": 1241.9892, "eval_samples_per_second": 0.081, "eval_steps_per_second": 0.01, "step": 600 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.767578125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 491.46875, "completions/mean_terminated_length": 423.66387939453125, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.4118653400801122, "epoch": 0.685290763968073, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.013207325711846352, "kl": 0.044261983421165496, "learning_rate": 1.3725692183360528e-06, "loss": 0.0107, "num_tokens": 192397905.0, "reward": 0.9532226920127869, "reward_std": 1.0466067790985107, "rewards/code_complexity_reward/mean": 0.4102539122104645, "rewards/code_complexity_reward/std": 0.4273860454559326, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.2421875, "rewards/code_syntax_reward/std": 0.2501222789287567, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 601, "step_time": 54.64077556505799 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.8125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 496.681640625, "completions/mean_terminated_length": 430.3020935058594, "completions/min_length": 286.0, "completions/min_terminated_length": 286.0, "entropy": 0.4190473407506943, "epoch": 0.6864310148232611, "frac_reward_zero_std": 0.28125, "grad_norm": 0.013964480720460415, "kl": 0.040110529924277216, "learning_rate": 1.3636935446332628e-06, "loss": 0.0058, "num_tokens": 192717546.0, "reward": 0.8700195550918579, "reward_std": 0.9948028326034546, "rewards/code_complexity_reward/mean": 0.3983398675918579, "rewards/code_complexity_reward/std": 0.4273229241371155, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.2353515625, "rewards/code_syntax_reward/std": 0.24981455504894257, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 602, "step_time": 45.80286479834467 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.810546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 495.0390625, "completions/mean_terminated_length": 422.4742126464844, "completions/min_length": 217.0, "completions/min_terminated_length": 217.0, "entropy": 0.4105501612648368, "epoch": 0.6875712656784493, "frac_reward_zero_std": 0.296875, "grad_norm": 0.015832215547561646, "kl": 0.0423325413139537, "learning_rate": 1.3548358861948196e-06, "loss": 0.0082, "num_tokens": 193031022.0, "reward": 0.839794933795929, "reward_std": 1.0081393718719482, "rewards/code_complexity_reward/mean": 0.36572265625, "rewards/code_complexity_reward/std": 0.41713759303092957, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.220703125, "rewards/code_syntax_reward/std": 0.24852026998996735, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.001220703125, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 603, "step_time": 45.1618845295161 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.736328125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 488.16796875, "completions/mean_terminated_length": 421.61480712890625, "completions/min_length": 200.0, "completions/min_terminated_length": 200.0, "entropy": 0.3953092242591083, "epoch": 0.6887115165336374, "frac_reward_zero_std": 0.234375, "grad_norm": 0.014656512066721916, "kl": 0.043236729165073484, "learning_rate": 1.3459963834520806e-06, "loss": 0.0114, "num_tokens": 193340180.0, "reward": 1.0135741233825684, "reward_std": 1.0178236961364746, "rewards/code_complexity_reward/mean": 0.4515624940395355, "rewards/code_complexity_reward/std": 0.4251366853713989, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.2685546875, "rewards/code_syntax_reward/std": 0.2495543211698532, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 604, "step_time": 53.10596133582294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.794921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 494.74609375, "completions/mean_terminated_length": 427.8666687011719, "completions/min_length": 265.0, "completions/min_terminated_length": 265.0, "entropy": 0.3990088440477848, "epoch": 0.6898517673888256, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.013662147335708141, "kl": 0.046590237820055336, "learning_rate": 1.3371751765485568e-06, "loss": 0.0028, "num_tokens": 193652622.0, "reward": 0.9642578363418579, "reward_std": 1.031766414642334, "rewards/code_complexity_reward/mean": 0.4173828065395355, "rewards/code_complexity_reward/std": 0.4220912754535675, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.251953125, "rewards/code_syntax_reward/std": 0.2502368688583374, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 605, "step_time": 46.16429033689201 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.794921875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 494.1953125, "completions/mean_terminated_length": 425.18096923828125, "completions/min_length": 230.0, "completions/min_terminated_length": 230.0, "entropy": 0.4028046331368387, "epoch": 0.6909920182440137, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.013955817557871342, "kl": 0.04496132774511352, "learning_rate": 1.3283724053376985e-06, "loss": 0.0077, "num_tokens": 193964902.0, "reward": 0.997998058795929, "reward_std": 1.0487481355667114, "rewards/code_complexity_reward/mean": 0.42900389432907104, "rewards/code_complexity_reward/std": 0.4278394281864166, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.25390625, "rewards/code_syntax_reward/std": 0.25021395087242126, "rewards/reasoning_present_reward_func/mean": 0.0003906250058207661, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.002197265625, "rewards/xmlcount_reward_func/std": 0.028648728504776955, "step": 606, "step_time": 55.06515332683921 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 491.02734375, "completions/mean_terminated_length": 419.4310302734375, "completions/min_length": 261.0, "completions/min_terminated_length": 261.0, "entropy": 0.3895409959368408, "epoch": 0.6921322690992018, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.013210243545472622, "kl": 0.04531146731460467, "learning_rate": 1.319588209380664e-06, "loss": 0.008, "num_tokens": 194274032.0, "reward": 0.9587891101837158, "reward_std": 1.03441321849823, "rewards/code_complexity_reward/mean": 0.42167967557907104, "rewards/code_complexity_reward/std": 0.43027400970458984, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.248046875, "rewards/code_syntax_reward/std": 0.2502368688583374, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 607, "step_time": 45.551768438890576 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.79296875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 491.9921875, "completions/mean_terminated_length": 415.3584899902344, "completions/min_length": 205.0, "completions/min_terminated_length": 205.0, "entropy": 0.3987792292609811, "epoch": 0.69327251995439, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.015057597309350967, "kl": 0.04685309779597446, "learning_rate": 1.3108227279441243e-06, "loss": 0.0074, "num_tokens": 194585500.0, "reward": 1.0020508766174316, "reward_std": 1.0317848920822144, "rewards/code_complexity_reward/mean": 0.4424804747104645, "rewards/code_complexity_reward/std": 0.4265148937702179, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.2626953125, "rewards/code_syntax_reward/std": 0.24992163479328156, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 608, "step_time": 46.605205262079835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.759765625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 491.09765625, "completions/mean_terminated_length": 424.9918518066406, "completions/min_length": 270.0, "completions/min_terminated_length": 270.0, "entropy": 0.396685142070055, "epoch": 0.6944127708095781, "frac_reward_zero_std": 0.265625, "grad_norm": 0.013672838918864727, "kl": 0.04366894101258367, "learning_rate": 1.3020760999980386e-06, "loss": 0.0133, "num_tokens": 194895910.0, "reward": 0.9668945074081421, "reward_std": 1.0017902851104736, "rewards/code_complexity_reward/mean": 0.4395507872104645, "rewards/code_complexity_reward/std": 0.4248208999633789, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.26171875, "rewards/code_syntax_reward/std": 0.24996942281723022, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 609, "step_time": 53.270918706431985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.771484375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 492.541015625, "completions/mean_terminated_length": 426.8461608886719, "completions/min_length": 260.0, "completions/min_terminated_length": 260.0, "entropy": 0.39994997670874, "epoch": 0.6955530216647663, "frac_reward_zero_std": 0.25, "grad_norm": 0.014535106718540192, "kl": 0.04433067259378731, "learning_rate": 1.2933484642134631e-06, "loss": 0.0092, "num_tokens": 195206963.0, "reward": 0.958300769329071, "reward_std": 1.0054265260696411, "rewards/code_complexity_reward/mean": 0.43828123807907104, "rewards/code_complexity_reward/std": 0.43047937750816345, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.2578125, "rewards/code_syntax_reward/std": 0.2501222789287567, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 610, "step_time": 51.205408696085215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.736328125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 487.681640625, "completions/mean_terminated_length": 419.7703552246094, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.3996958080679178, "epoch": 0.6966932725199544, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.014905142597854137, "kl": 0.04381495981942862, "learning_rate": 1.2846399589603453e-06, "loss": 0.0111, "num_tokens": 195514304.0, "reward": 1.0556640625, "reward_std": 1.0086888074874878, "rewards/code_complexity_reward/mean": 0.4755859375, "rewards/code_complexity_reward/std": 0.4224609136581421, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.283203125, "rewards/code_syntax_reward/std": 0.24802762269973755, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 611, "step_time": 45.43962806742638 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.77734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 492.50390625, "completions/mean_terminated_length": 424.4385986328125, "completions/min_length": 275.0, "completions/min_terminated_length": 275.0, "entropy": 0.40202374942600727, "epoch": 0.6978335233751425, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.014541557058691978, "kl": 0.04571516846772283, "learning_rate": 1.2759507223053341e-06, "loss": 0.0103, "num_tokens": 195827274.0, "reward": 0.8968749642372131, "reward_std": 1.0017400979995728, "rewards/code_complexity_reward/mean": 0.4066406190395355, "rewards/code_complexity_reward/std": 0.42789801955223083, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.240234375, "rewards/code_syntax_reward/std": 0.2500534951686859, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 612, "step_time": 45.542295683175325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.77734375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 491.51171875, "completions/mean_terminated_length": 419.9824523925781, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "entropy": 0.40784698026254773, "epoch": 0.6989737742303307, "frac_reward_zero_std": 0.203125, "grad_norm": 0.015937095507979393, "kl": 0.04574282839894295, "learning_rate": 1.2672808920095914e-06, "loss": 0.0117, "num_tokens": 196140284.0, "reward": 0.9556640386581421, "reward_std": 0.9946712851524353, "rewards/code_complexity_reward/mean": 0.43222659826278687, "rewards/code_complexity_reward/std": 0.42158785462379456, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.259765625, "rewards/code_syntax_reward/std": 0.2500534951686859, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 613, "step_time": 48.14829544629902 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.73046875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 487.63671875, "completions/mean_terminated_length": 421.60870361328125, "completions/min_length": 270.0, "completions/min_terminated_length": 270.0, "entropy": 0.3998939348384738, "epoch": 0.7001140250855188, "frac_reward_zero_std": 0.3125, "grad_norm": 0.014512279070913792, "kl": 0.04703648330178112, "learning_rate": 1.2586306055266007e-06, "loss": 0.0119, "num_tokens": 196449542.0, "reward": 0.9512695074081421, "reward_std": 1.016937494277954, "rewards/code_complexity_reward/mean": 0.4307617247104645, "rewards/code_complexity_reward/std": 0.4315538704395294, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.2529296875, "rewards/code_syntax_reward/std": 0.25022733211517334, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 614, "step_time": 45.08392224833369 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.818359375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 495.474609375, "completions/mean_terminated_length": 421.0215148925781, "completions/min_length": 179.0, "completions/min_terminated_length": 179.0, "entropy": 0.4028781387023628, "epoch": 0.701254275940707, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.014865049161016941, "kl": 0.04522994830040261, "learning_rate": 1.2500000000000007e-06, "loss": 0.0076, "num_tokens": 196763241.0, "reward": 0.8856445550918579, "reward_std": 1.0064343214035034, "rewards/code_complexity_reward/mean": 0.3915039002895355, "rewards/code_complexity_reward/std": 0.4193591773509979, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.236328125, "rewards/code_syntax_reward/std": 0.24987001717090607, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 615, "step_time": 54.07384843938053 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 496.71484375, "completions/mean_terminated_length": 436.7500305175781, "completions/min_length": 264.0, "completions/min_terminated_length": 264.0, "entropy": 0.40358306700363755, "epoch": 0.7023945267958951, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.015964843332767487, "kl": 0.047030316141899675, "learning_rate": 1.2413892122613968e-06, "loss": 0.01, "num_tokens": 197078819.0, "reward": 0.9285156726837158, "reward_std": 1.022059679031372, "rewards/code_complexity_reward/mean": 0.40214842557907104, "rewards/code_complexity_reward/std": 0.4191073477268219, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.2431640625, "rewards/code_syntax_reward/std": 0.2501509189605713, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 616, "step_time": 50.63140221312642 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.779296875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 493.404296875, "completions/mean_terminated_length": 427.74334716796875, "completions/min_length": 268.0, "completions/min_terminated_length": 268.0, "entropy": 0.39285068353638053, "epoch": 0.7035347776510832, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.017550960183143616, "kl": 0.04466325987596065, "learning_rate": 1.2327983788282033e-06, "loss": 0.0116, "num_tokens": 197391766.0, "reward": 1.0038573741912842, "reward_std": 1.0112991333007812, "rewards/code_complexity_reward/mean": 0.4453125, "rewards/code_complexity_reward/std": 0.42214566469192505, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.2666015625, "rewards/code_syntax_reward/std": 0.2496921271085739, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 617, "step_time": 45.913616858422756 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.802734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 495.3828125, "completions/mean_terminated_length": 427.7623596191406, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.3894562483765185, "epoch": 0.7046750285062714, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.014036158099770546, "kl": 0.04838390805525705, "learning_rate": 1.2242276359014724e-06, "loss": 0.0048, "num_tokens": 197702994.0, "reward": 0.9522460699081421, "reward_std": 0.992397665977478, "rewards/code_complexity_reward/mean": 0.4268554747104645, "rewards/code_complexity_reward/std": 0.4158979654312134, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.259765625, "rewards/code_syntax_reward/std": 0.2500534951686859, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 618, "step_time": 45.8118268577382 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.78515625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 495.712890625, "completions/mean_terminated_length": 436.1908874511719, "completions/min_length": 277.0, "completions/min_terminated_length": 277.0, "entropy": 0.4169274768792093, "epoch": 0.7058152793614595, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.015164831653237343, "kl": 0.04462362063350156, "learning_rate": 1.215677119363736e-06, "loss": 0.0122, "num_tokens": 198018391.0, "reward": 0.9507323503494263, "reward_std": 0.9919000267982483, "rewards/code_complexity_reward/mean": 0.43681639432907104, "rewards/code_complexity_reward/std": 0.42566514015197754, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.259765625, "rewards/code_syntax_reward/std": 0.2500534951686859, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 619, "step_time": 55.269378155469894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.76171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 490.080078125, "completions/mean_terminated_length": 420.0081787109375, "completions/min_length": 238.0, "completions/min_terminated_length": 238.0, "entropy": 0.3944694581441581, "epoch": 0.7069555302166477, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.014835000038146973, "kl": 0.045581088576000184, "learning_rate": 1.2071469647768564e-06, "loss": 0.0073, "num_tokens": 198328568.0, "reward": 1.0289063453674316, "reward_std": 1.0097782611846924, "rewards/code_complexity_reward/mean": 0.46250003576278687, "rewards/code_complexity_reward/std": 0.423039972782135, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.275390625, "rewards/code_syntax_reward/std": 0.2489505261182785, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 620, "step_time": 59.339633071795106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.80078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 496.3359375, "completions/mean_terminated_length": 433.37255859375, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "entropy": 0.3941877852194011, "epoch": 0.7080957810718358, "frac_reward_zero_std": 0.21875, "grad_norm": 0.0158513393253088, "kl": 0.047293150855693966, "learning_rate": 1.198637307379867e-06, "loss": 0.0114, "num_tokens": 198641612.0, "reward": 0.9791015386581421, "reward_std": 0.9859396815299988, "rewards/code_complexity_reward/mean": 0.4498046934604645, "rewards/code_complexity_reward/std": 0.421916663646698, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.26953125, "rewards/code_syntax_reward/std": 0.24947965145111084, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 621, "step_time": 54.054602523334324 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 491.037109375, "completions/mean_terminated_length": 428.1484375, "completions/min_length": 239.0, "completions/min_terminated_length": 239.0, "entropy": 0.3919572331942618, "epoch": 0.7092360319270239, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.01430302020162344, "kl": 0.047402121126651764, "learning_rate": 1.190148282086837e-06, "loss": 0.0115, "num_tokens": 198951419.0, "reward": 1.052832007408142, "reward_std": 1.0510021448135376, "rewards/code_complexity_reward/mean": 0.4561523497104645, "rewards/code_complexity_reward/std": 0.4296908974647522, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.2685546875, "rewards/code_syntax_reward/std": 0.2495543211698532, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 622, "step_time": 62.050336002372205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.775390625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 492.63671875, "completions/mean_terminated_length": 425.7912902832031, "completions/min_length": 243.0, "completions/min_terminated_length": 243.0, "entropy": 0.40897202398627996, "epoch": 0.7103762827822121, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.014526291750371456, "kl": 0.04620525718200952, "learning_rate": 1.1816800234847304e-06, "loss": 0.0068, "num_tokens": 199263249.0, "reward": 0.982226550579071, "reward_std": 1.0044550895690918, "rewards/code_complexity_reward/mean": 0.44316408038139343, "rewards/code_complexity_reward/std": 0.425165057182312, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.263671875, "rewards/code_syntax_reward/std": 0.24987001717090607, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 623, "step_time": 53.37590180337429 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.734375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 485.40234375, "completions/mean_terminated_length": 411.8676452636719, "completions/min_length": 164.0, "completions/min_terminated_length": 164.0, "entropy": 0.40543680638074875, "epoch": 0.7115165336374002, "frac_reward_zero_std": 0.390625, "grad_norm": 0.012916786596179008, "kl": 0.04796962207183242, "learning_rate": 1.1732326658312693e-06, "loss": 0.0069, "num_tokens": 199570183.0, "reward": 0.96435546875, "reward_std": 1.0399335622787476, "rewards/code_complexity_reward/mean": 0.42333984375, "rewards/code_complexity_reward/std": 0.43200844526290894, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.248046875, "rewards/code_syntax_reward/std": 0.2502368688583374, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 624, "step_time": 45.9325024811551 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.72265625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 484.529296875, "completions/mean_terminated_length": 412.95068359375, "completions/min_length": 180.0, "completions/min_terminated_length": 180.0, "entropy": 0.39681567810475826, "epoch": 0.7126567844925884, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.015221507288515568, "kl": 0.0452012182213366, "learning_rate": 1.1648063430528084e-06, "loss": 0.0068, "num_tokens": 199877270.0, "reward": 1.060791015625, "reward_std": 1.0437833070755005, "rewards/code_complexity_reward/mean": 0.46806639432907104, "rewards/code_complexity_reward/std": 0.43550899624824524, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.271484375, "rewards/code_syntax_reward/std": 0.24931873381137848, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 625, "step_time": 54.911645194515586 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.732421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 483.74609375, "completions/mean_terminated_length": 406.40875244140625, "completions/min_length": 185.0, "completions/min_terminated_length": 185.0, "entropy": 0.38137551909312606, "epoch": 0.7137970353477765, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.014564958401024342, "kl": 0.04984432691708207, "learning_rate": 1.1564011887422098e-06, "loss": 0.0096, "num_tokens": 200181604.0, "reward": 1.0920898914337158, "reward_std": 1.0176576375961304, "rewards/code_complexity_reward/mean": 0.48369139432907104, "rewards/code_complexity_reward/std": 0.4219081699848175, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.2880859375, "rewards/code_syntax_reward/std": 0.24732354283332825, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 626, "step_time": 62.40383490547538 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.751953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 487.30859375, "completions/mean_terminated_length": 412.4566955566406, "completions/min_length": 253.0, "completions/min_terminated_length": 253.0, "entropy": 0.4102002624422312, "epoch": 0.7149372862029647, "frac_reward_zero_std": 0.3125, "grad_norm": 0.013027150183916092, "kl": 0.04824733512941748, "learning_rate": 1.1480173361567287e-06, "loss": 0.0063, "num_tokens": 200489786.0, "reward": 1.0202147960662842, "reward_std": 1.0457533597946167, "rewards/code_complexity_reward/mean": 0.44453126192092896, "rewards/code_complexity_reward/std": 0.4313308894634247, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.2607421875, "rewards/code_syntax_reward/std": 0.2500133812427521, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 627, "step_time": 45.87120832502842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.736328125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.880859375, "completions/mean_terminated_length": 409.14813232421875, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "entropy": 0.3940558088943362, "epoch": 0.7160775370581528, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.015181094408035278, "kl": 0.047480270732194185, "learning_rate": 1.1396549182158933e-06, "loss": 0.0091, "num_tokens": 200797469.0, "reward": 1.0951659679412842, "reward_std": 1.0362497568130493, "rewards/code_complexity_reward/mean": 0.47734373807907104, "rewards/code_complexity_reward/std": 0.4271078407764435, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.2822265625, "rewards/code_syntax_reward/std": 0.24815665185451508, "rewards/reasoning_present_reward_func/mean": 0.0003906250058207661, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.001220703125, "rewards/xmlcount_reward_func/std": 0.019900046288967133, "step": 628, "step_time": 45.670489048585296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.78515625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 493.33984375, "completions/mean_terminated_length": 425.14544677734375, "completions/min_length": 261.0, "completions/min_terminated_length": 261.0, "entropy": 0.39672257797792554, "epoch": 0.7172177879133409, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.014653796330094337, "kl": 0.048006658791564405, "learning_rate": 1.1313140674994053e-06, "loss": 0.0071, "num_tokens": 201109283.0, "reward": 1.024999976158142, "reward_std": 1.0345263481140137, "rewards/code_complexity_reward/mean": 0.4380859136581421, "rewards/code_complexity_reward/std": 0.41710567474365234, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.2666015625, "rewards/code_syntax_reward/std": 0.2496921271085739, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 629, "step_time": 54.572547546587884 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.80078125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 493.72265625, "completions/mean_terminated_length": 420.2549133300781, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "entropy": 0.40641485154628754, "epoch": 0.7183580387685291, "frac_reward_zero_std": 0.265625, "grad_norm": 0.01559380628168583, "kl": 0.04778615414397791, "learning_rate": 1.1229949162450331e-06, "loss": 0.0119, "num_tokens": 201422753.0, "reward": 0.9537109732627869, "reward_std": 1.0042502880096436, "rewards/code_complexity_reward/mean": 0.4292968511581421, "rewards/code_complexity_reward/std": 0.422323614358902, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.2568359375, "rewards/code_syntax_reward/std": 0.2501509189605713, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 630, "step_time": 54.767893847078085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 486.044921875, "completions/mean_terminated_length": 408.1796875, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.39750278554856777, "epoch": 0.7194982896237172, "frac_reward_zero_std": 0.171875, "grad_norm": 0.01742270216345787, "kl": 0.05156538815936074, "learning_rate": 1.1146975963465179e-06, "loss": 0.0113, "num_tokens": 201730272.0, "reward": 1.082617163658142, "reward_std": 1.0121766328811646, "rewards/code_complexity_reward/mean": 0.4820312261581421, "rewards/code_complexity_reward/std": 0.4213009476661682, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.2880859375, "rewards/code_syntax_reward/std": 0.24732354283332825, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 631, "step_time": 61.87906232755631 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.759765625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 488.9765625, "completions/mean_terminated_length": 416.16259765625, "completions/min_length": 230.0, "completions/min_terminated_length": 230.0, "entropy": 0.3925730939954519, "epoch": 0.7206385404789054, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.015221184119582176, "kl": 0.059704028361011297, "learning_rate": 1.106422239351481e-06, "loss": 0.0104, "num_tokens": 202039556.0, "reward": 0.9794921875, "reward_std": 1.0076563358306885, "rewards/code_complexity_reward/mean": 0.4404296875, "rewards/code_complexity_reward/std": 0.42281579971313477, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.263671875, "rewards/code_syntax_reward/std": 0.24987001717090607, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 632, "step_time": 45.73406750243157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.76171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 491.26953125, "completions/mean_terminated_length": 424.9999694824219, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.39298125775530934, "epoch": 0.7217787913340935, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.014258020557463169, "kl": 0.0496606879751198, "learning_rate": 1.0981689764593384e-06, "loss": 0.0084, "num_tokens": 202351122.0, "reward": 1.0441405773162842, "reward_std": 1.009749412536621, "rewards/code_complexity_reward/mean": 0.46113282442092896, "rewards/code_complexity_reward/std": 0.4147159457206726, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.2802734375, "rewards/code_syntax_reward/std": 0.24840296804904938, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 633, "step_time": 45.371377697214484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.74609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 489.03515625, "completions/mean_terminated_length": 421.5538330078125, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.3984777610749006, "epoch": 0.7229190421892816, "frac_reward_zero_std": 0.234375, "grad_norm": 0.014784089289605618, "kl": 0.05049849330680445, "learning_rate": 1.0899379385192222e-06, "loss": 0.0138, "num_tokens": 202659612.0, "reward": 0.9741699695587158, "reward_std": 1.0008376836776733, "rewards/code_complexity_reward/mean": 0.44462889432907104, "rewards/code_complexity_reward/std": 0.42652449011802673, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.263671875, "rewards/code_syntax_reward/std": 0.24987001717090607, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 634, "step_time": 54.972470360808074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.779296875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 492.091796875, "completions/mean_terminated_length": 421.79644775390625, "completions/min_length": 216.0, "completions/min_terminated_length": 216.0, "entropy": 0.3842613659799099, "epoch": 0.7240592930444698, "frac_reward_zero_std": 0.234375, "grad_norm": 0.016822384670376778, "kl": 0.0474961296422407, "learning_rate": 1.0817292560279038e-06, "loss": 0.0069, "num_tokens": 202972115.0, "reward": 1.027734398841858, "reward_std": 1.0395504236221313, "rewards/code_complexity_reward/mean": 0.4486328363418579, "rewards/code_complexity_reward/std": 0.4264816343784332, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.265625, "rewards/code_syntax_reward/std": 0.24975526332855225, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 635, "step_time": 63.11701699811965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.744140625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 487.10546875, "completions/mean_terminated_length": 414.7023010253906, "completions/min_length": 230.0, "completions/min_terminated_length": 230.0, "entropy": 0.3906774022616446, "epoch": 0.7251995438996579, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.014412233605980873, "kl": 0.04764261585660279, "learning_rate": 1.0735430591277268e-06, "loss": 0.0081, "num_tokens": 203281301.0, "reward": 1.064843773841858, "reward_std": 1.0317912101745605, "rewards/code_complexity_reward/mean": 0.4613281488418579, "rewards/code_complexity_reward/std": 0.4203583598136902, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.27734375, "rewards/code_syntax_reward/std": 0.24874316155910492, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 636, "step_time": 55.06046987045556 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.76171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 490.5859375, "completions/mean_terminated_length": 422.1311340332031, "completions/min_length": 193.0, "completions/min_terminated_length": 193.0, "entropy": 0.4035233440808952, "epoch": 0.7263397947548461, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.017399052157998085, "kl": 0.04571218689670786, "learning_rate": 1.0653794776045435e-06, "loss": 0.0056, "num_tokens": 203593945.0, "reward": 0.985546886920929, "reward_std": 1.0285580158233643, "rewards/code_complexity_reward/mean": 0.42402344942092896, "rewards/code_complexity_reward/std": 0.42038214206695557, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.2568359375, "rewards/code_syntax_reward/std": 0.2501509189605713, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 637, "step_time": 54.672821594402194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.771484375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 489.82421875, "completions/mean_terminated_length": 414.957275390625, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "entropy": 0.4039645274169743, "epoch": 0.7274800456100342, "frac_reward_zero_std": 0.203125, "grad_norm": 0.015124338679015636, "kl": 0.048399656487163156, "learning_rate": 1.0572386408856553e-06, "loss": 0.0068, "num_tokens": 203903271.0, "reward": 0.9895020127296448, "reward_std": 1.0085971355438232, "rewards/code_complexity_reward/mean": 0.443359375, "rewards/code_complexity_reward/std": 0.4228026568889618, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.2646484375, "rewards/code_syntax_reward/std": 0.24981455504894257, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 638, "step_time": 54.26999957021326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.767578125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 488.822265625, "completions/mean_terminated_length": 412.27734375, "completions/min_length": 232.0, "completions/min_terminated_length": 232.0, "entropy": 0.3994159339927137, "epoch": 0.7286202964652223, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.014941238798201084, "kl": 0.04732334124855697, "learning_rate": 1.0491206780377636e-06, "loss": 0.0095, "num_tokens": 204212192.0, "reward": 0.9553711414337158, "reward_std": 1.0068092346191406, "rewards/code_complexity_reward/mean": 0.42705076932907104, "rewards/code_complexity_reward/std": 0.4227573573589325, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.2568359375, "rewards/code_syntax_reward/std": 0.2501509189605713, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 639, "step_time": 50.88818213716149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 487.599609375, "completions/mean_terminated_length": 414.3984375, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "entropy": 0.38598283333703876, "epoch": 0.7297605473204105, "frac_reward_zero_std": 0.265625, "grad_norm": 0.01604459062218666, "kl": 0.05229453992797062, "learning_rate": 1.0410257177649217e-06, "loss": 0.0079, "num_tokens": 204519875.0, "reward": 1.0456055402755737, "reward_std": 1.0063376426696777, "rewards/code_complexity_reward/mean": 0.47138673067092896, "rewards/code_complexity_reward/std": 0.42279312014579773, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.28125, "rewards/code_syntax_reward/std": 0.24828176200389862, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 640, "step_time": 61.98550589941442 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.767578125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 490.16796875, "completions/mean_terminated_length": 418.0672607421875, "completions/min_length": 256.0, "completions/min_terminated_length": 256.0, "entropy": 0.3883255999535322, "epoch": 0.7309007981755986, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.01572849042713642, "kl": 0.04914459242718294, "learning_rate": 1.0329538884064948e-06, "loss": 0.0107, "num_tokens": 204829517.0, "reward": 1.0107421875, "reward_std": 1.005752444267273, "rewards/code_complexity_reward/mean": 0.4482421875, "rewards/code_complexity_reward/std": 0.4175302982330322, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.271484375, "rewards/code_syntax_reward/std": 0.24931873381137848, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 641, "step_time": 62.5622599683702 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.755859375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 490.291015625, "completions/mean_terminated_length": 423.08001708984375, "completions/min_length": 221.0, "completions/min_terminated_length": 221.0, "entropy": 0.39252569805830717, "epoch": 0.7320410490307868, "frac_reward_zero_std": 0.1875, "grad_norm": 0.017397969961166382, "kl": 0.04896959906909615, "learning_rate": 1.0249053179351257e-06, "loss": 0.0106, "num_tokens": 205138554.0, "reward": 1.0609374046325684, "reward_std": 1.0354044437408447, "rewards/code_complexity_reward/mean": 0.4632812440395355, "rewards/code_complexity_reward/std": 0.42369791865348816, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.275390625, "rewards/code_syntax_reward/std": 0.2489505261182785, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 642, "step_time": 46.06060379277915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.759765625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 489.3046875, "completions/mean_terminated_length": 417.5284423828125, "completions/min_length": 253.0, "completions/min_terminated_length": 253.0, "entropy": 0.4040879965759814, "epoch": 0.7331812998859749, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.014269494451582432, "kl": 0.04740786843467504, "learning_rate": 1.0168801339547046e-06, "loss": 0.0065, "num_tokens": 205450450.0, "reward": 0.9969726800918579, "reward_std": 1.0165776014328003, "rewards/code_complexity_reward/mean": 0.4452148675918579, "rewards/code_complexity_reward/std": 0.4259231388568878, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.2646484375, "rewards/code_syntax_reward/std": 0.24981455504894257, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 643, "step_time": 63.273946510627866 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.4296875, "completions/mean_terminated_length": 418.4857177734375, "completions/min_length": 246.0, "completions/min_terminated_length": 246.0, "entropy": 0.3888565399684012, "epoch": 0.734321550741163, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.015458904206752777, "kl": 0.047940944554284215, "learning_rate": 1.0088784636983472e-06, "loss": 0.0068, "num_tokens": 205759498.0, "reward": 1.07177734375, "reward_std": 0.9940327405929565, "rewards/code_complexity_reward/mean": 0.48662108182907104, "rewards/code_complexity_reward/std": 0.4191732108592987, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.291015625, "rewards/code_syntax_reward/std": 0.24685366451740265, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 644, "step_time": 46.220364067703485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.720703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 485.462890625, "completions/mean_terminated_length": 416.98602294921875, "completions/min_length": 241.0, "completions/min_terminated_length": 241.0, "entropy": 0.40524064004421234, "epoch": 0.7354618015963512, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.015646692365407944, "kl": 0.048025466792751104, "learning_rate": 1.0009004340263778e-06, "loss": 0.0101, "num_tokens": 206069663.0, "reward": 1.14501953125, "reward_std": 1.0560096502304077, "rewards/code_complexity_reward/mean": 0.49462890625, "rewards/code_complexity_reward/std": 0.42912036180496216, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.2890625, "rewards/code_syntax_reward/std": 0.24717088043689728, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 645, "step_time": 45.380115321837366 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.798828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 493.609375, "completions/mean_terminated_length": 420.58251953125, "completions/min_length": 280.0, "completions/min_terminated_length": 280.0, "entropy": 0.3954313676804304, "epoch": 0.7366020524515393, "frac_reward_zero_std": 0.203125, "grad_norm": 0.01752094179391861, "kl": 0.04815385339315981, "learning_rate": 9.929461714243166e-07, "loss": 0.0117, "num_tokens": 206380347.0, "reward": 0.9076172113418579, "reward_std": 1.0025312900543213, "rewards/code_complexity_reward/mean": 0.4007812738418579, "rewards/code_complexity_reward/std": 0.41702911257743835, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.2451171875, "rewards/code_syntax_reward/std": 0.25019675493240356, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 646, "step_time": 45.365295676514506 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.712890625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 482.4296875, "completions/mean_terminated_length": 409.0068054199219, "completions/min_length": 220.0, "completions/min_terminated_length": 220.0, "entropy": 0.39123289892449975, "epoch": 0.7377423033067275, "frac_reward_zero_std": 0.265625, "grad_norm": 0.01499419566243887, "kl": 0.05358021392021328, "learning_rate": 9.850158020008757e-07, "loss": 0.0071, "num_tokens": 206686675.0, "reward": 1.1709473133087158, "reward_std": 1.025414228439331, "rewards/code_complexity_reward/mean": 0.509570300579071, "rewards/code_complexity_reward/std": 0.4166380763053894, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.3037109375, "rewards/code_syntax_reward/std": 0.24440090358257294, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 647, "step_time": 52.23570995032787 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.740234375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 490.412109375, "completions/mean_terminated_length": 428.8947448730469, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "entropy": 0.3998180590569973, "epoch": 0.7388825541619156, "frac_reward_zero_std": 0.1875, "grad_norm": 0.016390560194849968, "kl": 0.04908984946087003, "learning_rate": 9.771094514859587e-07, "loss": 0.009, "num_tokens": 206995354.0, "reward": 0.9491211175918579, "reward_std": 0.9979059100151062, "rewards/code_complexity_reward/mean": 0.4334960877895355, "rewards/code_complexity_reward/std": 0.42574453353881836, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.2578125, "rewards/code_syntax_reward/std": 0.2501222789287567, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 648, "step_time": 59.39364284276962 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.74609375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 489.044921875, "completions/mean_terminated_length": 421.5923156738281, "completions/min_length": 198.0, "completions/min_terminated_length": 198.0, "entropy": 0.38748633256182075, "epoch": 0.7400228050171037, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.017623258754611015, "kl": 0.04861816862830892, "learning_rate": 9.692272452286686e-07, "loss": 0.0078, "num_tokens": 207306073.0, "reward": 1.0731933116912842, "reward_std": 1.020004391670227, "rewards/code_complexity_reward/mean": 0.47607421875, "rewards/code_complexity_reward/std": 0.4231889247894287, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.283203125, "rewards/code_syntax_reward/std": 0.24802762269973755, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.001220703125, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 649, "step_time": 65.68651519715786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.76171875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 490.251953125, "completions/mean_terminated_length": 420.7294921875, "completions/min_length": 261.0, "completions/min_terminated_length": 261.0, "entropy": 0.39234346989542246, "epoch": 0.7411630558722919, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.0136997289955616, "kl": 0.050044092000462115, "learning_rate": 9.613693081953195e-07, "loss": 0.0077, "num_tokens": 207617298.0, "reward": 1.0232911109924316, "reward_std": 1.0243492126464844, "rewards/code_complexity_reward/mean": 0.4507812559604645, "rewards/code_complexity_reward/std": 0.42306235432624817, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.26953125, "rewards/code_syntax_reward/std": 0.24947965145111084, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 650, "step_time": 57.584434256888926 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.724609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.560546875, "completions/mean_terminated_length": 415.9928894042969, "completions/min_length": 224.0, "completions/min_terminated_length": 224.0, "entropy": 0.3928950629197061, "epoch": 0.74230330672748, "frac_reward_zero_std": 0.203125, "grad_norm": 0.014914528466761112, "kl": 0.05085370223969221, "learning_rate": 9.535357649674554e-07, "loss": 0.0066, "num_tokens": 207924589.0, "reward": 1.073876976966858, "reward_std": 1.0069948434829712, "rewards/code_complexity_reward/mean": 0.48017579317092896, "rewards/code_complexity_reward/std": 0.41876548528671265, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.2880859375, "rewards/code_syntax_reward/std": 0.24732354283332825, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 651, "step_time": 45.59163629449904 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.70703125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 484.318359375, "completions/mean_terminated_length": 417.5133361816406, "completions/min_length": 229.0, "completions/min_terminated_length": 229.0, "entropy": 0.38845905289053917, "epoch": 0.7434435575826682, "frac_reward_zero_std": 0.140625, "grad_norm": 0.015819557011127472, "kl": 0.05001734185498208, "learning_rate": 9.457267397398756e-07, "loss": 0.0107, "num_tokens": 208231736.0, "reward": 1.1685547828674316, "reward_std": 1.003212809562683, "rewards/code_complexity_reward/mean": 0.5269531011581421, "rewards/code_complexity_reward/std": 0.4210996627807617, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.3095703125, "rewards/code_syntax_reward/std": 0.24303650856018066, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 652, "step_time": 45.51612049154937 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.724609375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 484.703125, "completions/mean_terminated_length": 412.8794250488281, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.4030157523229718, "epoch": 0.7445838084378563, "frac_reward_zero_std": 0.296875, "grad_norm": 0.013952264562249184, "kl": 0.05088397237705067, "learning_rate": 9.379423563186652e-07, "loss": 0.0067, "num_tokens": 208539480.0, "reward": 1.1337401866912842, "reward_std": 1.053156852722168, "rewards/code_complexity_reward/mean": 0.48310545086860657, "rewards/code_complexity_reward/std": 0.4257291853427887, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.28515625, "rewards/code_syntax_reward/std": 0.24775780737400055, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 653, "step_time": 59.39657203666866 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.728515625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 482.830078125, "completions/mean_terminated_length": 404.553955078125, "completions/min_length": 199.0, "completions/min_terminated_length": 199.0, "entropy": 0.3994027986191213, "epoch": 0.7457240592930444, "frac_reward_zero_std": 0.203125, "grad_norm": 0.015499882400035858, "kl": 0.05030067096231505, "learning_rate": 9.301827381192321e-07, "loss": 0.0098, "num_tokens": 208844613.0, "reward": 1.053320288658142, "reward_std": 1.0059455633163452, "rewards/code_complexity_reward/mean": 0.4761718511581421, "rewards/code_complexity_reward/std": 0.4229458272457123, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.2841796875, "rewards/code_syntax_reward/std": 0.24789467453956604, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 654, "step_time": 52.76521549932659 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.736328125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 488.20703125, "completions/mean_terminated_length": 421.7629699707031, "completions/min_length": 227.0, "completions/min_terminated_length": 227.0, "entropy": 0.3988245353102684, "epoch": 0.7468643101482326, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.01561051607131958, "kl": 0.050284569966606796, "learning_rate": 9.224480081643516e-07, "loss": 0.0091, "num_tokens": 209153899.0, "reward": 1.110498070716858, "reward_std": 0.9877148270606995, "rewards/code_complexity_reward/mean": 0.5038086175918579, "rewards/code_complexity_reward/std": 0.4151015877723694, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.3017578125, "rewards/code_syntax_reward/std": 0.24482278525829315, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 655, "step_time": 55.569861124269664 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.732421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 487.353515625, "completions/mean_terminated_length": 419.8905029296875, "completions/min_length": 255.0, "completions/min_terminated_length": 255.0, "entropy": 0.38994683884084225, "epoch": 0.7480045610034207, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.014508092775940895, "kl": 0.051627222332172096, "learning_rate": 9.147382890822143e-07, "loss": 0.007, "num_tokens": 209465060.0, "reward": 1.0631835460662842, "reward_std": 1.0433402061462402, "rewards/code_complexity_reward/mean": 0.46259763836860657, "rewards/code_complexity_reward/std": 0.4280553460121155, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.2724609375, "rewards/code_syntax_reward/std": 0.2492324709892273, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 656, "step_time": 74.24067687243223 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.74609375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 486.767578125, "completions/mean_terminated_length": 412.6230773925781, "completions/min_length": 244.0, "completions/min_terminated_length": 244.0, "entropy": 0.39527717512100935, "epoch": 0.7491448118586089, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.014604002237319946, "kl": 0.050375382124911994, "learning_rate": 9.070537031044829e-07, "loss": 0.011, "num_tokens": 209773781.0, "reward": 1.1572754383087158, "reward_std": 1.0228276252746582, "rewards/code_complexity_reward/mean": 0.509570300579071, "rewards/code_complexity_reward/std": 0.4192017614841461, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.3017578125, "rewards/code_syntax_reward/std": 0.24482278525829315, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 657, "step_time": 83.54582098964602 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.72265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.15234375, "completions/mean_terminated_length": 418.80279541015625, "completions/min_length": 202.0, "completions/min_terminated_length": 202.0, "entropy": 0.38950205547735095, "epoch": 0.750285062713797, "frac_reward_zero_std": 0.1875, "grad_norm": 0.01596144028007984, "kl": 0.0515675576752983, "learning_rate": 8.993943720643536e-07, "loss": 0.007, "num_tokens": 210081807.0, "reward": 1.221777319908142, "reward_std": 1.040401816368103, "rewards/code_complexity_reward/mean": 0.5169921517372131, "rewards/code_complexity_reward/std": 0.4122444987297058, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.3095703125, "rewards/code_syntax_reward/std": 0.24303650856018066, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 658, "step_time": 45.37041860539466 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.71484375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.763671875, "completions/mean_terminated_length": 423.5, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "entropy": 0.3892208859324455, "epoch": 0.7514253135689852, "frac_reward_zero_std": 0.171875, "grad_norm": 0.01625392958521843, "kl": 0.05123944894876331, "learning_rate": 8.917604173946268e-07, "loss": 0.0102, "num_tokens": 210390006.0, "reward": 1.075927734375, "reward_std": 1.0003756284713745, "rewards/code_complexity_reward/mean": 0.48417967557907104, "rewards/code_complexity_reward/std": 0.41799506545066833, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.2900390625, "rewards/code_syntax_reward/std": 0.24701425433158875, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 659, "step_time": 45.25018537975848 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.697265625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 482.431640625, "completions/mean_terminated_length": 414.32904052734375, "completions/min_length": 182.0, "completions/min_terminated_length": 182.0, "entropy": 0.3912618951871991, "epoch": 0.7525655644241733, "frac_reward_zero_std": 0.140625, "grad_norm": 0.014623278751969337, "kl": 0.050026569224428385, "learning_rate": 8.841519601257756e-07, "loss": 0.0103, "num_tokens": 210697267.0, "reward": 1.213476538658142, "reward_std": 1.0017812252044678, "rewards/code_complexity_reward/mean": 0.5416015386581421, "rewards/code_complexity_reward/std": 0.4113575518131256, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.322265625, "rewards/code_syntax_reward/std": 0.23956161737442017, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 660, "step_time": 60.47607784718275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.720703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 489.31640625, "completions/mean_terminated_length": 430.783203125, "completions/min_length": 203.0, "completions/min_terminated_length": 203.0, "entropy": 0.4055361463688314, "epoch": 0.7537058152793614, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.01590397022664547, "kl": 0.04944407334551215, "learning_rate": 8.765691208840374e-07, "loss": 0.0052, "num_tokens": 211007729.0, "reward": 1.047753930091858, "reward_std": 1.0182578563690186, "rewards/code_complexity_reward/mean": 0.46767574548721313, "rewards/code_complexity_reward/std": 0.4256054162979126, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.27734375, "rewards/code_syntax_reward/std": 0.24874316155910492, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 661, "step_time": 54.43615236412734 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.740234375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 489.693359375, "completions/mean_terminated_length": 426.1278381347656, "completions/min_length": 239.0, "completions/min_terminated_length": 239.0, "entropy": 0.39650676446035504, "epoch": 0.7548460661345496, "frac_reward_zero_std": 0.15625, "grad_norm": 0.01662093959748745, "kl": 0.04957972915144637, "learning_rate": 8.690120198894919e-07, "loss": 0.0086, "num_tokens": 211318968.0, "reward": 1.104150414466858, "reward_std": 0.9869518876075745, "rewards/code_complexity_reward/mean": 0.5042968988418579, "rewards/code_complexity_reward/std": 0.4174063503742218, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.30078125, "rewards/code_syntax_reward/std": 0.2450276017189026, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 662, "step_time": 52.08527090866119 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.759765625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 491.841796875, "completions/mean_terminated_length": 428.08941650390625, "completions/min_length": 254.0, "completions/min_terminated_length": 254.0, "entropy": 0.39185379864647985, "epoch": 0.7559863169897377, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.01732599548995495, "kl": 0.049177328299265355, "learning_rate": 8.614807769541589e-07, "loss": 0.0116, "num_tokens": 211630107.0, "reward": 1.0763671398162842, "reward_std": 1.02317476272583, "rewards/code_complexity_reward/mean": 0.47871094942092896, "rewards/code_complexity_reward/std": 0.42434006929397583, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.283203125, "rewards/code_syntax_reward/std": 0.24802762269973755, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 663, "step_time": 46.104248768649995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.74609375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 485.77734375, "completions/mean_terminated_length": 408.72308349609375, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "entropy": 0.3919709846377373, "epoch": 0.7571265678449259, "frac_reward_zero_std": 0.203125, "grad_norm": 0.015674922615289688, "kl": 0.05313355510588735, "learning_rate": 8.539755114800996e-07, "loss": 0.0112, "num_tokens": 211936973.0, "reward": 1.0897459983825684, "reward_std": 1.0094791650772095, "rewards/code_complexity_reward/mean": 0.4862304627895355, "rewards/code_complexity_reward/std": 0.41934722661972046, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.291015625, "rewards/code_syntax_reward/std": 0.24685366451740265, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 664, "step_time": 53.15956997964531 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.751953125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 485.857421875, "completions/mean_terminated_length": 406.6062927246094, "completions/min_length": 167.0, "completions/min_terminated_length": 167.0, "entropy": 0.39224539790302515, "epoch": 0.758266818700114, "frac_reward_zero_std": 0.1875, "grad_norm": 0.016183873638510704, "kl": 0.05026664223987609, "learning_rate": 8.464963424575215e-07, "loss": 0.0106, "num_tokens": 212246580.0, "reward": 1.119726538658142, "reward_std": 1.0213159322738647, "rewards/code_complexity_reward/mean": 0.48115232586860657, "rewards/code_complexity_reward/std": 0.41003942489624023, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.294921875, "rewards/code_syntax_reward/std": 0.24617145955562592, "rewards/reasoning_present_reward_func/mean": 0.0003906250058207661, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.00146484375, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 665, "step_time": 51.51719525270164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.720703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 485.767578125, "completions/mean_terminated_length": 418.076904296875, "completions/min_length": 226.0, "completions/min_terminated_length": 226.0, "entropy": 0.3944085487164557, "epoch": 0.7594070695553021, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.014516005292534828, "kl": 0.04785110492957756, "learning_rate": 8.390433884628948e-07, "loss": 0.01, "num_tokens": 212554441.0, "reward": 1.079980492591858, "reward_std": 0.9962230920791626, "rewards/code_complexity_reward/mean": 0.4852539300918579, "rewards/code_complexity_reward/std": 0.4146560728549957, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.2939453125, "rewards/code_syntax_reward/std": 0.24634800851345062, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 666, "step_time": 51.31213280092925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.4375, "completions/mean_terminated_length": 414.8571472167969, "completions/min_length": 260.0, "completions/min_terminated_length": 260.0, "entropy": 0.3938609496690333, "epoch": 0.7605473204104903, "frac_reward_zero_std": 0.203125, "grad_norm": 0.017928045243024826, "kl": 0.07921067526331171, "learning_rate": 8.316167676570666e-07, "loss": 0.0132, "num_tokens": 212861177.0, "reward": 1.1383788585662842, "reward_std": 1.0332329273223877, "rewards/code_complexity_reward/mean": 0.49482423067092896, "rewards/code_complexity_reward/std": 0.4209968149662018, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.2939453125, "rewards/code_syntax_reward/std": 0.24634800851345062, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 667, "step_time": 53.43768129404634 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.666015625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 480.1328125, "completions/mean_terminated_length": 416.5848083496094, "completions/min_length": 220.0, "completions/min_terminated_length": 220.0, "entropy": 0.3853088356554508, "epoch": 0.7616875712656784, "frac_reward_zero_std": 0.21875, "grad_norm": 0.01450419146567583, "kl": 0.05177359079243615, "learning_rate": 8.242165977833974e-07, "loss": 0.0115, "num_tokens": 213165969.0, "reward": 1.212548851966858, "reward_std": 1.0359818935394287, "rewards/code_complexity_reward/mean": 0.525585949420929, "rewards/code_complexity_reward/std": 0.4195907711982727, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.30859375, "rewards/code_syntax_reward/std": 0.2432742565870285, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.001220703125, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 668, "step_time": 45.71952734421939 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.708984375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 488.583984375, "completions/mean_terminated_length": 431.53692626953125, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.3899611122906208, "epoch": 0.7628278221208666, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.01526669692248106, "kl": 0.05222095985664055, "learning_rate": 8.168429961658822e-07, "loss": 0.0073, "num_tokens": 213475456.0, "reward": 1.1952147483825684, "reward_std": 1.0186718702316284, "rewards/code_complexity_reward/mean": 0.5272461175918579, "rewards/code_complexity_reward/std": 0.41778564453125, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.310546875, "rewards/code_syntax_reward/std": 0.24279458820819855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 669, "step_time": 45.55884056072682 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7578125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 489.1875, "completions/mean_terminated_length": 417.8064270019531, "completions/min_length": 172.0, "completions/min_terminated_length": 172.0, "entropy": 0.3957027168944478, "epoch": 0.7639680729760547, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.015457039698958397, "kl": 0.05093203095020726, "learning_rate": 8.094960797073023e-07, "loss": 0.0077, "num_tokens": 213787132.0, "reward": 1.0470702648162842, "reward_std": 0.9877088069915771, "rewards/code_complexity_reward/mean": 0.47480469942092896, "rewards/code_complexity_reward/std": 0.4162195324897766, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.287109375, "rewards/code_syntax_reward/std": 0.24747224152088165, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 670, "step_time": 45.532146711833775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.671875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 477.57421875, "completions/mean_terminated_length": 407.0833435058594, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.38976697670295835, "epoch": 0.7651083238312428, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.016527537256479263, "kl": 0.05170449789147824, "learning_rate": 8.021759648873642e-07, "loss": 0.0099, "num_tokens": 214091282.0, "reward": 1.23095703125, "reward_std": 1.034277319908142, "rewards/code_complexity_reward/mean": 0.52880859375, "rewards/code_complexity_reward/std": 0.41514283418655396, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.3134765625, "rewards/code_syntax_reward/std": 0.24204368889331818, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 671, "step_time": 59.28425691369921 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.755859375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 489.0390625, "completions/mean_terminated_length": 417.9520263671875, "completions/min_length": 213.0, "completions/min_terminated_length": 213.0, "entropy": 0.3966873036697507, "epoch": 0.766248574686431, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.014877214096486568, "kl": 0.05312607652740553, "learning_rate": 7.94882767760852e-07, "loss": 0.0101, "num_tokens": 214402950.0, "reward": 1.0492186546325684, "reward_std": 1.0034136772155762, "rewards/code_complexity_reward/mean": 0.4642578363418579, "rewards/code_complexity_reward/std": 0.41552767157554626, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.2822265625, "rewards/code_syntax_reward/std": 0.24815665185451508, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 672, "step_time": 46.28371483460069 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.744140625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 485.203125, "completions/mean_terminated_length": 407.2671813964844, "completions/min_length": 193.0, "completions/min_terminated_length": 193.0, "entropy": 0.3865043483674526, "epoch": 0.7673888255416191, "frac_reward_zero_std": 0.234375, "grad_norm": 0.01615297608077526, "kl": 0.052413745666854084, "learning_rate": 7.876166039557967e-07, "loss": 0.0114, "num_tokens": 214709474.0, "reward": 1.0613281726837158, "reward_std": 1.026411771774292, "rewards/code_complexity_reward/mean": 0.46660158038139343, "rewards/code_complexity_reward/std": 0.4230489730834961, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.2783203125, "rewards/code_syntax_reward/std": 0.24863366782665253, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 673, "step_time": 55.18160243425518 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.708984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.3671875, "completions/mean_terminated_length": 420.48321533203125, "completions/min_length": 214.0, "completions/min_terminated_length": 214.0, "entropy": 0.392764987424016, "epoch": 0.7685290763968073, "frac_reward_zero_std": 0.234375, "grad_norm": 0.016346128657460213, "kl": 0.0507352850981988, "learning_rate": 7.8037758867163e-07, "loss": 0.0068, "num_tokens": 215018594.0, "reward": 1.122167944908142, "reward_std": 1.0187536478042603, "rewards/code_complexity_reward/mean": 0.4981445074081421, "rewards/code_complexity_reward/std": 0.4195229113101959, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.2958984375, "rewards/code_syntax_reward/std": 0.24599088728427887, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 674, "step_time": 66.61618690099567 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.73828125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 489.310546875, "completions/mean_terminated_length": 425.30596923828125, "completions/min_length": 225.0, "completions/min_terminated_length": 225.0, "entropy": 0.3978952239267528, "epoch": 0.7696693272519954, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.016033465042710304, "kl": 0.05126120516797528, "learning_rate": 7.731658366773717e-07, "loss": 0.0073, "num_tokens": 215328949.0, "reward": 1.0452148914337158, "reward_std": 0.9626801013946533, "rewards/code_complexity_reward/mean": 0.48759764432907104, "rewards/code_complexity_reward/std": 0.41472113132476807, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.2939453125, "rewards/code_syntax_reward/std": 0.24634800851345062, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 675, "step_time": 55.50179318897426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.654296875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 475.75390625, "completions/mean_terminated_length": 407.1525573730469, "completions/min_length": 192.0, "completions/min_terminated_length": 192.0, "entropy": 0.3899144404567778, "epoch": 0.7708095781071835, "frac_reward_zero_std": 0.203125, "grad_norm": 0.016198424622416496, "kl": 0.05121268308721483, "learning_rate": 7.659814623097958e-07, "loss": 0.0125, "num_tokens": 215631899.0, "reward": 1.171728491783142, "reward_std": 1.0479373931884766, "rewards/code_complexity_reward/mean": 0.4996093511581421, "rewards/code_complexity_reward/std": 0.4201386272907257, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.296875, "rewards/code_syntax_reward/std": 0.24580632150173187, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 676, "step_time": 45.130355722270906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7109375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 483.55859375, "completions/mean_terminated_length": 413.6081237792969, "completions/min_length": 256.0, "completions/min_terminated_length": 256.0, "entropy": 0.3870549057610333, "epoch": 0.7719498289623717, "frac_reward_zero_std": 0.171875, "grad_norm": 0.01645049639046192, "kl": 0.05414984206436202, "learning_rate": 7.588245794716315e-07, "loss": 0.011, "num_tokens": 215938393.0, "reward": 1.059960961341858, "reward_std": 0.9865278005599976, "rewards/code_complexity_reward/mean": 0.48847657442092896, "rewards/code_complexity_reward/std": 0.42036470770835876, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.291015625, "rewards/code_syntax_reward/std": 0.24685366451740265, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 677, "step_time": 45.484199687838554 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.71875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.779296875, "completions/mean_terminated_length": 415.21527099609375, "completions/min_length": 153.0, "completions/min_terminated_length": 153.0, "entropy": 0.39421014953404665, "epoch": 0.7730900798175598, "frac_reward_zero_std": 0.203125, "grad_norm": 0.01641525886952877, "kl": 0.052511111192870885, "learning_rate": 7.516953016297479e-07, "loss": 0.0046, "num_tokens": 216247312.0, "reward": 1.0971190929412842, "reward_std": 0.9848076701164246, "rewards/code_complexity_reward/mean": 0.50830078125, "rewards/code_complexity_reward/std": 0.4205629825592041, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.30078125, "rewards/code_syntax_reward/std": 0.2450276017189026, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 678, "step_time": 55.45148311275989 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.74609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 489.705078125, "completions/mean_terminated_length": 424.19232177734375, "completions/min_length": 215.0, "completions/min_terminated_length": 215.0, "entropy": 0.3954656985588372, "epoch": 0.774230330672748, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.01403803937137127, "kl": 0.052218802797142416, "learning_rate": 7.445937418133564e-07, "loss": 0.0114, "num_tokens": 216559073.0, "reward": 0.9966309070587158, "reward_std": 1.0050287246704102, "rewards/code_complexity_reward/mean": 0.44755858182907104, "rewards/code_complexity_reward/std": 0.42196181416511536, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.267578125, "rewards/code_syntax_reward/std": 0.24962514638900757, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 679, "step_time": 58.22362224012613 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.775390625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 493.24609375, "completions/mean_terminated_length": 428.50433349609375, "completions/min_length": 262.0, "completions/min_terminated_length": 262.0, "entropy": 0.40428188629448414, "epoch": 0.7753705815279361, "frac_reward_zero_std": 0.28125, "grad_norm": 0.013840852305293083, "kl": 0.0512924839858897, "learning_rate": 7.375200126122256e-07, "loss": 0.0078, "num_tokens": 216871859.0, "reward": 0.9842284917831421, "reward_std": 1.0291122198104858, "rewards/code_complexity_reward/mean": 0.4332031011581421, "rewards/code_complexity_reward/std": 0.4253478944301605, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.2578125, "rewards/code_syntax_reward/std": 0.2501222789287567, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 680, "step_time": 45.45657093543559 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.705078125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 480.19921875, "completions/mean_terminated_length": 404.17218017578125, "completions/min_length": 252.0, "completions/min_terminated_length": 252.0, "entropy": 0.38332271995022893, "epoch": 0.7765108323831242, "frac_reward_zero_std": 0.140625, "grad_norm": 0.015357979573309422, "kl": 0.056073543906677514, "learning_rate": 7.304742261748848e-07, "loss": 0.0135, "num_tokens": 217175209.0, "reward": 1.137939453125, "reward_std": 1.024681806564331, "rewards/code_complexity_reward/mean": 0.49052733182907104, "rewards/code_complexity_reward/std": 0.4141729772090912, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.296875, "rewards/code_syntax_reward/std": 0.24580632150173187, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 681, "step_time": 60.27883082255721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7578125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 489.580078125, "completions/mean_terminated_length": 419.4273986816406, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.39601021027192473, "epoch": 0.7776510832383124, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.014755605719983578, "kl": 0.05107360990950838, "learning_rate": 7.23456494206859e-07, "loss": 0.0093, "num_tokens": 217485538.0, "reward": 1.015625, "reward_std": 1.0184212923049927, "rewards/code_complexity_reward/mean": 0.4560546576976776, "rewards/code_complexity_reward/std": 0.42886605858802795, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.2685546875, "rewards/code_syntax_reward/std": 0.2495543211698532, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 682, "step_time": 54.36926992982626 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.701171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 484.28515625, "completions/mean_terminated_length": 419.2549133300781, "completions/min_length": 245.0, "completions/min_terminated_length": 245.0, "entropy": 0.3960731280967593, "epoch": 0.7787913340935005, "frac_reward_zero_std": 0.1875, "grad_norm": 0.01678909920156002, "kl": 0.05366974661592394, "learning_rate": 7.164669279688846e-07, "loss": 0.0155, "num_tokens": 217792796.0, "reward": 1.0502440929412842, "reward_std": 0.9859166145324707, "rewards/code_complexity_reward/mean": 0.48359373211860657, "rewards/code_complexity_reward/std": 0.4203650951385498, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.2890625, "rewards/code_syntax_reward/std": 0.24717088043689728, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 683, "step_time": 81.56248586531729 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.71484375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 485.658203125, "completions/mean_terminated_length": 419.623291015625, "completions/min_length": 234.0, "completions/min_terminated_length": 234.0, "entropy": 0.38224842213094234, "epoch": 0.7799315849486887, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.015897270292043686, "kl": 0.050408363342285156, "learning_rate": 7.095056382751559e-07, "loss": 0.0111, "num_tokens": 218100985.0, "reward": 1.013671875, "reward_std": 0.9727644324302673, "rewards/code_complexity_reward/mean": 0.4755859375, "rewards/code_complexity_reward/std": 0.4208016097545624, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.2841796875, "rewards/code_syntax_reward/std": 0.24789467453956604, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 684, "step_time": 64.63091274630278 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.73046875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 488.484375, "completions/mean_terminated_length": 424.7536315917969, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "entropy": 0.4004338518716395, "epoch": 0.7810718358038768, "frac_reward_zero_std": 0.203125, "grad_norm": 0.016971273347735405, "kl": 0.049644947750493884, "learning_rate": 7.025727354915655e-07, "loss": 0.0087, "num_tokens": 218412353.0, "reward": 1.0832030773162842, "reward_std": 1.0172442197799683, "rewards/code_complexity_reward/mean": 0.47871091961860657, "rewards/code_complexity_reward/std": 0.4211575388908386, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.2861328125, "rewards/code_syntax_reward/std": 0.2476169914007187, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 685, "step_time": 46.4555323086679 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.70703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 482.994140625, "completions/mean_terminated_length": 412.99334716796875, "completions/min_length": 194.0, "completions/min_terminated_length": 194.0, "entropy": 0.3969178479164839, "epoch": 0.7822120866590649, "frac_reward_zero_std": 0.21875, "grad_norm": 0.015953587368130684, "kl": 0.05271007918054238, "learning_rate": 6.956683295339483e-07, "loss": 0.0129, "num_tokens": 218718170.0, "reward": 1.0958983898162842, "reward_std": 0.9927342534065247, "rewards/code_complexity_reward/mean": 0.49921873211860657, "rewards/code_complexity_reward/std": 0.4186449348926544, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.2978515625, "rewards/code_syntax_reward/std": 0.24561770260334015, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 686, "step_time": 46.28185324091464 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.705078125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 483.837890625, "completions/mean_terminated_length": 416.5099182128906, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "entropy": 0.39332465128973126, "epoch": 0.7833523375142531, "frac_reward_zero_std": 0.125, "grad_norm": 0.017778996378183365, "kl": 0.053136955946683884, "learning_rate": 6.887925298663506e-07, "loss": 0.0138, "num_tokens": 219024687.0, "reward": 1.1418944597244263, "reward_std": 1.0122236013412476, "rewards/code_complexity_reward/mean": 0.508105456829071, "rewards/code_complexity_reward/std": 0.4182571768760681, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.3017578125, "rewards/code_syntax_reward/std": 0.24482278525829315, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 687, "step_time": 44.909415762871504 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.708984375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 483.521484375, "completions/mean_terminated_length": 414.14093017578125, "completions/min_length": 233.0, "completions/min_terminated_length": 233.0, "entropy": 0.38640962028875947, "epoch": 0.7844925883694412, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.01570132188498974, "kl": 0.05018667422700673, "learning_rate": 6.81945445499281e-07, "loss": 0.0041, "num_tokens": 219333734.0, "reward": 1.167822241783142, "reward_std": 1.013814091682434, "rewards/code_complexity_reward/mean": 0.5103515386581421, "rewards/code_complexity_reward/std": 0.4106343984603882, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.3076171875, "rewards/code_syntax_reward/std": 0.24350784718990326, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 688, "step_time": 53.60374896321446 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 480.865234375, "completions/mean_terminated_length": 409.8141174316406, "completions/min_length": 191.0, "completions/min_terminated_length": 191.0, "entropy": 0.38999371975660324, "epoch": 0.7856328392246295, "frac_reward_zero_std": 0.265625, "grad_norm": 0.016077054664492607, "kl": 0.05333544185850769, "learning_rate": 6.751271849879959e-07, "loss": 0.0141, "num_tokens": 219640397.0, "reward": 1.13623046875, "reward_std": 0.9881840348243713, "rewards/code_complexity_reward/mean": 0.52099609375, "rewards/code_complexity_reward/std": 0.41797149181365967, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.30859375, "rewards/code_syntax_reward/std": 0.2432742565870285, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 689, "step_time": 54.76593547128141 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.72265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.767578125, "completions/mean_terminated_length": 417.4154968261719, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.4034030809998512, "epoch": 0.7867730900798175, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.01667516492307186, "kl": 0.053170416271314025, "learning_rate": 6.68337856430766e-07, "loss": 0.0126, "num_tokens": 219948294.0, "reward": 1.1249022483825684, "reward_std": 0.9996183514595032, "rewards/code_complexity_reward/mean": 0.505566418170929, "rewards/code_complexity_reward/std": 0.4136412739753723, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.3037109375, "rewards/code_syntax_reward/std": 0.24440090358257294, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 690, "step_time": 54.96190141327679 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.693359375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 484.109375, "completions/mean_terminated_length": 421.0445861816406, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.40032680332660675, "epoch": 0.7879133409350056, "frac_reward_zero_std": 0.21875, "grad_norm": 0.015112584456801414, "kl": 0.05479734594700858, "learning_rate": 6.615775674671706e-07, "loss": 0.0115, "num_tokens": 220255470.0, "reward": 1.0794920921325684, "reward_std": 1.006266713142395, "rewards/code_complexity_reward/mean": 0.4886718690395355, "rewards/code_complexity_reward/std": 0.4248731732368469, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.2880859375, "rewards/code_syntax_reward/std": 0.24732354283332825, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 691, "step_time": 59.54866865091026 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.70703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.001953125, "completions/mean_terminated_length": 419.8466796875, "completions/min_length": 264.0, "completions/min_terminated_length": 264.0, "entropy": 0.39124689530581236, "epoch": 0.7890535917901939, "frac_reward_zero_std": 0.171875, "grad_norm": 0.015113532543182373, "kl": 0.05254792713094503, "learning_rate": 6.548464252763906e-07, "loss": 0.0121, "num_tokens": 220564871.0, "reward": 1.1251953840255737, "reward_std": 1.0017842054367065, "rewards/code_complexity_reward/mean": 0.500195324420929, "rewards/code_complexity_reward/std": 0.41297444701194763, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.30078125, "rewards/code_syntax_reward/std": 0.2450276017189026, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 692, "step_time": 54.82108165230602 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6484375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 479.427734375, "completions/mean_terminated_length": 419.3500061035156, "completions/min_length": 203.0, "completions/min_terminated_length": 203.0, "entropy": 0.39479596773162484, "epoch": 0.790193842645382, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.016011420637369156, "kl": 0.05258695757947862, "learning_rate": 6.481445365755021e-07, "loss": 0.0078, "num_tokens": 220868878.0, "reward": 1.1416993141174316, "reward_std": 0.9994540810585022, "rewards/code_complexity_reward/mean": 0.5225585699081421, "rewards/code_complexity_reward/std": 0.4226047992706299, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.306640625, "rewards/code_syntax_reward/std": 0.24373729526996613, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 693, "step_time": 44.76687756925821 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.69921875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 483.41796875, "completions/mean_terminated_length": 416.9740295410156, "completions/min_length": 190.0, "completions/min_terminated_length": 190.0, "entropy": 0.3947923807427287, "epoch": 0.7913340935005702, "frac_reward_zero_std": 0.171875, "grad_norm": 0.01645997352898121, "kl": 0.05408213206101209, "learning_rate": 6.414720076177958e-07, "loss": 0.0125, "num_tokens": 221176248.0, "reward": 1.1321289539337158, "reward_std": 1.0126731395721436, "rewards/code_complexity_reward/mean": 0.507128894329071, "rewards/code_complexity_reward/std": 0.42024704813957214, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.30078125, "rewards/code_syntax_reward/std": 0.2450276017189026, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 694, "step_time": 54.53674156963825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.712890625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 489.130859375, "completions/mean_terminated_length": 432.346923828125, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "entropy": 0.3920977753587067, "epoch": 0.7924743443557583, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.014870732091367245, "kl": 0.05286173988133669, "learning_rate": 6.348289441910807e-07, "loss": 0.0132, "num_tokens": 221486467.0, "reward": 1.1363282203674316, "reward_std": 1.0181660652160645, "rewards/code_complexity_reward/mean": 0.5005859136581421, "rewards/code_complexity_reward/std": 0.4161604046821594, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.2998046875, "rewards/code_syntax_reward/std": 0.2452283650636673, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 695, "step_time": 45.49142815824598 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 495.16796875, "completions/mean_terminated_length": 429.1346435546875, "completions/min_length": 244.0, "completions/min_terminated_length": 244.0, "entropy": 0.387673185672611, "epoch": 0.7936145952109465, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.01696641743183136, "kl": 0.05434558220440522, "learning_rate": 6.282154516160157e-07, "loss": 0.0106, "num_tokens": 221800005.0, "reward": 1.035546898841858, "reward_std": 1.0026755332946777, "rewards/code_complexity_reward/mean": 0.4623046815395355, "rewards/code_complexity_reward/std": 0.4173522889614105, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.2802734375, "rewards/code_syntax_reward/std": 0.24840296804904938, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 696, "step_time": 45.5534935714677 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.771484375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 493.083984375, "completions/mean_terminated_length": 429.22222900390625, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.3908200184814632, "epoch": 0.7947548460661346, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.01314042042940855, "kl": 0.051311654679011554, "learning_rate": 6.216316347444362e-07, "loss": 0.003, "num_tokens": 222112248.0, "reward": 1.005859375, "reward_std": 0.985261857509613, "rewards/code_complexity_reward/mean": 0.46074217557907104, "rewards/code_complexity_reward/std": 0.4204668700695038, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.2763671875, "rewards/code_syntax_reward/std": 0.2488487958908081, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 697, "step_time": 50.224896639585495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 481.6953125, "completions/mean_terminated_length": 409.9210510253906, "completions/min_length": 167.0, "completions/min_terminated_length": 167.0, "entropy": 0.38554739439859986, "epoch": 0.7958950969213227, "frac_reward_zero_std": 0.1875, "grad_norm": 0.01699039340019226, "kl": 0.050249580293893814, "learning_rate": 6.150775979576906e-07, "loss": 0.0129, "num_tokens": 222420908.0, "reward": 1.0936036109924316, "reward_std": 1.0179260969161987, "rewards/code_complexity_reward/mean": 0.49208980798721313, "rewards/code_complexity_reward/std": 0.4253016710281372, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.2900390625, "rewards/code_syntax_reward/std": 0.24701425433158875, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 698, "step_time": 49.982982876710594 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.765625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 487.583984375, "completions/mean_terminated_length": 407.82501220703125, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "entropy": 0.38400587951764464, "epoch": 0.7970353477765109, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.01428379025310278, "kl": 0.05285971541889012, "learning_rate": 6.085534451649905e-07, "loss": 0.0053, "num_tokens": 222729795.0, "reward": 1.0785155296325684, "reward_std": 1.0185737609863281, "rewards/code_complexity_reward/mean": 0.4691406190395355, "rewards/code_complexity_reward/std": 0.41551944613456726, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.28515625, "rewards/code_syntax_reward/std": 0.24775780737400055, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 699, "step_time": 54.11837884876877 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.673828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 478.5546875, "completions/mean_terminated_length": 409.4610900878906, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "entropy": 0.3928482630290091, "epoch": 0.798175598631699, "frac_reward_zero_std": 0.203125, "grad_norm": 0.015902472659945488, "kl": 0.05406068341108039, "learning_rate": 6.020592798017554e-07, "loss": 0.0092, "num_tokens": 223034515.0, "reward": 1.131445288658142, "reward_std": 0.9783955216407776, "rewards/code_complexity_reward/mean": 0.5240234136581421, "rewards/code_complexity_reward/std": 0.4164182245731354, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.310546875, "rewards/code_syntax_reward/std": 0.24279458820819855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 700, "step_time": 53.00152881536633 }, { "epoch": 0.798175598631699, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.7275, "eval_completions/max_length": 511.98, "eval_completions/max_terminated_length": 365.34, "eval_completions/mean_length": 485.2225, "eval_completions/mean_terminated_length": 337.5509539794922, "eval_completions/min_length": 411.36, "eval_completions/min_terminated_length": 308.96, "eval_entropy": 0.4001522672176361, "eval_frac_reward_zero_std": 0.26, "eval_kl": 0.05241265408694744, "eval_loss": 0.006237081717699766, "eval_num_tokens": 223034515.0, "eval_reward": 1.0176875081658363, "eval_reward_std": 0.7935651442408562, "eval_rewards/code_complexity_reward/mean": 0.469874999076128, "eval_rewards/code_complexity_reward/std": 0.35141404539346693, "eval_rewards/code_execution_reward/mean": 0.2675, "eval_rewards/code_execution_reward/std": 0.3057731783390045, "eval_rewards/code_syntax_reward/mean": 0.28, "eval_rewards/code_syntax_reward/std": 0.20919544011354446, "eval_rewards/reasoning_present_reward_func/mean": 0.0, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.0003125, "eval_rewards/xmlcount_reward_func/std": 0.000883883461356163, "eval_runtime": 1286.1704, "eval_samples_per_second": 0.078, "eval_steps_per_second": 0.01, "step": 700 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.724609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.30078125, "completions/mean_terminated_length": 415.04962158203125, "completions/min_length": 224.0, "completions/min_terminated_length": 224.0, "entropy": 0.39873423846438527, "epoch": 0.7993158494868872, "frac_reward_zero_std": 0.21875, "grad_norm": 0.01544927153736353, "kl": 0.05055102542974055, "learning_rate": 5.955952048279795e-07, "loss": 0.0142, "num_tokens": 223345425.0, "reward": 0.9870116710662842, "reward_std": 1.0133802890777588, "rewards/code_complexity_reward/mean": 0.44501951336860657, "rewards/code_complexity_reward/std": 0.4290224611759186, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.2626953125, "rewards/code_syntax_reward/std": 0.24992163479328156, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 701, "step_time": 49.97776315547526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.76171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 490.494140625, "completions/mean_terminated_length": 421.7458801269531, "completions/min_length": 225.0, "completions/min_terminated_length": 225.0, "entropy": 0.39308557426556945, "epoch": 0.8004561003420753, "frac_reward_zero_std": 0.234375, "grad_norm": 0.014243941754102707, "kl": 0.052229501598048955, "learning_rate": 5.891613227265971e-07, "loss": 0.0103, "num_tokens": 223656246.0, "reward": 0.9909180402755737, "reward_std": 0.9890597462654114, "rewards/code_complexity_reward/mean": 0.45771485567092896, "rewards/code_complexity_reward/std": 0.4254778027534485, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.271484375, "rewards/code_syntax_reward/std": 0.24931873381137848, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 702, "step_time": 51.812333134002984 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.669921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 482.697265625, "completions/mean_terminated_length": 423.224853515625, "completions/min_length": 233.0, "completions/min_terminated_length": 233.0, "entropy": 0.38137213233858347, "epoch": 0.8015963511972634, "frac_reward_zero_std": 0.109375, "grad_norm": 0.018179258331656456, "kl": 0.05313341360306367, "learning_rate": 5.827577355018577e-07, "loss": 0.0157, "num_tokens": 223962055.0, "reward": 1.1796386241912842, "reward_std": 0.9590778946876526, "rewards/code_complexity_reward/mean": 0.5419921875, "rewards/code_complexity_reward/std": 0.403936505317688, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.326171875, "rewards/code_syntax_reward/std": 0.23834596574306488, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 703, "step_time": 53.4166681682691 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.708984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 483.18359375, "completions/mean_terminated_length": 412.9798583984375, "completions/min_length": 227.0, "completions/min_terminated_length": 227.0, "entropy": 0.3965249736793339, "epoch": 0.8027366020524516, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.0169968381524086, "kl": 0.051697729737497866, "learning_rate": 5.763845446777077e-07, "loss": 0.0127, "num_tokens": 224268021.0, "reward": 1.0586915016174316, "reward_std": 0.9733923673629761, "rewards/code_complexity_reward/mean": 0.4874023199081421, "rewards/code_complexity_reward/std": 0.4130958914756775, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.2958984375, "rewards/code_syntax_reward/std": 0.24599088728427887, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 704, "step_time": 45.60604127403349 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.443359375, "completions/mean_terminated_length": 418.53570556640625, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 0.3928482015617192, "epoch": 0.8038768529076397, "frac_reward_zero_std": 0.203125, "grad_norm": 0.014574069529771805, "kl": 0.05153268453432247, "learning_rate": 5.700418512961825e-07, "loss": 0.0068, "num_tokens": 224575692.0, "reward": 1.0908691883087158, "reward_std": 0.9796090722084045, "rewards/code_complexity_reward/mean": 0.49199217557907104, "rewards/code_complexity_reward/std": 0.40988799929618835, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.2998046875, "rewards/code_syntax_reward/std": 0.2452283650636673, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 705, "step_time": 56.6852987492457 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.697265625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 482.95703125, "completions/mean_terminated_length": 416.06451416015625, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.39033042965456843, "epoch": 0.8050171037628279, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.014870263636112213, "kl": 0.05386444600299001, "learning_rate": 5.637297559158067e-07, "loss": 0.0067, "num_tokens": 224881786.0, "reward": 1.1441407203674316, "reward_std": 1.0044397115707397, "rewards/code_complexity_reward/mean": 0.5030273199081421, "rewards/code_complexity_reward/std": 0.41052162647247314, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.3046875, "rewards/code_syntax_reward/std": 0.24418380856513977, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 706, "step_time": 53.729901489801705 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.697265625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 483.708984375, "completions/mean_terminated_length": 418.5483703613281, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "entropy": 0.4024571473710239, "epoch": 0.806157354618016, "frac_reward_zero_std": 0.21875, "grad_norm": 0.014733821153640747, "kl": 0.0521113978466019, "learning_rate": 5.574483586099924e-07, "loss": 0.0071, "num_tokens": 225189813.0, "reward": 1.1494629383087158, "reward_std": 1.0191588401794434, "rewards/code_complexity_reward/mean": 0.503710925579071, "rewards/code_complexity_reward/std": 0.4151790738105774, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.3017578125, "rewards/code_syntax_reward/std": 0.24482278525829315, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 707, "step_time": 55.79963883943856 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.724609375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 485.89453125, "completions/mean_terminated_length": 417.2056579589844, "completions/min_length": 246.0, "completions/min_terminated_length": 246.0, "entropy": 0.40118865156546235, "epoch": 0.8072976054732041, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.0193207748234272, "kl": 0.057617355952970684, "learning_rate": 5.511977589654601e-07, "loss": 0.0084, "num_tokens": 225498175.0, "reward": 1.068359375, "reward_std": 0.976191520690918, "rewards/code_complexity_reward/mean": 0.4931640625, "rewards/code_complexity_reward/std": 0.41680940985679626, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.2958984375, "rewards/code_syntax_reward/std": 0.24599088728427887, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 708, "step_time": 54.75758449733257 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.693359375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 479.841796875, "completions/mean_terminated_length": 407.12738037109375, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.3879412072710693, "epoch": 0.8084378563283923, "frac_reward_zero_std": 0.21875, "grad_norm": 0.01576506532728672, "kl": 0.054811683716252446, "learning_rate": 5.449780560806572e-07, "loss": 0.0124, "num_tokens": 225803542.0, "reward": 1.17041015625, "reward_std": 1.0149751901626587, "rewards/code_complexity_reward/mean": 0.51904296875, "rewards/code_complexity_reward/std": 0.4174562394618988, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.3076171875, "rewards/code_syntax_reward/std": 0.24350784718990326, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 709, "step_time": 46.84987996518612 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65234375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 477.05859375, "completions/mean_terminated_length": 411.494384765625, "completions/min_length": 247.0, "completions/min_terminated_length": 247.0, "entropy": 0.38619173131883144, "epoch": 0.8095781071835804, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.013243460096418858, "kl": 0.053650699788704515, "learning_rate": 5.387893485641862e-07, "loss": 0.0025, "num_tokens": 226105060.0, "reward": 1.216552734375, "reward_std": 1.014076828956604, "rewards/code_complexity_reward/mean": 0.53564453125, "rewards/code_complexity_reward/std": 0.413168728351593, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.3173828125, "rewards/code_syntax_reward/std": 0.24098336696624756, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 710, "step_time": 59.77332477271557 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.716796875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 484.25, "completions/mean_terminated_length": 414.0137939453125, "completions/min_length": 164.0, "completions/min_terminated_length": 164.0, "entropy": 0.3862016936764121, "epoch": 0.8107183580387686, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.01773235760629177, "kl": 0.053550231095869094, "learning_rate": 5.326317345332416e-07, "loss": 0.0117, "num_tokens": 226410512.0, "reward": 1.092431664466858, "reward_std": 0.9818821549415588, "rewards/code_complexity_reward/mean": 0.49873045086860657, "rewards/code_complexity_reward/std": 0.4134993851184845, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.2998046875, "rewards/code_syntax_reward/std": 0.2452283650636673, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 711, "step_time": 53.782075569964945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.658203125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 475.0703125, "completions/mean_terminated_length": 403.95428466796875, "completions/min_length": 222.0, "completions/min_terminated_length": 222.0, "entropy": 0.39741548197343946, "epoch": 0.8118586088939567, "frac_reward_zero_std": 0.171875, "grad_norm": 0.015603304840624332, "kl": 0.05468733294401318, "learning_rate": 5.26505311612055e-07, "loss": 0.0117, "num_tokens": 226713132.0, "reward": 1.212255835533142, "reward_std": 0.9799318909645081, "rewards/code_complexity_reward/mean": 0.5501953363418579, "rewards/code_complexity_reward/std": 0.40656304359436035, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.3271484375, "rewards/code_syntax_reward/std": 0.2380310446023941, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 712, "step_time": 53.82113594003022 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7109375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.720703125, "completions/mean_terminated_length": 417.6283874511719, "completions/min_length": 257.0, "completions/min_terminated_length": 257.0, "entropy": 0.3984707738272846, "epoch": 0.8129988597491448, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.014280908741056919, "kl": 0.05607636185595766, "learning_rate": 5.204101769303474e-07, "loss": 0.0095, "num_tokens": 227019569.0, "reward": 1.1526367664337158, "reward_std": 1.019956350326538, "rewards/code_complexity_reward/mean": 0.509082019329071, "rewards/code_complexity_reward/std": 0.4197550117969513, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.3017578125, "rewards/code_syntax_reward/std": 0.24482278525829315, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 713, "step_time": 46.709317764267325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.642578125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 475.513671875, "completions/mean_terminated_length": 409.91802978515625, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 0.3754503121599555, "epoch": 0.814139110604333, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.015146980062127113, "kl": 0.05197082500671968, "learning_rate": 5.143464271217876e-07, "loss": 0.0126, "num_tokens": 227320552.0, "reward": 1.2787108421325684, "reward_std": 0.9924598336219788, "rewards/code_complexity_reward/mean": 0.5726562738418579, "rewards/code_complexity_reward/std": 0.407949835062027, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.3349609375, "rewards/code_syntax_reward/std": 0.23535043001174927, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 714, "step_time": 54.473797897808254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.685546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 481.94140625, "completions/mean_terminated_length": 416.4099426269531, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "entropy": 0.3725440693087876, "epoch": 0.8152793614595211, "frac_reward_zero_std": 0.1875, "grad_norm": 0.014583363197743893, "kl": 0.05409052694449201, "learning_rate": 5.083141583224627e-07, "loss": 0.0084, "num_tokens": 227625798.0, "reward": 1.251953125, "reward_std": 1.0203487873077393, "rewards/code_complexity_reward/mean": 0.541015625, "rewards/code_complexity_reward/std": 0.4096408486366272, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.322265625, "rewards/code_syntax_reward/std": 0.23956161737442017, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 715, "step_time": 57.81047624628991 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.78515625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 491.666015625, "completions/mean_terminated_length": 417.3545227050781, "completions/min_length": 243.0, "completions/min_terminated_length": 243.0, "entropy": 0.39507920388132334, "epoch": 0.8164196123147093, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.018410516902804375, "kl": 0.054856409318745136, "learning_rate": 5.023134661693518e-07, "loss": 0.0113, "num_tokens": 227940623.0, "reward": 0.9937012195587158, "reward_std": 1.0101131200790405, "rewards/code_complexity_reward/mean": 0.44169920682907104, "rewards/code_complexity_reward/std": 0.41956138610839844, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.2666015625, "rewards/code_syntax_reward/std": 0.2496921271085739, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 716, "step_time": 54.08116790652275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.74609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.783203125, "completions/mean_terminated_length": 408.74615478515625, "completions/min_length": 232.0, "completions/min_terminated_length": 232.0, "entropy": 0.3857972901314497, "epoch": 0.8175598631698974, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.015846913680434227, "kl": 0.051738182082772255, "learning_rate": 4.963444457988109e-07, "loss": 0.0125, "num_tokens": 228249164.0, "reward": 1.07958984375, "reward_std": 0.9902106523513794, "rewards/code_complexity_reward/mean": 0.49560546875, "rewards/code_complexity_reward/std": 0.42008668184280396, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.294921875, "rewards/code_syntax_reward/std": 0.24617145955562592, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 717, "step_time": 54.82619623746723 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.67578125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 480.236328125, "completions/mean_terminated_length": 414.03009033203125, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "entropy": 0.3910762704908848, "epoch": 0.8187001140250855, "frac_reward_zero_std": 0.171875, "grad_norm": 0.019458385184407234, "kl": 0.05516119598178193, "learning_rate": 4.904071918450643e-07, "loss": 0.0139, "num_tokens": 228554605.0, "reward": 1.2401854991912842, "reward_std": 0.9980644583702087, "rewards/code_complexity_reward/mean": 0.5458984375, "rewards/code_complexity_reward/std": 0.4064219892024994, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.326171875, "rewards/code_syntax_reward/std": 0.23834596574306488, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 718, "step_time": 45.202755889855325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.69140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.072265625, "completions/mean_terminated_length": 421.5, "completions/min_length": 227.0, "completions/min_terminated_length": 227.0, "entropy": 0.3945623622275889, "epoch": 0.8198403648802737, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.016130974516272545, "kl": 0.055129102896898985, "learning_rate": 4.84501798438704e-07, "loss": 0.0112, "num_tokens": 228861914.0, "reward": 1.1759765148162842, "reward_std": 1.0015822649002075, "rewards/code_complexity_reward/mean": 0.52099609375, "rewards/code_complexity_reward/std": 0.4115903675556183, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.3115234375, "rewards/code_syntax_reward/std": 0.24254849553108215, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.00146484375, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 719, "step_time": 50.170029564760625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.693359375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 482.271484375, "completions/mean_terminated_length": 415.05096435546875, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "entropy": 0.390036647208035, "epoch": 0.8209806157354618, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.017380565404891968, "kl": 0.0539783505955711, "learning_rate": 4.78628359205198e-07, "loss": 0.0173, "num_tokens": 229171097.0, "reward": 1.1775389909744263, "reward_std": 0.9832568168640137, "rewards/code_complexity_reward/mean": 0.537890613079071, "rewards/code_complexity_reward/std": 0.4123459756374359, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.3193359375, "rewards/code_syntax_reward/std": 0.24042759835720062, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 720, "step_time": 54.71423668041825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.74609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 488.875, "completions/mean_terminated_length": 420.9230651855469, "completions/min_length": 264.0, "completions/min_terminated_length": 264.0, "entropy": 0.39224279718473554, "epoch": 0.82212086659065, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.015614758245646954, "kl": 0.0533807166502811, "learning_rate": 4.727869672634044e-07, "loss": 0.0065, "num_tokens": 229480533.0, "reward": 1.083984375, "reward_std": 0.9673433899879456, "rewards/code_complexity_reward/mean": 0.5007812976837158, "rewards/code_complexity_reward/std": 0.4115259051322937, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.302734375, "rewards/code_syntax_reward/std": 0.2446138858795166, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 721, "step_time": 50.463237242773175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.744140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 487.197265625, "completions/mean_terminated_length": 415.0610656738281, "completions/min_length": 244.0, "completions/min_terminated_length": 244.0, "entropy": 0.385035565122962, "epoch": 0.8232611174458381, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.016311807557940483, "kl": 0.05249590432504192, "learning_rate": 4.669777152240976e-07, "loss": 0.0131, "num_tokens": 229789870.0, "reward": 1.0458496809005737, "reward_std": 0.9789285063743591, "rewards/code_complexity_reward/mean": 0.47822266817092896, "rewards/code_complexity_reward/std": 0.41429847478866577, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.2900390625, "rewards/code_syntax_reward/std": 0.24701425433158875, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 722, "step_time": 46.180815608240664 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.716796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.935546875, "completions/mean_terminated_length": 419.96551513671875, "completions/min_length": 270.0, "completions/min_terminated_length": 270.0, "entropy": 0.3951973761431873, "epoch": 0.8244013683010262, "frac_reward_zero_std": 0.21875, "grad_norm": 0.01573183946311474, "kl": 0.056315304012969136, "learning_rate": 4.612006951884973e-07, "loss": 0.0125, "num_tokens": 230099353.0, "reward": 1.0544922351837158, "reward_std": 1.012234091758728, "rewards/code_complexity_reward/mean": 0.4708007872104645, "rewards/code_complexity_reward/std": 0.4206415116786957, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.2822265625, "rewards/code_syntax_reward/std": 0.24815665185451508, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 723, "step_time": 59.42537247110158 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.712890625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.302734375, "completions/mean_terminated_length": 415.5306091308594, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.3973926976323128, "epoch": 0.8255416191562144, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.015389705076813698, "kl": 0.05302017356734723, "learning_rate": 4.5545599874681076e-07, "loss": 0.0128, "num_tokens": 230407320.0, "reward": 1.0893065929412842, "reward_std": 0.9984498620033264, "rewards/code_complexity_reward/mean": 0.50244140625, "rewards/code_complexity_reward/std": 0.4275480806827545, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.29296875, "rewards/code_syntax_reward/std": 0.24652054905891418, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 724, "step_time": 60.59444725327194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 482.208984375, "completions/mean_terminated_length": 411.65130615234375, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.3823750982992351, "epoch": 0.8266818700114025, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.014122956432402134, "kl": 0.05478108732495457, "learning_rate": 4.4974371697677847e-07, "loss": 0.0107, "num_tokens": 230712071.0, "reward": 1.1730468273162842, "reward_std": 1.0264582633972168, "rewards/code_complexity_reward/mean": 0.5089843273162842, "rewards/code_complexity_reward/std": 0.41542819142341614, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.3046875, "rewards/code_syntax_reward/std": 0.24418380856513977, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 725, "step_time": 45.79881720524281 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.697265625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 487.7578125, "completions/mean_terminated_length": 431.9225769042969, "completions/min_length": 263.0, "completions/min_terminated_length": 263.0, "entropy": 0.38656838750466704, "epoch": 0.8278221208665907, "frac_reward_zero_std": 0.15625, "grad_norm": 0.014800123870372772, "kl": 0.05200675979722291, "learning_rate": 4.4406394044223174e-07, "loss": 0.0065, "num_tokens": 231021423.0, "reward": 1.130859375, "reward_std": 0.99615079164505, "rewards/code_complexity_reward/mean": 0.5105469226837158, "rewards/code_complexity_reward/std": 0.41441264748573303, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.3046875, "rewards/code_syntax_reward/std": 0.24418380856513977, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 726, "step_time": 45.65118422731757 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.671875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 479.486328125, "completions/mean_terminated_length": 412.9107360839844, "completions/min_length": 234.0, "completions/min_terminated_length": 234.0, "entropy": 0.3825922552496195, "epoch": 0.8289623717217788, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.015166605822741985, "kl": 0.05332154541974887, "learning_rate": 4.384167591916566e-07, "loss": 0.0021, "num_tokens": 231324948.0, "reward": 1.2380859851837158, "reward_std": 0.9849675297737122, "rewards/code_complexity_reward/mean": 0.552539050579071, "rewards/code_complexity_reward/std": 0.4021996855735779, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.33203125, "rewards/code_syntax_reward/std": 0.23638953268527985, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 727, "step_time": 52.759046195074916 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.69921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 482.576171875, "completions/mean_terminated_length": 414.1753234863281, "completions/min_length": 199.0, "completions/min_terminated_length": 199.0, "entropy": 0.3941151350736618, "epoch": 0.830102622576967, "frac_reward_zero_std": 0.1875, "grad_norm": 0.015353125520050526, "kl": 0.05504166369792074, "learning_rate": 4.328022627567657e-07, "loss": 0.0067, "num_tokens": 231631391.0, "reward": 1.1540038585662842, "reward_std": 0.9998867511749268, "rewards/code_complexity_reward/mean": 0.5153319835662842, "rewards/code_complexity_reward/std": 0.4102577567100525, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.310546875, "rewards/code_syntax_reward/std": 0.24279458820819855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 728, "step_time": 54.46339970827103 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.701171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 482.376953125, "completions/mean_terminated_length": 412.8692932128906, "completions/min_length": 216.0, "completions/min_terminated_length": 216.0, "entropy": 0.3809489537961781, "epoch": 0.8312428734321551, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.016855910420417786, "kl": 0.05294360750121996, "learning_rate": 4.2722054015107874e-07, "loss": 0.0136, "num_tokens": 231937792.0, "reward": 1.1337890625, "reward_std": 1.0299434661865234, "rewards/code_complexity_reward/mean": 0.486328125, "rewards/code_complexity_reward/std": 0.4142266809940338, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.2939453125, "rewards/code_syntax_reward/std": 0.24634800851345062, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 729, "step_time": 67.3098726477474 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.673828125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 479.9140625, "completions/mean_terminated_length": 413.6287536621094, "completions/min_length": 190.0, "completions/min_terminated_length": 190.0, "entropy": 0.3811980397440493, "epoch": 0.8323831242873432, "frac_reward_zero_std": 0.15625, "grad_norm": 0.01673085428774357, "kl": 0.05419038067338988, "learning_rate": 4.216716798685125e-07, "loss": 0.0068, "num_tokens": 232242276.0, "reward": 1.19970703125, "reward_std": 0.9548541307449341, "rewards/code_complexity_reward/mean": 0.54833984375, "rewards/code_complexity_reward/std": 0.39733970165252686, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.3330078125, "rewards/code_syntax_reward/std": 0.23604771494865417, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 730, "step_time": 63.39947733748704 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.669921875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 476.486328125, "completions/mean_terminated_length": 404.4082946777344, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "entropy": 0.3850765135139227, "epoch": 0.8335233751425314, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.01532130315899849, "kl": 0.05785072228172794, "learning_rate": 4.161557698819757e-07, "loss": 0.0136, "num_tokens": 232544893.0, "reward": 1.18115234375, "reward_std": 1.0144352912902832, "rewards/code_complexity_reward/mean": 0.515625, "rewards/code_complexity_reward/std": 0.41132453083992004, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.3095703125, "rewards/code_syntax_reward/std": 0.24303650856018066, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 731, "step_time": 53.22309519350529 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.669921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 478.76953125, "completions/mean_terminated_length": 411.325439453125, "completions/min_length": 185.0, "completions/min_terminated_length": 185.0, "entropy": 0.38528412068262696, "epoch": 0.8346636259977195, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.015919726341962814, "kl": 0.05494710337370634, "learning_rate": 4.106728976419763e-07, "loss": 0.007, "num_tokens": 232847183.0, "reward": 1.1507811546325684, "reward_std": 0.9943674206733704, "rewards/code_complexity_reward/mean": 0.5199218392372131, "rewards/code_complexity_reward/std": 0.41308537125587463, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.310546875, "rewards/code_syntax_reward/std": 0.24279458820819855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 732, "step_time": 55.05858478322625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.728515625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 486.15234375, "completions/mean_terminated_length": 416.7913818359375, "completions/min_length": 253.0, "completions/min_terminated_length": 253.0, "entropy": 0.3932869993150234, "epoch": 0.8358038768529077, "frac_reward_zero_std": 0.234375, "grad_norm": 0.014594940468668938, "kl": 0.05342419567750767, "learning_rate": 4.0522315007523486e-07, "loss": 0.0098, "num_tokens": 233157169.0, "reward": 1.085107445716858, "reward_std": 0.9927768111228943, "rewards/code_complexity_reward/mean": 0.4872070550918579, "rewards/code_complexity_reward/std": 0.4123905599117279, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.296875, "rewards/code_syntax_reward/std": 0.24580632150173187, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 733, "step_time": 61.06966927461326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.744140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.54296875, "completions/mean_terminated_length": 412.5038146972656, "completions/min_length": 214.0, "completions/min_terminated_length": 214.0, "entropy": 0.37978535145521164, "epoch": 0.8369441277080958, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.01742478273808956, "kl": 0.05380589491687715, "learning_rate": 3.998066135833031e-07, "loss": 0.0115, "num_tokens": 233465303.0, "reward": 1.138769507408142, "reward_std": 1.018020510673523, "rewards/code_complexity_reward/mean": 0.5020507574081421, "rewards/code_complexity_reward/std": 0.41602054238319397, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.30078125, "rewards/code_syntax_reward/std": 0.2450276017189026, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 734, "step_time": 53.13435816857964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.71484375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.033203125, "completions/mean_terminated_length": 417.4315185546875, "completions/min_length": 232.0, "completions/min_terminated_length": 232.0, "entropy": 0.3872310910373926, "epoch": 0.8380843785632839, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.01721447892487049, "kl": 0.05206582183018327, "learning_rate": 3.944233740412001e-07, "loss": 0.0167, "num_tokens": 233772956.0, "reward": 1.154443383216858, "reward_std": 0.9997142553329468, "rewards/code_complexity_reward/mean": 0.5119140148162842, "rewards/code_complexity_reward/std": 0.4107706546783447, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.3095703125, "rewards/code_syntax_reward/std": 0.24303650856018066, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 735, "step_time": 46.11051477864385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.716796875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 485.935546875, "completions/mean_terminated_length": 419.96551513671875, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "entropy": 0.39431147556751966, "epoch": 0.8392246294184721, "frac_reward_zero_std": 0.1875, "grad_norm": 0.01491142250597477, "kl": 0.05122629157267511, "learning_rate": 3.890735167960455e-07, "loss": 0.0098, "num_tokens": 234080291.0, "reward": 1.04052734375, "reward_std": 0.9718610048294067, "rewards/code_complexity_reward/mean": 0.48095703125, "rewards/code_complexity_reward/std": 0.41557690501213074, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.2900390625, "rewards/code_syntax_reward/std": 0.24701425433158875, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 736, "step_time": 45.079165865667164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.60546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 470.3203125, "completions/mean_terminated_length": 406.3564147949219, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.38188764778897166, "epoch": 0.8403648802736602, "frac_reward_zero_std": 0.21875, "grad_norm": 0.013657448813319206, "kl": 0.055759074282832444, "learning_rate": 3.8375712666570865e-07, "loss": 0.0083, "num_tokens": 234382655.0, "reward": 1.3442871570587158, "reward_std": 0.9919898509979248, "rewards/code_complexity_reward/mean": 0.5865234136581421, "rewards/code_complexity_reward/std": 0.3975881338119507, "rewards/code_execution_reward/mean": 0.41015625, "rewards/code_execution_reward/std": 0.49234291911125183, "rewards/code_syntax_reward/mean": 0.3466796875, "rewards/code_syntax_reward/std": 0.2307749092578888, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 737, "step_time": 45.876152059063315 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.71875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 484.6875, "completions/mean_terminated_length": 414.8888854980469, "completions/min_length": 225.0, "completions/min_terminated_length": 225.0, "entropy": 0.39145630644634366, "epoch": 0.8415051311288484, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.016296887770295143, "kl": 0.05835375376045704, "learning_rate": 3.784742879374631e-07, "loss": 0.0058, "num_tokens": 234689827.0, "reward": 1.0535156726837158, "reward_std": 0.9779653549194336, "rewards/code_complexity_reward/mean": 0.48515623807907104, "rewards/code_complexity_reward/std": 0.41517728567123413, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.29296875, "rewards/code_syntax_reward/std": 0.24652054905891418, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 738, "step_time": 53.421770486049354 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.73828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.845703125, "completions/mean_terminated_length": 408.2462463378906, "completions/min_length": 225.0, "completions/min_terminated_length": 225.0, "entropy": 0.3971648942679167, "epoch": 0.8426453819840365, "frac_reward_zero_std": 0.203125, "grad_norm": 0.01570054329931736, "kl": 0.05474814778426662, "learning_rate": 3.7322508436665184e-07, "loss": 0.0118, "num_tokens": 234996404.0, "reward": 1.0949218273162842, "reward_std": 1.0082504749298096, "rewards/code_complexity_reward/mean": 0.48554688692092896, "rewards/code_complexity_reward/std": 0.41633251309394836, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.29296875, "rewards/code_syntax_reward/std": 0.24652054905891418, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 739, "step_time": 54.502292320132256 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7109375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 485.291015625, "completions/mean_terminated_length": 419.6013488769531, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.3894678223878145, "epoch": 0.8437856328392246, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.01516488566994667, "kl": 0.05146078870166093, "learning_rate": 3.680095991753577e-07, "loss": 0.0036, "num_tokens": 235304265.0, "reward": 1.06689453125, "reward_std": 0.9818572998046875, "rewards/code_complexity_reward/mean": 0.48681640625, "rewards/code_complexity_reward/std": 0.41457900404930115, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.294921875, "rewards/code_syntax_reward/std": 0.24617145955562592, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 740, "step_time": 45.79918297473341 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.736328125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 491.42578125, "completions/mean_terminated_length": 433.9703674316406, "completions/min_length": 226.0, "completions/min_terminated_length": 226.0, "entropy": 0.39614483853802085, "epoch": 0.8449258836944128, "frac_reward_zero_std": 0.171875, "grad_norm": 0.01928062178194523, "kl": 0.05307173420442268, "learning_rate": 3.6282791505108327e-07, "loss": 0.0135, "num_tokens": 235615171.0, "reward": 1.0447266101837158, "reward_std": 0.9864798188209534, "rewards/code_complexity_reward/mean": 0.47050780057907104, "rewards/code_complexity_reward/std": 0.41369572281837463, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.287109375, "rewards/code_syntax_reward/std": 0.24747224152088165, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 741, "step_time": 52.216168120503426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.73046875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 483.927734375, "completions/mean_terminated_length": 407.84783935546875, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.3864088552072644, "epoch": 0.8460661345496009, "frac_reward_zero_std": 0.1875, "grad_norm": 0.015809832140803337, "kl": 0.05468007770832628, "learning_rate": 3.576801141454439e-07, "loss": 0.0077, "num_tokens": 235923594.0, "reward": 1.0965821743011475, "reward_std": 0.9856917262077332, "rewards/code_complexity_reward/mean": 0.49648433923721313, "rewards/code_complexity_reward/std": 0.4114046096801758, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.30078125, "rewards/code_syntax_reward/std": 0.2450276017189026, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 742, "step_time": 63.23762433230877 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.69921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 482.62109375, "completions/mean_terminated_length": 414.3246765136719, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 0.39038194669410586, "epoch": 0.8472063854047891, "frac_reward_zero_std": 0.15625, "grad_norm": 0.01674637384712696, "kl": 0.053377335076220334, "learning_rate": 3.5256627807286086e-07, "loss": 0.0105, "num_tokens": 236232296.0, "reward": 1.1240723133087158, "reward_std": 0.9953796863555908, "rewards/code_complexity_reward/mean": 0.5027344226837158, "rewards/code_complexity_reward/std": 0.4133325517177582, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.302734375, "rewards/code_syntax_reward/std": 0.2446138858795166, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 743, "step_time": 50.767022288404405 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6484375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 477.17578125, "completions/mean_terminated_length": 412.9444580078125, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "entropy": 0.38096662843599916, "epoch": 0.8483466362599772, "frac_reward_zero_std": 0.15625, "grad_norm": 0.01749703846871853, "kl": 0.055073359748348594, "learning_rate": 3.474864879092693e-07, "loss": 0.0141, "num_tokens": 236534974.0, "reward": 1.189550757408142, "reward_std": 0.9804264307022095, "rewards/code_complexity_reward/mean": 0.5401366949081421, "rewards/code_complexity_reward/std": 0.4064371585845947, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.3232421875, "rewards/code_syntax_reward/std": 0.23926427960395813, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 744, "step_time": 57.233612682670355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.689453125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 480.736328125, "completions/mean_terminated_length": 411.3270263671875, "completions/min_length": 232.0, "completions/min_terminated_length": 232.0, "entropy": 0.3914263630285859, "epoch": 0.8494868871151653, "frac_reward_zero_std": 0.15625, "grad_norm": 0.015463405288755894, "kl": 0.05557085241889581, "learning_rate": 3.424408241908336e-07, "loss": 0.015, "num_tokens": 236841463.0, "reward": 1.1837890148162842, "reward_std": 0.9543396234512329, "rewards/code_complexity_reward/mean": 0.5509765148162842, "rewards/code_complexity_reward/std": 0.4036635160446167, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.330078125, "rewards/code_syntax_reward/std": 0.2370595932006836, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 745, "step_time": 52.07427075505257 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.658203125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 480.060546875, "completions/mean_terminated_length": 418.5542907714844, "completions/min_length": 219.0, "completions/min_terminated_length": 219.0, "entropy": 0.3803029339760542, "epoch": 0.8506271379703535, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.015621735714375973, "kl": 0.05648795660817996, "learning_rate": 3.374293669126669e-07, "loss": 0.0129, "num_tokens": 237145818.0, "reward": 1.2497069835662842, "reward_std": 0.9825806021690369, "rewards/code_complexity_reward/mean": 0.5583007335662842, "rewards/code_complexity_reward/std": 0.4021401107311249, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.333984375, "rewards/code_syntax_reward/std": 0.23570136725902557, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 746, "step_time": 61.82877649925649 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.66796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 478.2421875, "completions/mean_terminated_length": 410.32940673828125, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "entropy": 0.3830849980004132, "epoch": 0.8517673888255416, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.016364922747015953, "kl": 0.059159452095627785, "learning_rate": 3.324521955275697e-07, "loss": 0.0059, "num_tokens": 237449750.0, "reward": 1.2366211414337158, "reward_std": 1.0121723413467407, "rewards/code_complexity_reward/mean": 0.5393555164337158, "rewards/code_complexity_reward/std": 0.40758341550827026, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.322265625, "rewards/code_syntax_reward/std": 0.23956161737442017, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 747, "step_time": 53.200981684960425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.724609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.8515625, "completions/mean_terminated_length": 413.4184265136719, "completions/min_length": 224.0, "completions/min_terminated_length": 224.0, "entropy": 0.38400873728096485, "epoch": 0.8529076396807298, "frac_reward_zero_std": 0.171875, "grad_norm": 0.015889598056674004, "kl": 0.053585863613989204, "learning_rate": 3.2750938894476223e-07, "loss": 0.0087, "num_tokens": 237757590.0, "reward": 1.1202149391174316, "reward_std": 1.0016067028045654, "rewards/code_complexity_reward/mean": 0.4979492127895355, "rewards/code_complexity_reward/std": 0.4122876822948456, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.30078125, "rewards/code_syntax_reward/std": 0.2450276017189026, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 748, "step_time": 45.064053698442876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.068359375, "completions/mean_terminated_length": 417.91448974609375, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.39066761545836926, "epoch": 0.8540478905359179, "frac_reward_zero_std": 0.234375, "grad_norm": 0.01497552078217268, "kl": 0.054277699731756, "learning_rate": 3.2260102552863994e-07, "loss": 0.0074, "num_tokens": 238066077.0, "reward": 1.1282227039337158, "reward_std": 0.9987648129463196, "rewards/code_complexity_reward/mean": 0.5088866949081421, "rewards/code_complexity_reward/std": 0.4152594804763794, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.3037109375, "rewards/code_syntax_reward/std": 0.24440090358257294, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 749, "step_time": 58.36955517809838 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.697265625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 486.767578125, "completions/mean_terminated_length": 428.651611328125, "completions/min_length": 267.0, "completions/min_terminated_length": 267.0, "entropy": 0.38534263288602233, "epoch": 0.855188141391106, "frac_reward_zero_std": 0.1875, "grad_norm": 0.01407216489315033, "kl": 0.05423286574659869, "learning_rate": 3.177271830975276e-07, "loss": 0.0075, "num_tokens": 238373442.0, "reward": 1.155126929283142, "reward_std": 0.9843040704727173, "rewards/code_complexity_reward/mean": 0.5337890386581421, "rewards/code_complexity_reward/std": 0.41629114747047424, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.314453125, "rewards/code_syntax_reward/std": 0.2417849749326706, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 750, "step_time": 54.410889892838895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.693359375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 483.9296875, "completions/mean_terminated_length": 420.4586181640625, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.38188255298882723, "epoch": 0.8563283922462942, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.016813868656754494, "kl": 0.054958428954705596, "learning_rate": 3.1288793892244427e-07, "loss": 0.0096, "num_tokens": 238681110.0, "reward": 1.203857421875, "reward_std": 0.981173574924469, "rewards/code_complexity_reward/mean": 0.54345703125, "rewards/code_complexity_reward/std": 0.4070936143398285, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.326171875, "rewards/code_syntax_reward/std": 0.23834596574306488, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 751, "step_time": 53.21880661416799 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7109375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 482.51171875, "completions/mean_terminated_length": 409.98651123046875, "completions/min_length": 194.0, "completions/min_terminated_length": 194.0, "entropy": 0.39405127242207527, "epoch": 0.8574686431014823, "frac_reward_zero_std": 0.171875, "grad_norm": 0.016928303986787796, "kl": 0.05356652312912047, "learning_rate": 3.080833697258842e-07, "loss": 0.0108, "num_tokens": 238988196.0, "reward": 1.1368651390075684, "reward_std": 0.9940820932388306, "rewards/code_complexity_reward/mean": 0.5067383050918579, "rewards/code_complexity_reward/std": 0.4098452031612396, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.3076171875, "rewards/code_syntax_reward/std": 0.24350784718990326, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 752, "step_time": 73.64896343275905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.736328125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.095703125, "completions/mean_terminated_length": 413.75555419921875, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "entropy": 0.40484615648165345, "epoch": 0.8586088939566705, "frac_reward_zero_std": 0.171875, "grad_norm": 0.015342549420893192, "kl": 0.053742498450446874, "learning_rate": 3.0331355168059214e-07, "loss": 0.0104, "num_tokens": 239297069.0, "reward": 1.044921875, "reward_std": 0.951249897480011, "rewards/code_complexity_reward/mean": 0.494140625, "rewards/code_complexity_reward/std": 0.41311055421829224, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.298828125, "rewards/code_syntax_reward/std": 0.2454250603914261, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 753, "step_time": 45.23703632503748 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.658203125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 477.296875, "completions/mean_terminated_length": 410.46856689453125, "completions/min_length": 178.0, "completions/min_terminated_length": 178.0, "entropy": 0.3989123096689582, "epoch": 0.8597491448118586, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.015234585851430893, "kl": 0.0593709553941153, "learning_rate": 2.985785604083649e-07, "loss": 0.0161, "num_tokens": 239602049.0, "reward": 1.1464354991912842, "reward_std": 0.9957582950592041, "rewards/code_complexity_reward/mean": 0.520214855670929, "rewards/code_complexity_reward/std": 0.41763541102409363, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.3076171875, "rewards/code_syntax_reward/std": 0.24350784718990326, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 754, "step_time": 55.09344610851258 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.69140625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 482.93359375, "completions/mean_terminated_length": 417.8101501464844, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.3817577352747321, "epoch": 0.8608893956670467, "frac_reward_zero_std": 0.15625, "grad_norm": 0.015602712519466877, "kl": 0.05660721834283322, "learning_rate": 2.9387847097884254e-07, "loss": 0.0111, "num_tokens": 239907723.0, "reward": 1.21240234375, "reward_std": 0.981231689453125, "rewards/code_complexity_reward/mean": 0.54541015625, "rewards/code_complexity_reward/std": 0.4032444655895233, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.3271484375, "rewards/code_syntax_reward/std": 0.2380310446023941, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 755, "step_time": 45.738210829906166 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6953125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 482.0546875, "completions/mean_terminated_length": 413.71795654296875, "completions/min_length": 249.0, "completions/min_terminated_length": 249.0, "entropy": 0.3871076088398695, "epoch": 0.8620296465222349, "frac_reward_zero_std": 0.15625, "grad_norm": 0.016213195398449898, "kl": 0.0545759589294903, "learning_rate": 2.8921335790832757e-07, "loss": 0.0141, "num_tokens": 240214419.0, "reward": 1.1492187976837158, "reward_std": 0.9848854541778564, "rewards/code_complexity_reward/mean": 0.5212891101837158, "rewards/code_complexity_reward/std": 0.4111649990081787, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.3134765625, "rewards/code_syntax_reward/std": 0.24204368889331818, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 756, "step_time": 59.05572192184627 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.650390625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 476.54296875, "completions/mean_terminated_length": 410.58099365234375, "completions/min_length": 265.0, "completions/min_terminated_length": 265.0, "entropy": 0.38451345870271325, "epoch": 0.863169897377423, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.01755792833864689, "kl": 0.05570023966720328, "learning_rate": 2.845832951585969e-07, "loss": 0.0149, "num_tokens": 240517645.0, "reward": 1.1899902820587158, "reward_std": 0.9631537199020386, "rewards/code_complexity_reward/mean": 0.545214831829071, "rewards/code_complexity_reward/std": 0.40569761395454407, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.328125, "rewards/code_syntax_reward/std": 0.23771169781684875, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 757, "step_time": 53.2402451178059 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.671875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 480.53515625, "completions/mean_terminated_length": 416.1071472167969, "completions/min_length": 208.0, "completions/min_terminated_length": 208.0, "entropy": 0.3878611489199102, "epoch": 0.8643101482326112, "frac_reward_zero_std": 0.203125, "grad_norm": 0.01603812538087368, "kl": 0.05368958553299308, "learning_rate": 2.799883561357314e-07, "loss": 0.0079, "num_tokens": 240824279.0, "reward": 1.23876953125, "reward_std": 1.0083147287368774, "rewards/code_complexity_reward/mean": 0.55029296875, "rewards/code_complexity_reward/std": 0.41351306438446045, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.3232421875, "rewards/code_syntax_reward/std": 0.23926427960395813, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 758, "step_time": 46.07275320775807 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.70703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 483.828125, "completions/mean_terminated_length": 415.8399963378906, "completions/min_length": 176.0, "completions/min_terminated_length": 176.0, "entropy": 0.37611649138852954, "epoch": 0.8654503990877993, "frac_reward_zero_std": 0.265625, "grad_norm": 0.014178668148815632, "kl": 0.05491204746067524, "learning_rate": 2.7542861368895444e-07, "loss": 0.0069, "num_tokens": 241130895.0, "reward": 1.1309571266174316, "reward_std": 0.9966633319854736, "rewards/code_complexity_reward/mean": 0.5064453482627869, "rewards/code_complexity_reward/std": 0.4113095998764038, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.3056640625, "rewards/code_syntax_reward/std": 0.2439626157283783, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 759, "step_time": 55.2064966596663 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.673828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 481.65625, "completions/mean_terminated_length": 418.9700622558594, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "entropy": 0.3829010846093297, "epoch": 0.8665906499429875, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.015198067761957645, "kl": 0.05437722970964387, "learning_rate": 2.709041401094717e-07, "loss": 0.0095, "num_tokens": 241439091.0, "reward": 1.169677734375, "reward_std": 0.9861143231391907, "rewards/code_complexity_reward/mean": 0.52685546875, "rewards/code_complexity_reward/std": 0.40979501605033875, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.31640625, "rewards/code_syntax_reward/std": 0.24125482141971588, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 760, "step_time": 45.800663580186665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.671875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 476.998046875, "completions/mean_terminated_length": 405.327392578125, "completions/min_length": 226.0, "completions/min_terminated_length": 226.0, "entropy": 0.37549165775999427, "epoch": 0.8677309007981756, "frac_reward_zero_std": 0.203125, "grad_norm": 0.015970731154084206, "kl": 0.05796687625115737, "learning_rate": 2.664150071293314e-07, "loss": 0.0122, "num_tokens": 241740802.0, "reward": 1.1647460460662842, "reward_std": 1.0133320093154907, "rewards/code_complexity_reward/mean": 0.510449230670929, "rewards/code_complexity_reward/std": 0.41322699189186096, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.306640625, "rewards/code_syntax_reward/std": 0.24373729526996613, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 761, "step_time": 52.57255509868264 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.64453125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 477.634765625, "completions/mean_terminated_length": 415.3241882324219, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.3927794168703258, "epoch": 0.8688711516533637, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.016938364133238792, "kl": 0.059988126915413886, "learning_rate": 2.619612859202808e-07, "loss": 0.0128, "num_tokens": 242043503.0, "reward": 1.195654273033142, "reward_std": 0.9796932339668274, "rewards/code_complexity_reward/mean": 0.5414062738418579, "rewards/code_complexity_reward/std": 0.4050976037979126, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.3251953125, "rewards/code_syntax_reward/std": 0.23865646123886108, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 762, "step_time": 54.11125605739653 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.685546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 479.9921875, "completions/mean_terminated_length": 410.211181640625, "completions/min_length": 208.0, "completions/min_terminated_length": 208.0, "entropy": 0.3909789430908859, "epoch": 0.8700114025085519, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.016370372846722603, "kl": 0.055301724933087826, "learning_rate": 2.575430470926421e-07, "loss": 0.0081, "num_tokens": 242347451.0, "reward": 1.18310546875, "reward_std": 0.9835549592971802, "rewards/code_complexity_reward/mean": 0.53466796875, "rewards/code_complexity_reward/std": 0.4085249602794647, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.3203125, "rewards/code_syntax_reward/std": 0.24014326930046082, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 763, "step_time": 45.31834208685905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7421875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 486.62890625, "completions/mean_terminated_length": 413.5909118652344, "completions/min_length": 198.0, "completions/min_terminated_length": 198.0, "entropy": 0.39811359578743577, "epoch": 0.87115165336374, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.017084963619709015, "kl": 0.05441714689368382, "learning_rate": 2.531603606941929e-07, "loss": 0.0107, "num_tokens": 242655741.0, "reward": 1.0770020484924316, "reward_std": 0.9702016115188599, "rewards/code_complexity_reward/mean": 0.4976562261581421, "rewards/code_complexity_reward/std": 0.4126597046852112, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.2998046875, "rewards/code_syntax_reward/std": 0.2452283650636673, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 764, "step_time": 53.80175864137709 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.765625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 488.208984375, "completions/mean_terminated_length": 410.49169921875, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.3989649033173919, "epoch": 0.8722919042189282, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.01594815030694008, "kl": 0.05696057161549106, "learning_rate": 2.4881329620905144e-07, "loss": 0.0095, "num_tokens": 242965456.0, "reward": 0.973583996295929, "reward_std": 0.9729722142219543, "rewards/code_complexity_reward/mean": 0.45283204317092896, "rewards/code_complexity_reward/std": 0.42072853446006775, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.2724609375, "rewards/code_syntax_reward/std": 0.2492324709892273, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 765, "step_time": 45.39946374669671 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.595703125, "completions/mean_terminated_length": 419.6907958984375, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 0.3957516518421471, "epoch": 0.8734321550741163, "frac_reward_zero_std": 0.1875, "grad_norm": 0.017499810084700584, "kl": 0.0560564745683223, "learning_rate": 2.4450192255658115e-07, "loss": 0.0085, "num_tokens": 243274025.0, "reward": 1.0872559547424316, "reward_std": 0.988495409488678, "rewards/code_complexity_reward/mean": 0.4922851324081421, "rewards/code_complexity_reward/std": 0.41371312737464905, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.2978515625, "rewards/code_syntax_reward/std": 0.24561770260334015, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 766, "step_time": 54.18279389757663 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.669921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 480.93359375, "completions/mean_terminated_length": 417.88165283203125, "completions/min_length": 173.0, "completions/min_terminated_length": 173.0, "entropy": 0.3875516033731401, "epoch": 0.8745724059293044, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.016329532489180565, "kl": 0.05782750074286014, "learning_rate": 2.402263080902917e-07, "loss": 0.0087, "num_tokens": 243579047.0, "reward": 1.1800780296325684, "reward_std": 0.9685280919075012, "rewards/code_complexity_reward/mean": 0.5443359613418579, "rewards/code_complexity_reward/std": 0.4063791036605835, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.3251953125, "rewards/code_syntax_reward/std": 0.23865646123886108, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 767, "step_time": 45.51672369334847 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.708984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 480.75, "completions/mean_terminated_length": 404.6174621582031, "completions/min_length": 210.0, "completions/min_terminated_length": 210.0, "entropy": 0.3984620333649218, "epoch": 0.8757126567844926, "frac_reward_zero_std": 0.203125, "grad_norm": 0.01740574836730957, "kl": 0.05527364619774744, "learning_rate": 2.359865205967618e-07, "loss": 0.0061, "num_tokens": 243883835.0, "reward": 1.079345703125, "reward_std": 0.9907791018486023, "rewards/code_complexity_reward/mean": 0.4931640625, "rewards/code_complexity_reward/std": 0.4184260666370392, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.294921875, "rewards/code_syntax_reward/std": 0.24617145955562592, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 768, "step_time": 53.30746115371585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.69921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 483.345703125, "completions/mean_terminated_length": 416.7337646484375, "completions/min_length": 171.0, "completions/min_terminated_length": 171.0, "entropy": 0.3894861415028572, "epoch": 0.8768529076396807, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.01647273451089859, "kl": 0.05299527238821611, "learning_rate": 2.317826272945578e-07, "loss": 0.0172, "num_tokens": 244191484.0, "reward": 1.1284668445587158, "reward_std": 1.0013964176177979, "rewards/code_complexity_reward/mean": 0.504199206829071, "rewards/code_complexity_reward/std": 0.41250863671302795, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.3037109375, "rewards/code_syntax_reward/std": 0.24440090358257294, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 769, "step_time": 66.09144799411297 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.697265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 481.62109375, "completions/mean_terminated_length": 411.651611328125, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 0.3919669738970697, "epoch": 0.8779931584948689, "frac_reward_zero_std": 0.203125, "grad_norm": 0.015842720866203308, "kl": 0.054317958652973175, "learning_rate": 2.2761469483317257e-07, "loss": 0.0124, "num_tokens": 244498774.0, "reward": 1.112890601158142, "reward_std": 1.0181570053100586, "rewards/code_complexity_reward/mean": 0.4901367127895355, "rewards/code_complexity_reward/std": 0.41791507601737976, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.2939453125, "rewards/code_syntax_reward/std": 0.24634800851345062, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 770, "step_time": 59.577196050435305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6953125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 483.62109375, "completions/mean_terminated_length": 418.8589782714844, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "entropy": 0.38371078623458743, "epoch": 0.879133409350057, "frac_reward_zero_std": 0.171875, "grad_norm": 0.015566810965538025, "kl": 0.056391672987956554, "learning_rate": 2.2348278929196886e-07, "loss": 0.0099, "num_tokens": 244804184.0, "reward": 1.228613257408142, "reward_std": 0.996989905834198, "rewards/code_complexity_reward/mean": 0.5372070074081421, "rewards/code_complexity_reward/std": 0.4024842083454132, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.32421875, "rewards/code_syntax_reward/std": 0.2389625608921051, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 771, "step_time": 46.30047306418419 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7265625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 482.701171875, "completions/mean_terminated_length": 404.8500061035156, "completions/min_length": 178.0, "completions/min_terminated_length": 178.0, "entropy": 0.39485615864396095, "epoch": 0.8802736602052451, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.015153131447732449, "kl": 0.05339943105354905, "learning_rate": 2.1938697617912759e-07, "loss": 0.0094, "num_tokens": 245111059.0, "reward": 1.095117211341858, "reward_std": 1.0001055002212524, "rewards/code_complexity_reward/mean": 0.4945312738418579, "rewards/code_complexity_reward/std": 0.4191471338272095, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.2958984375, "rewards/code_syntax_reward/std": 0.24599088728427887, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 772, "step_time": 58.69415467325598 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.740234375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 489.759765625, "completions/mean_terminated_length": 426.38348388671875, "completions/min_length": 203.0, "completions/min_terminated_length": 203.0, "entropy": 0.39450653828680515, "epoch": 0.8814139110604333, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.017474742606282234, "kl": 0.05427186453016475, "learning_rate": 2.1532732043061527e-07, "loss": 0.0074, "num_tokens": 245422296.0, "reward": 1.0166504383087158, "reward_std": 0.9585475921630859, "rewards/code_complexity_reward/mean": 0.47343748807907104, "rewards/code_complexity_reward/std": 0.41149991750717163, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.2890625, "rewards/code_syntax_reward/std": 0.24717088043689728, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 773, "step_time": 60.36251379735768 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.720703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 483.345703125, "completions/mean_terminated_length": 409.40557861328125, "completions/min_length": 200.0, "completions/min_terminated_length": 200.0, "entropy": 0.38761178171262145, "epoch": 0.8825541619156214, "frac_reward_zero_std": 0.125, "grad_norm": 0.01697627268731594, "kl": 0.05401688563870266, "learning_rate": 2.1130388640914794e-07, "loss": 0.0154, "num_tokens": 245729253.0, "reward": 1.118066430091858, "reward_std": 0.9917964339256287, "rewards/code_complexity_reward/mean": 0.5038086175918579, "rewards/code_complexity_reward/std": 0.4129272997379303, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.3037109375, "rewards/code_syntax_reward/std": 0.24440090358257294, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 774, "step_time": 52.91379221715033 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.642578125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 476.666015625, "completions/mean_terminated_length": 413.1420593261719, "completions/min_length": 243.0, "completions/min_terminated_length": 243.0, "entropy": 0.3835231503471732, "epoch": 0.8836944127708096, "frac_reward_zero_std": 0.171875, "grad_norm": 0.016870597377419472, "kl": 0.05574074696050957, "learning_rate": 2.0731673790317596e-07, "loss": 0.0118, "num_tokens": 246031162.0, "reward": 1.294189453125, "reward_std": 0.9883131384849548, "rewards/code_complexity_reward/mean": 0.568359375, "rewards/code_complexity_reward/std": 0.3986705541610718, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.3408203125, "rewards/code_syntax_reward/std": 0.23314768075942993, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 775, "step_time": 45.476342289708555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 474.3359375, "completions/mean_terminated_length": 402.43182373046875, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.3849490755237639, "epoch": 0.8848346636259977, "frac_reward_zero_std": 0.15625, "grad_norm": 0.01563892886042595, "kl": 0.055635390570387244, "learning_rate": 2.0336593812587096e-07, "loss": 0.0132, "num_tokens": 246332718.0, "reward": 1.3109374046325684, "reward_std": 0.9894391298294067, "rewards/code_complexity_reward/mean": 0.5785156488418579, "rewards/code_complexity_reward/std": 0.39864903688430786, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.34375, "rewards/code_syntax_reward/std": 0.23198285698890686, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 776, "step_time": 45.702550483867526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.697265625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 480.830078125, "completions/mean_terminated_length": 409.0386962890625, "completions/min_length": 208.0, "completions/min_terminated_length": 208.0, "entropy": 0.39043072424829006, "epoch": 0.8859749144811858, "frac_reward_zero_std": 0.09375, "grad_norm": 0.018218116834759712, "kl": 0.055159661802463233, "learning_rate": 1.9945154971412168e-07, "loss": 0.0128, "num_tokens": 246637923.0, "reward": 1.2169921398162842, "reward_std": 0.9758403897285461, "rewards/code_complexity_reward/mean": 0.5460937023162842, "rewards/code_complexity_reward/std": 0.4015069007873535, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.3291015625, "rewards/code_syntax_reward/std": 0.23738788068294525, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 777, "step_time": 54.99546112958342 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.658203125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 474.591796875, "completions/mean_terminated_length": 402.5542907714844, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.3893895335495472, "epoch": 0.887115165336374, "frac_reward_zero_std": 0.1875, "grad_norm": 0.01751638762652874, "kl": 0.05590817774645984, "learning_rate": 1.9557363472754582e-07, "loss": 0.008, "num_tokens": 246940086.0, "reward": 1.1607909202575684, "reward_std": 0.9972427487373352, "rewards/code_complexity_reward/mean": 0.528027355670929, "rewards/code_complexity_reward/std": 0.41802656650543213, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.3115234375, "rewards/code_syntax_reward/std": 0.24254849553108215, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 778, "step_time": 53.70588385593146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.69140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 479.66796875, "completions/mean_terminated_length": 407.22784423828125, "completions/min_length": 223.0, "completions/min_terminated_length": 223.0, "entropy": 0.3815963198430836, "epoch": 0.8882554161915621, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.01520970556885004, "kl": 0.05766374716768041, "learning_rate": 1.9173225464749867e-07, "loss": 0.0068, "num_tokens": 247244096.0, "reward": 1.192773461341858, "reward_std": 1.0014454126358032, "rewards/code_complexity_reward/mean": 0.5277343988418579, "rewards/code_complexity_reward/std": 0.4111012816429138, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.3154296875, "rewards/code_syntax_reward/std": 0.24152201414108276, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 779, "step_time": 47.07709497958422 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.693359375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 481.890625, "completions/mean_terminated_length": 413.8089294433594, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.3780671274289489, "epoch": 0.8893956670467503, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.014299621805548668, "kl": 0.057583401387091726, "learning_rate": 1.879274703761072e-07, "loss": 0.0051, "num_tokens": 247550828.0, "reward": 1.138671875, "reward_std": 0.9786661267280579, "rewards/code_complexity_reward/mean": 0.525390625, "rewards/code_complexity_reward/std": 0.41461771726608276, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.3125, "rewards/code_syntax_reward/std": 0.2422981858253479, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 780, "step_time": 64.43424165900797 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.720703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 483.24609375, "completions/mean_terminated_length": 409.0489501953125, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.3915279661305249, "epoch": 0.8905359179019384, "frac_reward_zero_std": 0.203125, "grad_norm": 0.015955297276377678, "kl": 0.05789039476076141, "learning_rate": 1.8415934223529665e-07, "loss": 0.0124, "num_tokens": 247857094.0, "reward": 1.09130859375, "reward_std": 0.9903880953788757, "rewards/code_complexity_reward/mean": 0.49853515625, "rewards/code_complexity_reward/std": 0.41672801971435547, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.2978515625, "rewards/code_syntax_reward/std": 0.24561770260334015, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 781, "step_time": 53.50928711052984 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.70703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.3671875, "completions/mean_terminated_length": 421.0933532714844, "completions/min_length": 254.0, "completions/min_terminated_length": 254.0, "entropy": 0.38866491988301277, "epoch": 0.8916761687571265, "frac_reward_zero_std": 0.125, "grad_norm": 0.01713121123611927, "kl": 0.054969930672086775, "learning_rate": 1.8042792996583902e-07, "loss": 0.0112, "num_tokens": 248164918.0, "reward": 1.0745117664337158, "reward_std": 0.9475169777870178, "rewards/code_complexity_reward/mean": 0.504199206829071, "rewards/code_complexity_reward/std": 0.4066556692123413, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.30859375, "rewards/code_syntax_reward/std": 0.2432742565870285, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 782, "step_time": 52.74827292282134 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 476.00390625, "completions/mean_terminated_length": 407.2840881347656, "completions/min_length": 174.0, "completions/min_terminated_length": 174.0, "entropy": 0.3850144539028406, "epoch": 0.8928164196123147, "frac_reward_zero_std": 0.15625, "grad_norm": 0.016519881784915924, "kl": 0.05693793838145211, "learning_rate": 1.76733292726404e-07, "loss": 0.0132, "num_tokens": 248467892.0, "reward": 1.225341796875, "reward_std": 1.01715886592865, "rewards/code_complexity_reward/mean": 0.5320311784744263, "rewards/code_complexity_reward/std": 0.41175174713134766, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.3173828125, "rewards/code_syntax_reward/std": 0.24098336696624756, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 783, "step_time": 45.79280070774257 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.642578125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 475.361328125, "completions/mean_terminated_length": 409.4917907714844, "completions/min_length": 172.0, "completions/min_terminated_length": 172.0, "entropy": 0.3832142222672701, "epoch": 0.8939566704675028, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.018045924603939056, "kl": 0.054856448143254966, "learning_rate": 1.7307548909262118e-07, "loss": 0.0203, "num_tokens": 248771097.0, "reward": 1.1360352039337158, "reward_std": 0.9857324957847595, "rewards/code_complexity_reward/mean": 0.517871081829071, "rewards/code_complexity_reward/std": 0.4135285019874573, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.3095703125, "rewards/code_syntax_reward/std": 0.24303650856018066, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 784, "step_time": 72.18394994642586 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.701171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 482.529296875, "completions/mean_terminated_length": 413.37908935546875, "completions/min_length": 147.0, "completions/min_terminated_length": 147.0, "entropy": 0.38349762512370944, "epoch": 0.895096921322691, "frac_reward_zero_std": 0.171875, "grad_norm": 0.0162428617477417, "kl": 0.055054253549315035, "learning_rate": 1.694545770561537e-07, "loss": 0.0146, "num_tokens": 249077960.0, "reward": 1.146386742591858, "reward_std": 0.9869211912155151, "rewards/code_complexity_reward/mean": 0.5165039300918579, "rewards/code_complexity_reward/std": 0.4090537130832672, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.3115234375, "rewards/code_syntax_reward/std": 0.24254849553108215, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 785, "step_time": 51.283332596533 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.638671875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 472.775390625, "completions/mean_terminated_length": 403.4432678222656, "completions/min_length": 199.0, "completions/min_terminated_length": 199.0, "entropy": 0.3873316920362413, "epoch": 0.8962371721778791, "frac_reward_zero_std": 0.15625, "grad_norm": 0.016758568584918976, "kl": 0.0584030884783715, "learning_rate": 1.658706140237737e-07, "loss": 0.0172, "num_tokens": 249378593.0, "reward": 1.2330079078674316, "reward_std": 1.0299072265625, "rewards/code_complexity_reward/mean": 0.5308593511581421, "rewards/code_complexity_reward/std": 0.4132881462574005, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.3154296875, "rewards/code_syntax_reward/std": 0.24152201414108276, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 786, "step_time": 54.41173950303346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 481.298828125, "completions/mean_terminated_length": 422.6875, "completions/min_length": 257.0, "completions/min_terminated_length": 257.0, "entropy": 0.3840023218654096, "epoch": 0.8973774230330672, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.01665656268596649, "kl": 0.05366018234053627, "learning_rate": 1.623236568164574e-07, "loss": 0.0136, "num_tokens": 249683166.0, "reward": 1.2285645008087158, "reward_std": 0.9913499355316162, "rewards/code_complexity_reward/mean": 0.542773425579071, "rewards/code_complexity_reward/std": 0.4038543999195099, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.326171875, "rewards/code_syntax_reward/std": 0.23834596574306488, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 787, "step_time": 53.74242296721786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.716796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.82421875, "completions/mean_terminated_length": 419.5724182128906, "completions/min_length": 250.0, "completions/min_terminated_length": 250.0, "entropy": 0.38741324795410037, "epoch": 0.8985176738882554, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.016108358278870583, "kl": 0.05540177249349654, "learning_rate": 1.5881376166848151e-07, "loss": 0.0083, "num_tokens": 249990720.0, "reward": 1.1150391101837158, "reward_std": 1.000317096710205, "rewards/code_complexity_reward/mean": 0.501757800579071, "rewards/code_complexity_reward/std": 0.41639214754104614, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.30078125, "rewards/code_syntax_reward/std": 0.2450276017189026, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 788, "step_time": 45.367290989495814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.849609375, "completions/mean_terminated_length": 417.3161926269531, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "entropy": 0.37671558232977986, "epoch": 0.8996579247434435, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.017166590318083763, "kl": 0.05546099634375423, "learning_rate": 1.5534098422653243e-07, "loss": 0.0118, "num_tokens": 250300327.0, "reward": 1.1347168684005737, "reward_std": 1.0017094612121582, "rewards/code_complexity_reward/mean": 0.49580079317092896, "rewards/code_complexity_reward/std": 0.4065353274345398, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.3046875, "rewards/code_syntax_reward/std": 0.24418380856513977, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 789, "step_time": 58.630896128714085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7109375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.224609375, "completions/mean_terminated_length": 422.8310852050781, "completions/min_length": 214.0, "completions/min_terminated_length": 214.0, "entropy": 0.39652628684416413, "epoch": 0.9007981755986317, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.014694043435156345, "kl": 0.0549516924074851, "learning_rate": 1.5190537954882373e-07, "loss": 0.0072, "num_tokens": 250608838.0, "reward": 1.1313965320587158, "reward_std": 0.9937042593955994, "rewards/code_complexity_reward/mean": 0.5012695789337158, "rewards/code_complexity_reward/std": 0.40861961245536804, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.3056640625, "rewards/code_syntax_reward/std": 0.2439626157283783, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 790, "step_time": 58.16139207594097 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.720703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.705078125, "completions/mean_terminated_length": 414.2727355957031, "completions/min_length": 225.0, "completions/min_terminated_length": 225.0, "entropy": 0.38921136409044266, "epoch": 0.9019384264538198, "frac_reward_zero_std": 0.203125, "grad_norm": 0.016780609264969826, "kl": 0.057610903517343104, "learning_rate": 1.485070021042237e-07, "loss": 0.0071, "num_tokens": 250916691.0, "reward": 1.100976586341858, "reward_std": 0.9788212776184082, "rewards/code_complexity_reward/mean": 0.504589855670929, "rewards/code_complexity_reward/std": 0.41349947452545166, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.3046875, "rewards/code_syntax_reward/std": 0.24418380856513977, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 791, "step_time": 52.76263571064919 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.724609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.50390625, "completions/mean_terminated_length": 412.156005859375, "completions/min_length": 171.0, "completions/min_terminated_length": 171.0, "entropy": 0.39226270839571953, "epoch": 0.9030786773090079, "frac_reward_zero_std": 0.21875, "grad_norm": 0.016259681433439255, "kl": 0.055678183503914624, "learning_rate": 1.4514590577139136e-07, "loss": 0.0075, "num_tokens": 251225769.0, "reward": 1.08154296875, "reward_std": 0.9971565008163452, "rewards/code_complexity_reward/mean": 0.48974609375, "rewards/code_complexity_reward/std": 0.4181514382362366, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.29296875, "rewards/code_syntax_reward/std": 0.24652054905891418, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 792, "step_time": 46.06103719957173 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.697265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 478.033203125, "completions/mean_terminated_length": 399.79998779296875, "completions/min_length": 203.0, "completions/min_terminated_length": 203.0, "entropy": 0.3927192986011505, "epoch": 0.9042189281641961, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.014745771884918213, "kl": 0.05441470455843955, "learning_rate": 1.4182214383792192e-07, "loss": 0.0081, "num_tokens": 251530794.0, "reward": 1.1615722179412842, "reward_std": 0.9819875955581665, "rewards/code_complexity_reward/mean": 0.525585949420929, "rewards/code_complexity_reward/std": 0.40924930572509766, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.3154296875, "rewards/code_syntax_reward/std": 0.24152201414108276, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 793, "step_time": 51.44373522885144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.73828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 489.30078125, "completions/mean_terminated_length": 425.2686462402344, "completions/min_length": 222.0, "completions/min_terminated_length": 222.0, "entropy": 0.39589724503457546, "epoch": 0.9053591790193842, "frac_reward_zero_std": 0.21875, "grad_norm": 0.01447091344743967, "kl": 0.053496287669986486, "learning_rate": 1.3853576899950343e-07, "loss": 0.0061, "num_tokens": 251840304.0, "reward": 0.980175793170929, "reward_std": 0.9699560403823853, "rewards/code_complexity_reward/mean": 0.45185548067092896, "rewards/code_complexity_reward/std": 0.41550368070602417, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.2763671875, "rewards/code_syntax_reward/std": 0.2488487958908081, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 794, "step_time": 73.49118506815284 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.03515625, "completions/mean_terminated_length": 420.21795654296875, "completions/min_length": 240.0, "completions/min_terminated_length": 240.0, "entropy": 0.37706570606678724, "epoch": 0.9064994298745724, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.01599183678627014, "kl": 0.05598545726388693, "learning_rate": 1.3528683335907928e-07, "loss": 0.0107, "num_tokens": 252148918.0, "reward": 1.230566382408142, "reward_std": 0.9672464728355408, "rewards/code_complexity_reward/mean": 0.5596679449081421, "rewards/code_complexity_reward/std": 0.40014585852622986, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.3349609375, "rewards/code_syntax_reward/std": 0.23535043001174927, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 795, "step_time": 53.45740006305277 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.619140625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 473.5234375, "completions/mean_terminated_length": 410.974365234375, "completions/min_length": 202.0, "completions/min_terminated_length": 202.0, "entropy": 0.3901880090124905, "epoch": 0.9076396807297605, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.015005821362137794, "kl": 0.05634443857707083, "learning_rate": 1.3207538842602396e-07, "loss": 0.0107, "num_tokens": 252449806.0, "reward": 1.229101538658142, "reward_std": 0.9612974524497986, "rewards/code_complexity_reward/mean": 0.5699218511581421, "rewards/code_complexity_reward/std": 0.4018835127353668, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.3388671875, "rewards/code_syntax_reward/std": 0.23390056192874908, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 796, "step_time": 53.87137156911194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 471.1796875, "completions/mean_terminated_length": 393.25, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "entropy": 0.3918442805297673, "epoch": 0.9087799315849487, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.015469147823750973, "kl": 0.05512112524593249, "learning_rate": 1.289014851153253e-07, "loss": 0.0063, "num_tokens": 252749270.0, "reward": 1.1545898914337158, "reward_std": 1.0024888515472412, "rewards/code_complexity_reward/mean": 0.515917956829071, "rewards/code_complexity_reward/std": 0.4146362543106079, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.30859375, "rewards/code_syntax_reward/std": 0.2432742565870285, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 797, "step_time": 50.706998627632856 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.666015625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 477.44921875, "completions/mean_terminated_length": 408.5497131347656, "completions/min_length": 226.0, "completions/min_terminated_length": 226.0, "entropy": 0.3828963120467961, "epoch": 0.9099201824401368, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.018109383061528206, "kl": 0.05580217583337799, "learning_rate": 1.2576517374677745e-07, "loss": 0.0119, "num_tokens": 253051356.0, "reward": 1.2457032203674316, "reward_std": 0.9739770293235779, "rewards/code_complexity_reward/mean": 0.5621093511581421, "rewards/code_complexity_reward/std": 0.3998163044452667, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.3359375, "rewards/code_syntax_reward/std": 0.23499488830566406, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 798, "step_time": 45.886188911274076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.759765625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 491.623046875, "completions/mean_terminated_length": 427.1788330078125, "completions/min_length": 270.0, "completions/min_terminated_length": 270.0, "entropy": 0.38582940213382244, "epoch": 0.9110604332953249, "frac_reward_zero_std": 0.203125, "grad_norm": 0.015824513509869576, "kl": 0.053528685006313026, "learning_rate": 1.2266650404418378e-07, "loss": 0.0032, "num_tokens": 253362763.0, "reward": 1.04150390625, "reward_std": 0.9821205139160156, "rewards/code_complexity_reward/mean": 0.47412109375, "rewards/code_complexity_reward/std": 0.4137547016143799, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.2880859375, "rewards/code_syntax_reward/std": 0.24732354283332825, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 799, "step_time": 45.05831072013825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.73046875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.884765625, "completions/mean_terminated_length": 418.81884765625, "completions/min_length": 187.0, "completions/min_terminated_length": 187.0, "entropy": 0.3878536452539265, "epoch": 0.9122006841505131, "frac_reward_zero_std": 0.1875, "grad_norm": 0.015595006756484509, "kl": 0.055863751797005534, "learning_rate": 1.1960552513456764e-07, "loss": 0.01, "num_tokens": 253671324.0, "reward": 1.1432616710662842, "reward_std": 1.0163383483886719, "rewards/code_complexity_reward/mean": 0.49775391817092896, "rewards/code_complexity_reward/std": 0.41259506344795227, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.3017578125, "rewards/code_syntax_reward/std": 0.24482278525829315, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 800, "step_time": 57.66674091760069 }, { "epoch": 0.9122006841505131, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.7025, "eval_completions/max_length": 512.0, "eval_completions/max_terminated_length": 355.54, "eval_completions/mean_length": 479.59, "eval_completions/mean_terminated_length": 319.8173358154297, "eval_completions/min_length": 396.62, "eval_completions/min_terminated_length": 283.98, "eval_entropy": 0.40322402358055115, "eval_frac_reward_zero_std": 0.23, "eval_kl": 0.0681039346754551, "eval_loss": 0.015396857634186745, "eval_num_tokens": 253671324.0, "eval_reward": 1.0793750113248826, "eval_reward_std": 0.8657220470905304, "eval_rewards/code_complexity_reward/mean": 0.49312500104308127, "eval_rewards/code_complexity_reward/std": 0.3728107272088528, "eval_rewards/code_execution_reward/mean": 0.2925, "eval_rewards/code_execution_reward/std": 0.3384459352493286, "eval_rewards/code_syntax_reward/mean": 0.29375, "eval_rewards/code_syntax_reward/std": 0.22101665019989014, "eval_rewards/reasoning_present_reward_func/mean": 0.0, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.0, "eval_rewards/xmlcount_reward_func/std": 0.0, "eval_runtime": 1245.461, "eval_samples_per_second": 0.08, "eval_steps_per_second": 0.01, "step": 800 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.689453125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 483.654296875, "completions/mean_terminated_length": 420.7232666015625, "completions/min_length": 240.0, "completions/min_terminated_length": 240.0, "entropy": 0.38884901255369186, "epoch": 0.9133409350057012, "frac_reward_zero_std": 0.15625, "grad_norm": 0.02023371122777462, "kl": 0.0548762675607577, "learning_rate": 1.1658228554739359e-07, "loss": 0.0086, "num_tokens": 253979147.0, "reward": 1.26806640625, "reward_std": 0.9913524389266968, "rewards/code_complexity_reward/mean": 0.54638671875, "rewards/code_complexity_reward/std": 0.39577221870422363, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.3330078125, "rewards/code_syntax_reward/std": 0.23604771494865417, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 801, "step_time": 60.59586180746555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 481.1875, "completions/mean_terminated_length": 410.8717956542969, "completions/min_length": 219.0, "completions/min_terminated_length": 219.0, "entropy": 0.39812028408050537, "epoch": 0.9144811858608894, "frac_reward_zero_std": 0.1875, "grad_norm": 0.016799790784716606, "kl": 0.05852890119422227, "learning_rate": 1.1359683321379878e-07, "loss": 0.0133, "num_tokens": 254284483.0, "reward": 1.0739257335662842, "reward_std": 0.9709258675575256, "rewards/code_complexity_reward/mean": 0.49628907442092896, "rewards/code_complexity_reward/std": 0.41575607657432556, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.2978515625, "rewards/code_syntax_reward/std": 0.24561770260334015, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 802, "step_time": 60.644819252192974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.69140625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 481.25390625, "completions/mean_terminated_length": 412.3670959472656, "completions/min_length": 217.0, "completions/min_terminated_length": 217.0, "entropy": 0.37217913568019867, "epoch": 0.9156214367160775, "frac_reward_zero_std": 0.140625, "grad_norm": 0.017930684611201286, "kl": 0.05656182218808681, "learning_rate": 1.106492154658323e-07, "loss": 0.0133, "num_tokens": 254588089.0, "reward": 1.222900390625, "reward_std": 0.997454822063446, "rewards/code_complexity_reward/mean": 0.537402331829071, "rewards/code_complexity_reward/std": 0.4063373804092407, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.3232421875, "rewards/code_syntax_reward/std": 0.23926427960395813, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 803, "step_time": 59.26238466054201 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.853515625, "completions/mean_terminated_length": 417.33087158203125, "completions/min_length": 256.0, "completions/min_terminated_length": 256.0, "entropy": 0.3989083543419838, "epoch": 0.9167616875712656, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.016675326973199844, "kl": 0.057207581237889826, "learning_rate": 1.0773947903570503e-07, "loss": 0.0095, "num_tokens": 254896794.0, "reward": 1.0748047828674316, "reward_std": 1.0010415315628052, "rewards/code_complexity_reward/mean": 0.4800781011581421, "rewards/code_complexity_reward/std": 0.4161062240600586, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.2900390625, "rewards/code_syntax_reward/std": 0.24701425433158875, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 804, "step_time": 44.64695203397423 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65234375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 476.57421875, "completions/mean_terminated_length": 410.10113525390625, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.38475854881107807, "epoch": 0.9179019384264538, "frac_reward_zero_std": 0.21875, "grad_norm": 0.01505669578909874, "kl": 0.05600192863494158, "learning_rate": 1.0486767005504911e-07, "loss": 0.0089, "num_tokens": 255199820.0, "reward": 1.2550780773162842, "reward_std": 1.0051937103271484, "rewards/code_complexity_reward/mean": 0.552929699420929, "rewards/code_complexity_reward/std": 0.40652844309806824, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.3291015625, "rewards/code_syntax_reward/std": 0.23738788068294525, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 805, "step_time": 53.82527507469058 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6640625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 478.361328125, "completions/mean_terminated_length": 411.86627197265625, "completions/min_length": 234.0, "completions/min_terminated_length": 234.0, "entropy": 0.377513169310987, "epoch": 0.9190421892816419, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.014871912077069283, "kl": 0.05570724495919421, "learning_rate": 1.0203383405418515e-07, "loss": 0.0078, "num_tokens": 255504881.0, "reward": 1.23388671875, "reward_std": 0.9846171140670776, "rewards/code_complexity_reward/mean": 0.55224609375, "rewards/code_complexity_reward/std": 0.40328866243362427, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.330078125, "rewards/code_syntax_reward/std": 0.2370595932006836, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 806, "step_time": 61.327820549719036 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.697265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 478.625, "completions/mean_terminated_length": 401.75482177734375, "completions/min_length": 180.0, "completions/min_terminated_length": 180.0, "entropy": 0.39424010273069143, "epoch": 0.9201824401368301, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.016774235293269157, "kl": 0.05738946341443807, "learning_rate": 9.923801596140258e-08, "loss": 0.0096, "num_tokens": 255809005.0, "reward": 1.116455078125, "reward_std": 0.9925623536109924, "rewards/code_complexity_reward/mean": 0.5097655653953552, "rewards/code_complexity_reward/std": 0.4185197949409485, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.3017578125, "rewards/code_syntax_reward/std": 0.24482278525829315, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 807, "step_time": 47.41491161752492 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.310546875, "completions/mean_terminated_length": 425.55487060546875, "completions/min_length": 232.0, "completions/min_terminated_length": 232.0, "entropy": 0.38788266759365797, "epoch": 0.9213226909920182, "frac_reward_zero_std": 0.171875, "grad_norm": 0.016002891585230827, "kl": 0.05575513036455959, "learning_rate": 9.64802601022452e-08, "loss": 0.0096, "num_tokens": 256116560.0, "reward": 1.1935547590255737, "reward_std": 0.995841383934021, "rewards/code_complexity_reward/mean": 0.5363280773162842, "rewards/code_complexity_reward/std": 0.41122767329216003, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.3193359375, "rewards/code_syntax_reward/std": 0.24042759835720062, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 808, "step_time": 70.3833782421425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 473.3984375, "completions/mean_terminated_length": 399.7045593261719, "completions/min_length": 250.0, "completions/min_terminated_length": 250.0, "entropy": 0.37925533251836896, "epoch": 0.9224629418472063, "frac_reward_zero_std": 0.1875, "grad_norm": 0.015416930429637432, "kl": 0.055305285088252276, "learning_rate": 9.376061019881006e-08, "loss": 0.0079, "num_tokens": 256418256.0, "reward": 1.2090821266174316, "reward_std": 1.011170744895935, "rewards/code_complexity_reward/mean": 0.5352538824081421, "rewards/code_complexity_reward/std": 0.4122065305709839, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.318359375, "rewards/code_syntax_reward/std": 0.2407076209783554, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 809, "step_time": 54.36339973285794 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.720703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.986328125, "completions/mean_terminated_length": 422.4405517578125, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.38534402661025524, "epoch": 0.9236031927023945, "frac_reward_zero_std": 0.15625, "grad_norm": 0.016769107431173325, "kl": 0.05891400226391852, "learning_rate": 9.107910936905301e-08, "loss": 0.011, "num_tokens": 256727121.0, "reward": 1.0880370140075684, "reward_std": 0.9357959628105164, "rewards/code_complexity_reward/mean": 0.5122069716453552, "rewards/code_complexity_reward/std": 0.4037645757198334, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.3134765625, "rewards/code_syntax_reward/std": 0.24204368889331818, "rewards/reasoning_present_reward_func/mean": 0.0003906250058207661, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.002197265625, "rewards/xmlcount_reward_func/std": 0.03168932721018791, "step": 810, "step_time": 55.010378072969615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.74609375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 490.48046875, "completions/mean_terminated_length": 427.24615478515625, "completions/min_length": 180.0, "completions/min_terminated_length": 180.0, "entropy": 0.3927961098961532, "epoch": 0.9247434435575826, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.016288630664348602, "kl": 0.056350490369368345, "learning_rate": 8.843580012610625e-08, "loss": 0.0107, "num_tokens": 257037615.0, "reward": 1.03759765625, "reward_std": 0.9440863132476807, "rewards/code_complexity_reward/mean": 0.49169921875, "rewards/code_complexity_reward/std": 0.4116508364677429, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.2978515625, "rewards/code_syntax_reward/std": 0.24561770260334015, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 811, "step_time": 62.45034656487405 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 482.92578125, "completions/mean_terminated_length": 416.5769348144531, "completions/min_length": 219.0, "completions/min_terminated_length": 219.0, "entropy": 0.3874737434089184, "epoch": 0.9258836944127709, "frac_reward_zero_std": 0.1875, "grad_norm": 0.01406517531722784, "kl": 0.056920311704743654, "learning_rate": 8.583072437760381e-08, "loss": 0.0126, "num_tokens": 257346869.0, "reward": 1.0882811546325684, "reward_std": 0.9839574098587036, "rewards/code_complexity_reward/mean": 0.500195324420929, "rewards/code_complexity_reward/std": 0.4180491864681244, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.2978515625, "rewards/code_syntax_reward/std": 0.24561770260334015, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 812, "step_time": 59.96713381819427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6640625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 480.369140625, "completions/mean_terminated_length": 417.843017578125, "completions/min_length": 250.0, "completions/min_terminated_length": 250.0, "entropy": 0.3917591799981892, "epoch": 0.927023945267959, "frac_reward_zero_std": 0.140625, "grad_norm": 0.016802389174699783, "kl": 0.05523341574007645, "learning_rate": 8.3263923425016e-08, "loss": 0.0125, "num_tokens": 257654310.0, "reward": 1.168212890625, "reward_std": 0.9978678822517395, "rewards/code_complexity_reward/mean": 0.5205078125, "rewards/code_complexity_reward/std": 0.40923693776130676, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.3134765625, "rewards/code_syntax_reward/std": 0.24204368889331818, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 813, "step_time": 53.02332460228354 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.673828125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 479.58203125, "completions/mean_terminated_length": 412.61077880859375, "completions/min_length": 205.0, "completions/min_terminated_length": 205.0, "entropy": 0.3822345989756286, "epoch": 0.928164196123147, "frac_reward_zero_std": 0.15625, "grad_norm": 0.01743876002728939, "kl": 0.05674429633654654, "learning_rate": 8.07354379629971e-08, "loss": 0.0132, "num_tokens": 257958224.0, "reward": 1.3023924827575684, "reward_std": 0.9570742249488831, "rewards/code_complexity_reward/mean": 0.5868163704872131, "rewards/code_complexity_reward/std": 0.3902142345905304, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.3515625, "rewards/code_syntax_reward/std": 0.22866390645503998, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 814, "step_time": 46.733795115724206 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.74609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 488.177734375, "completions/mean_terminated_length": 418.1769104003906, "completions/min_length": 239.0, "completions/min_terminated_length": 239.0, "entropy": 0.39282417856156826, "epoch": 0.9293044469783353, "frac_reward_zero_std": 0.140625, "grad_norm": 0.016886692494153976, "kl": 0.0535181172308512, "learning_rate": 7.82453080787382e-08, "loss": 0.0109, "num_tokens": 258265079.0, "reward": 1.0519530773162842, "reward_std": 0.9728989601135254, "rewards/code_complexity_reward/mean": 0.48066407442092896, "rewards/code_complexity_reward/std": 0.4098077714443207, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.2939453125, "rewards/code_syntax_reward/std": 0.24634800851345062, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 815, "step_time": 58.77490060310811 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65234375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 476.0546875, "completions/mean_terminated_length": 408.60675048828125, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.38688742415979505, "epoch": 0.9304446978335233, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.01766805350780487, "kl": 0.05460904573556036, "learning_rate": 7.579357325133208e-08, "loss": 0.0138, "num_tokens": 258568987.0, "reward": 1.2255859375, "reward_std": 0.9845514297485352, "rewards/code_complexity_reward/mean": 0.54541015625, "rewards/code_complexity_reward/std": 0.4003465175628662, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.330078125, "rewards/code_syntax_reward/std": 0.2370595932006836, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 816, "step_time": 46.58784963004291 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.67578125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 481.501953125, "completions/mean_terminated_length": 417.9337158203125, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.38310656463727355, "epoch": 0.9315849486887116, "frac_reward_zero_std": 0.1875, "grad_norm": 0.01491616852581501, "kl": 0.05789853830356151, "learning_rate": 7.338027235114759e-08, "loss": 0.0067, "num_tokens": 258873460.0, "reward": 1.2143065929412842, "reward_std": 1.0044283866882324, "rewards/code_complexity_reward/mean": 0.536328136920929, "rewards/code_complexity_reward/std": 0.4093078076839447, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.3203125, "rewards/code_syntax_reward/std": 0.24014326930046082, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 817, "step_time": 45.62096853274852 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.740234375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.58203125, "completions/mean_terminated_length": 406.4511413574219, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.3851535744033754, "epoch": 0.9327251995438997, "frac_reward_zero_std": 0.1875, "grad_norm": 0.015648262575268745, "kl": 0.05468508048215881, "learning_rate": 7.100544363921324e-08, "loss": 0.0124, "num_tokens": 259181818.0, "reward": 1.0732421875, "reward_std": 0.976047158241272, "rewards/code_complexity_reward/mean": 0.48046875, "rewards/code_complexity_reward/std": 0.40596020221710205, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.2978515625, "rewards/code_syntax_reward/std": 0.24561770260334015, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 818, "step_time": 54.972081150859594 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.669921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 476.478515625, "completions/mean_terminated_length": 404.3846130371094, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "entropy": 0.37771226558834314, "epoch": 0.9338654503990877, "frac_reward_zero_std": 0.140625, "grad_norm": 0.01564924605190754, "kl": 0.05839672475121915, "learning_rate": 6.866912476661075e-08, "loss": 0.009, "num_tokens": 259487479.0, "reward": 1.27880859375, "reward_std": 0.9769481420516968, "rewards/code_complexity_reward/mean": 0.57177734375, "rewards/code_complexity_reward/std": 0.40003591775894165, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.33984375, "rewards/code_syntax_reward/std": 0.23352646827697754, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 819, "step_time": 45.13061999157071 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.685546875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 477.66015625, "completions/mean_terminated_length": 402.7950439453125, "completions/min_length": 205.0, "completions/min_terminated_length": 205.0, "entropy": 0.3797367732040584, "epoch": 0.935005701254276, "frac_reward_zero_std": 0.1875, "grad_norm": 0.014731590636074543, "kl": 0.05650017625885084, "learning_rate": 6.637135277387713e-08, "loss": 0.0139, "num_tokens": 259789833.0, "reward": 1.2535157203674316, "reward_std": 0.9697616696357727, "rewards/code_complexity_reward/mean": 0.5638672113418579, "rewards/code_complexity_reward/std": 0.3967370390892029, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.3388671875, "rewards/code_syntax_reward/std": 0.23390056192874908, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 820, "step_time": 54.113782653585076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.69140625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 480.13671875, "completions/mean_terminated_length": 408.7468566894531, "completions/min_length": 242.0, "completions/min_terminated_length": 242.0, "entropy": 0.3866188391111791, "epoch": 0.936145952109464, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.014788891188800335, "kl": 0.054274900001473725, "learning_rate": 6.411216409041965e-08, "loss": 0.0163, "num_tokens": 260096803.0, "reward": 1.17529296875, "reward_std": 1.0112392902374268, "rewards/code_complexity_reward/mean": 0.5203125476837158, "rewards/code_complexity_reward/std": 0.4153459072113037, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.3095703125, "rewards/code_syntax_reward/std": 0.24303650856018066, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.00146484375, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 821, "step_time": 45.7633601045236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.70703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.12109375, "completions/mean_terminated_length": 423.66668701171875, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.39440637920051813, "epoch": 0.9372862029646523, "frac_reward_zero_std": 0.140625, "grad_norm": 0.01638827659189701, "kl": 0.056511485658120364, "learning_rate": 6.189159453393573e-08, "loss": 0.0082, "num_tokens": 260405769.0, "reward": 1.1461913585662842, "reward_std": 0.994161069393158, "rewards/code_complexity_reward/mean": 0.5087890625, "rewards/code_complexity_reward/std": 0.4095973074436188, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.30859375, "rewards/code_syntax_reward/std": 0.2432742565870285, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 822, "step_time": 53.67418342549354 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.708984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.884765625, "completions/mean_terminated_length": 418.82550048828125, "completions/min_length": 208.0, "completions/min_terminated_length": 208.0, "entropy": 0.38991081807762384, "epoch": 0.9384264538198404, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.016031110659241676, "kl": 0.05546095815952867, "learning_rate": 5.970967930984728e-08, "loss": 0.0104, "num_tokens": 260713642.0, "reward": 1.1021971702575684, "reward_std": 0.9864697456359863, "rewards/code_complexity_reward/mean": 0.49580076336860657, "rewards/code_complexity_reward/std": 0.41221198439598083, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.30078125, "rewards/code_syntax_reward/std": 0.2450276017189026, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 823, "step_time": 51.6293321037665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.705078125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 482.84375, "completions/mean_terminated_length": 413.1390686035156, "completions/min_length": 201.0, "completions/min_terminated_length": 201.0, "entropy": 0.39526344323530793, "epoch": 0.9395667046750285, "frac_reward_zero_std": 0.203125, "grad_norm": 0.015096420422196388, "kl": 0.054061130969785154, "learning_rate": 5.756645301074088e-08, "loss": 0.0123, "num_tokens": 261021726.0, "reward": 1.10107421875, "reward_std": 0.9849622249603271, "rewards/code_complexity_reward/mean": 0.49462890625, "rewards/code_complexity_reward/std": 0.40835869312286377, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.3017578125, "rewards/code_syntax_reward/std": 0.24482278525829315, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 824, "step_time": 45.540787477977574 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.71875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.158203125, "completions/mean_terminated_length": 420.1180725097656, "completions/min_length": 203.0, "completions/min_terminated_length": 203.0, "entropy": 0.38265862176194787, "epoch": 0.9407069555302167, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.01605106331408024, "kl": 0.05945501901442185, "learning_rate": 5.546194961581958e-08, "loss": 0.0147, "num_tokens": 261330719.0, "reward": 1.093359351158142, "reward_std": 0.9708192944526672, "rewards/code_complexity_reward/mean": 0.5044921636581421, "rewards/code_complexity_reward/std": 0.4117041826248169, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.3037109375, "rewards/code_syntax_reward/std": 0.24440090358257294, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 825, "step_time": 45.40829338785261 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.724609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 486.017578125, "completions/mean_terminated_length": 417.6524658203125, "completions/min_length": 252.0, "completions/min_terminated_length": 252.0, "entropy": 0.3844743245281279, "epoch": 0.9418472063854048, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.015745436772704124, "kl": 0.05248664505779743, "learning_rate": 5.3396202490365586e-08, "loss": 0.012, "num_tokens": 261640412.0, "reward": 1.094628930091858, "reward_std": 0.9943211674690247, "rewards/code_complexity_reward/mean": 0.49794918298721313, "rewards/code_complexity_reward/std": 0.4174410402774811, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.2978515625, "rewards/code_syntax_reward/std": 0.24561770260334015, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 826, "step_time": 55.58810133021325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.68359375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 476.205078125, "completions/mean_terminated_length": 398.870361328125, "completions/min_length": 196.0, "completions/min_terminated_length": 196.0, "entropy": 0.38210091181099415, "epoch": 0.942987457240593, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.01680058427155018, "kl": 0.05867984960786998, "learning_rate": 5.136924438520902e-08, "loss": 0.0153, "num_tokens": 261942401.0, "reward": 1.089990258216858, "reward_std": 1.005836009979248, "rewards/code_complexity_reward/mean": 0.4852539002895355, "rewards/code_complexity_reward/std": 0.4173375070095062, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.2919921875, "rewards/code_syntax_reward/std": 0.246689110994339, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 827, "step_time": 54.89368579350412 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.677734375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 481.220703125, "completions/mean_terminated_length": 416.49090576171875, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "entropy": 0.3798000440001488, "epoch": 0.9441277080957811, "frac_reward_zero_std": 0.171875, "grad_norm": 0.015016932971775532, "kl": 0.05620074877515435, "learning_rate": 4.9381107436211607e-08, "loss": 0.0079, "num_tokens": 262247386.0, "reward": 1.17822265625, "reward_std": 1.0111652612686157, "rewards/code_complexity_reward/mean": 0.52294921875, "rewards/code_complexity_reward/std": 0.41408151388168335, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.3115234375, "rewards/code_syntax_reward/std": 0.24254849553108215, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 828, "step_time": 62.05200508888811 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 482.919921875, "completions/mean_terminated_length": 421.2134094238281, "completions/min_length": 197.0, "completions/min_terminated_length": 197.0, "entropy": 0.3791320533491671, "epoch": 0.9452679589509693, "frac_reward_zero_std": 0.171875, "grad_norm": 0.016093790531158447, "kl": 0.057121495832689106, "learning_rate": 4.743182316375439e-08, "loss": 0.0094, "num_tokens": 262554621.0, "reward": 1.1624999046325684, "reward_std": 0.9915247559547424, "rewards/code_complexity_reward/mean": 0.5199218988418579, "rewards/code_complexity_reward/std": 0.4071085453033447, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.314453125, "rewards/code_syntax_reward/std": 0.2417849749326706, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 829, "step_time": 55.78904914390296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.677734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 479.630859375, "completions/mean_terminated_length": 411.55755615234375, "completions/min_length": 179.0, "completions/min_terminated_length": 179.0, "entropy": 0.384926734957844, "epoch": 0.9464082098061574, "frac_reward_zero_std": 0.203125, "grad_norm": 0.01675310730934143, "kl": 0.05535334983142093, "learning_rate": 4.55214224722389e-08, "loss": 0.0118, "num_tokens": 262859932.0, "reward": 1.17724609375, "reward_std": 1.000672459602356, "rewards/code_complexity_reward/mean": 0.52685546875, "rewards/code_complexity_reward/std": 0.41215208172798157, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.314453125, "rewards/code_syntax_reward/std": 0.2417849749326706, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 830, "step_time": 46.08748653810471 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.673828125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 479.994140625, "completions/mean_terminated_length": 413.874267578125, "completions/min_length": 216.0, "completions/min_terminated_length": 216.0, "entropy": 0.37755833752453327, "epoch": 0.9475484606613455, "frac_reward_zero_std": 0.140625, "grad_norm": 0.016544751822948456, "kl": 0.05666457279585302, "learning_rate": 4.364993564959841e-08, "loss": 0.0124, "num_tokens": 263163201.0, "reward": 1.2195312976837158, "reward_std": 0.9562644362449646, "rewards/code_complexity_reward/mean": 0.554492175579071, "rewards/code_complexity_reward/std": 0.3951384425163269, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.3369140625, "rewards/code_syntax_reward/std": 0.23463475704193115, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 831, "step_time": 53.352405971847475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.701171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 481.41015625, "completions/mean_terminated_length": 409.6340026855469, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "entropy": 0.3936948371119797, "epoch": 0.9486887115165337, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.0151524618268013, "kl": 0.05713615455897525, "learning_rate": 4.1817392366815535e-08, "loss": 0.0115, "num_tokens": 263470455.0, "reward": 1.1393554210662842, "reward_std": 0.9809830784797668, "rewards/code_complexity_reward/mean": 0.515332043170929, "rewards/code_complexity_reward/std": 0.4076617956161499, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.3115234375, "rewards/code_syntax_reward/std": 0.24254849553108215, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 832, "step_time": 50.765134998597205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.705078125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.189453125, "completions/mean_terminated_length": 421.09271240234375, "completions/min_length": 190.0, "completions/min_terminated_length": 190.0, "entropy": 0.382407084107399, "epoch": 0.9498289623717218, "frac_reward_zero_std": 0.140625, "grad_norm": 0.016972266137599945, "kl": 0.054499436693731695, "learning_rate": 4.002382167745428e-08, "loss": 0.0083, "num_tokens": 263777584.0, "reward": 1.10986328125, "reward_std": 0.9661164283752441, "rewards/code_complexity_reward/mean": 0.51513671875, "rewards/code_complexity_reward/std": 0.4118361175060272, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.3095703125, "rewards/code_syntax_reward/std": 0.24303650856018066, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 833, "step_time": 54.97376901563257 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.68359375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 482.982421875, "completions/mean_terminated_length": 420.2901306152344, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.3811411103233695, "epoch": 0.95096921322691, "frac_reward_zero_std": 0.140625, "grad_norm": 0.016641536727547646, "kl": 0.059398088487796485, "learning_rate": 3.8269252017197056e-08, "loss": 0.0144, "num_tokens": 264083255.0, "reward": 1.1570801734924316, "reward_std": 0.9731950759887695, "rewards/code_complexity_reward/mean": 0.5308593511581421, "rewards/code_complexity_reward/std": 0.41024652123451233, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.3173828125, "rewards/code_syntax_reward/std": 0.24098336696624756, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 834, "step_time": 54.66255990322679 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 478.587890625, "completions/mean_terminated_length": 405.0812683105469, "completions/min_length": 182.0, "completions/min_terminated_length": 182.0, "entropy": 0.3948093578219414, "epoch": 0.9521094640820981, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.017105132341384888, "kl": 0.05544598394772038, "learning_rate": 3.6553711203395906e-08, "loss": 0.0122, "num_tokens": 264387844.0, "reward": 1.0895507335662842, "reward_std": 0.9897879958152771, "rewards/code_complexity_reward/mean": 0.5006835460662842, "rewards/code_complexity_reward/std": 0.4198412299156189, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.2978515625, "rewards/code_syntax_reward/std": 0.24561770260334015, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 835, "step_time": 47.088040770962834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.66796875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 477.220703125, "completions/mean_terminated_length": 407.2529602050781, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.38542649475857615, "epoch": 0.9532497149372862, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.01438361220061779, "kl": 0.05537168314913288, "learning_rate": 3.487722643463032e-08, "loss": 0.0109, "num_tokens": 264691973.0, "reward": 1.236328125, "reward_std": 1.0052886009216309, "rewards/code_complexity_reward/mean": 0.541015625, "rewards/code_complexity_reward/std": 0.406535804271698, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.32421875, "rewards/code_syntax_reward/std": 0.2389625608921051, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 836, "step_time": 54.25900964066386 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.66015625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 473.943359375, "completions/mean_terminated_length": 400.0172424316406, "completions/min_length": 200.0, "completions/min_terminated_length": 200.0, "entropy": 0.3873175694607198, "epoch": 0.9543899657924744, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.015081549063324928, "kl": 0.05792701366590336, "learning_rate": 3.323982429027567e-08, "loss": 0.0129, "num_tokens": 264993320.0, "reward": 1.271728515625, "reward_std": 0.9939224720001221, "rewards/code_complexity_reward/mean": 0.5642578601837158, "rewards/code_complexity_reward/std": 0.40554118156433105, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.3349609375, "rewards/code_syntax_reward/std": 0.23535043001174927, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.001220703125, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 837, "step_time": 45.98623274452984 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 483.9140625, "completions/mean_terminated_length": 406.26470947265625, "completions/min_length": 196.0, "completions/min_terminated_length": 196.0, "entropy": 0.3871090793982148, "epoch": 0.9555302166476625, "frac_reward_zero_std": 0.171875, "grad_norm": 0.01576339267194271, "kl": 0.05556627351325005, "learning_rate": 3.164153073008297e-08, "loss": 0.0062, "num_tokens": 265300888.0, "reward": 1.092431664466858, "reward_std": 0.9896690845489502, "rewards/code_complexity_reward/mean": 0.4901367127895355, "rewards/code_complexity_reward/std": 0.41165196895599365, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.296875, "rewards/code_syntax_reward/std": 0.24580632150173187, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000732421875, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 838, "step_time": 55.28603669349104 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.654296875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 477.41015625, "completions/mean_terminated_length": 411.9435119628906, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.381664068903774, "epoch": 0.9566704675028507, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.014394368976354599, "kl": 0.05601549445418641, "learning_rate": 3.0082371093766435e-08, "loss": 0.0096, "num_tokens": 265604710.0, "reward": 1.2526366710662842, "reward_std": 1.004425287246704, "rewards/code_complexity_reward/mean": 0.54541015625, "rewards/code_complexity_reward/std": 0.4047219157218933, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.3271484375, "rewards/code_syntax_reward/std": 0.2380310446023941, "rewards/reasoning_present_reward_func/mean": 0.00019531250291038305, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.0009765625, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 839, "step_time": 54.536724189296365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.693359375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 479.09375, "completions/mean_terminated_length": 404.6878967285156, "completions/min_length": 219.0, "completions/min_terminated_length": 219.0, "entropy": 0.3856565929017961, "epoch": 0.9578107183580388, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.014933819882571697, "kl": 0.05575723358197138, "learning_rate": 2.856237010060242e-08, "loss": 0.0128, "num_tokens": 265909934.0, "reward": 1.2423827648162842, "reward_std": 1.037963628768921, "rewards/code_complexity_reward/mean": 0.5275390148162842, "rewards/code_complexity_reward/std": 0.4129193127155304, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.314453125, "rewards/code_syntax_reward/std": 0.2417849749326706, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 840, "step_time": 55.087741187773645 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.677734375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 482.447265625, "completions/mean_terminated_length": 420.2969665527344, "completions/min_length": 210.0, "completions/min_terminated_length": 210.0, "entropy": 0.38526094797998667, "epoch": 0.9589509692132269, "frac_reward_zero_std": 0.140625, "grad_norm": 0.017016177996993065, "kl": 0.0555556858307682, "learning_rate": 2.708155184903666e-08, "loss": 0.0105, "num_tokens": 266216343.0, "reward": 1.2173340320587158, "reward_std": 0.96548992395401, "rewards/code_complexity_reward/mean": 0.546191394329071, "rewards/code_complexity_reward/std": 0.3947182297706604, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.3330078125, "rewards/code_syntax_reward/std": 0.23604771494865417, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 841, "step_time": 55.640036195516586 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7109375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.345703125, "completions/mean_terminated_length": 416.3310852050781, "completions/min_length": 272.0, "completions/min_terminated_length": 272.0, "entropy": 0.3833375070244074, "epoch": 0.9600912200684151, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.015102621167898178, "kl": 0.05434073106152937, "learning_rate": 2.5639939816302917e-08, "loss": 0.0095, "num_tokens": 266523124.0, "reward": 1.1070799827575684, "reward_std": 0.9856603741645813, "rewards/code_complexity_reward/mean": 0.5033202767372131, "rewards/code_complexity_reward/std": 0.41421499848365784, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.302734375, "rewards/code_syntax_reward/std": 0.2446138858795166, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 842, "step_time": 70.348487268202 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.689453125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 482.232421875, "completions/mean_terminated_length": 416.1446533203125, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.37921990640461445, "epoch": 0.9612314709236032, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.016987185925245285, "kl": 0.05696100904606283, "learning_rate": 2.4237556858050794e-08, "loss": 0.0098, "num_tokens": 266828323.0, "reward": 1.2703125476837158, "reward_std": 0.994153618812561, "rewards/code_complexity_reward/mean": 0.5471678972244263, "rewards/code_complexity_reward/std": 0.397357702255249, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.333984375, "rewards/code_syntax_reward/std": 0.23570136725902557, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 843, "step_time": 72.828689112328 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.736328125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.7265625, "completions/mean_terminated_length": 408.5629577636719, "completions/min_length": 220.0, "completions/min_terminated_length": 220.0, "entropy": 0.4003842961974442, "epoch": 0.9623717217787914, "frac_reward_zero_std": 0.171875, "grad_norm": 0.016497213393449783, "kl": 0.05557218159083277, "learning_rate": 2.2874425207982386e-08, "loss": 0.0073, "num_tokens": 267137043.0, "reward": 1.1024413108825684, "reward_std": 0.9747491478919983, "rewards/code_complexity_reward/mean": 0.5096679925918579, "rewards/code_complexity_reward/std": 0.41355374455451965, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.3056640625, "rewards/code_syntax_reward/std": 0.2439626157283783, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 844, "step_time": 70.42167349439114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.716796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 482.908203125, "completions/mean_terminated_length": 409.2758483886719, "completions/min_length": 199.0, "completions/min_terminated_length": 199.0, "entropy": 0.39402964524924755, "epoch": 0.9635119726339795, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.017669416964054108, "kl": 0.05528331029927358, "learning_rate": 2.15505664775012e-08, "loss": 0.0102, "num_tokens": 267443364.0, "reward": 1.1149413585662842, "reward_std": 1.0003260374069214, "rewards/code_complexity_reward/mean": 0.49775391817092896, "rewards/code_complexity_reward/std": 0.41379091143608093, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.30078125, "rewards/code_syntax_reward/std": 0.2450276017189026, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 845, "step_time": 53.32698624394834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.677734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 477.27734375, "completions/mean_terminated_length": 404.2545471191406, "completions/min_length": 221.0, "completions/min_terminated_length": 221.0, "entropy": 0.39059283351525664, "epoch": 0.9646522234891676, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.015625666826963425, "kl": 0.058628648810554296, "learning_rate": 2.026600165536824e-08, "loss": 0.0079, "num_tokens": 267746858.0, "reward": 1.217919945716858, "reward_std": 0.9745627045631409, "rewards/code_complexity_reward/mean": 0.5555664300918579, "rewards/code_complexity_reward/std": 0.4053860306739807, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.330078125, "rewards/code_syntax_reward/std": 0.2370595932006836, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 846, "step_time": 55.21191224176437 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.673828125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 477.66796875, "completions/mean_terminated_length": 406.7425231933594, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "entropy": 0.3769142432138324, "epoch": 0.9657924743443558, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.015727976337075233, "kl": 0.0622077101143077, "learning_rate": 1.9020751107370894e-08, "loss": 0.0132, "num_tokens": 268049356.0, "reward": 1.2341797351837158, "reward_std": 0.9950962066650391, "rewards/code_complexity_reward/mean": 0.5520508289337158, "rewards/code_complexity_reward/std": 0.40751346945762634, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.328125, "rewards/code_syntax_reward/std": 0.23771169781684875, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 847, "step_time": 51.307922217063606 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.662109375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 476.138671875, "completions/mean_terminated_length": 405.8670349121094, "completions/min_length": 232.0, "completions/min_terminated_length": 232.0, "entropy": 0.3825904312543571, "epoch": 0.9669327251995439, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.017556915059685707, "kl": 0.05573295726208016, "learning_rate": 1.7814834575997364e-08, "loss": 0.0157, "num_tokens": 268353559.0, "reward": 1.137109398841858, "reward_std": 0.9657894968986511, "rewards/code_complexity_reward/mean": 0.5248047113418579, "rewards/code_complexity_reward/std": 0.4051888883113861, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.3173828125, "rewards/code_syntax_reward/std": 0.24098336696624756, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 848, "step_time": 44.92920726817101 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.671875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 479.732421875, "completions/mean_terminated_length": 413.6607360839844, "completions/min_length": 194.0, "completions/min_terminated_length": 194.0, "entropy": 0.3859880226664245, "epoch": 0.9680729760547321, "frac_reward_zero_std": 0.171875, "grad_norm": 0.01672026328742504, "kl": 0.05442721338476986, "learning_rate": 1.664827118012663e-08, "loss": 0.0149, "num_tokens": 268658670.0, "reward": 1.210302710533142, "reward_std": 1.0181270837783813, "rewards/code_complexity_reward/mean": 0.5284179449081421, "rewards/code_complexity_reward/std": 0.4137060046195984, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.314453125, "rewards/code_syntax_reward/std": 0.2417849749326706, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 849, "step_time": 44.48531174659729 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.70703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 483.953125, "completions/mean_terminated_length": 416.26666259765625, "completions/min_length": 264.0, "completions/min_terminated_length": 264.0, "entropy": 0.38935075094923377, "epoch": 0.9692132269099202, "frac_reward_zero_std": 0.15625, "grad_norm": 0.01574833132326603, "kl": 0.05432437197305262, "learning_rate": 1.5521079414723695e-08, "loss": 0.0105, "num_tokens": 268966278.0, "reward": 1.1255371570587158, "reward_std": 0.999856173992157, "rewards/code_complexity_reward/mean": 0.5028320550918579, "rewards/code_complexity_reward/std": 0.4150976240634918, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.302734375, "rewards/code_syntax_reward/std": 0.2446138858795166, "rewards/reasoning_present_reward_func/mean": 0.0003906250058207661, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.001220703125, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 850, "step_time": 52.89725970104337 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.7265625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 482.943359375, "completions/mean_terminated_length": 405.7357177734375, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "entropy": 0.40004150196909904, "epoch": 0.9703534777651083, "frac_reward_zero_std": 0.25, "grad_norm": 0.01612093672156334, "kl": 0.055199221649672836, "learning_rate": 1.4433277150545932e-08, "loss": 0.0094, "num_tokens": 269274537.0, "reward": 0.9786133170127869, "reward_std": 0.9767285585403442, "rewards/code_complexity_reward/mean": 0.4532226622104645, "rewards/code_complexity_reward/std": 0.42062103748321533, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.2734375, "rewards/code_syntax_reward/std": 0.2491423636674881, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 851, "step_time": 65.7136535346508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.69140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 483.271484375, "completions/mean_terminated_length": 418.9050598144531, "completions/min_length": 173.0, "completions/min_terminated_length": 173.0, "entropy": 0.38057516794651747, "epoch": 0.9714937286202965, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.01630145125091076, "kl": 0.05553764593787491, "learning_rate": 1.3384881633861923e-08, "loss": 0.011, "num_tokens": 269581212.0, "reward": 1.2087889909744263, "reward_std": 0.9520196318626404, "rewards/code_complexity_reward/mean": 0.5554687976837158, "rewards/code_complexity_reward/std": 0.3977300822734833, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.3349609375, "rewards/code_syntax_reward/std": 0.23535043001174927, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 852, "step_time": 55.45391982793808 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.701171875, "completions/mean_terminated_length": 402.8046875, "completions/min_length": 220.0, "completions/min_terminated_length": 220.0, "entropy": 0.39735572086647153, "epoch": 0.9726339794754846, "frac_reward_zero_std": 0.203125, "grad_norm": 0.017715517431497574, "kl": 0.05677483818726614, "learning_rate": 1.2375909486175008e-08, "loss": 0.0108, "num_tokens": 269890027.0, "reward": 1.0925781726837158, "reward_std": 1.002132534980774, "rewards/code_complexity_reward/mean": 0.48808592557907104, "rewards/code_complexity_reward/std": 0.41721123456954956, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.2939453125, "rewards/code_syntax_reward/std": 0.24634800851345062, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 853, "step_time": 53.71031263936311 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.689453125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 481.6796875, "completions/mean_terminated_length": 414.3647766113281, "completions/min_length": 234.0, "completions/min_terminated_length": 234.0, "entropy": 0.38907777797430754, "epoch": 0.9737742303306728, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.016241248697042465, "kl": 0.058466823771595955, "learning_rate": 1.1406376703962384e-08, "loss": 0.0108, "num_tokens": 270195551.0, "reward": 1.1018555164337158, "reward_std": 0.9895696043968201, "rewards/code_complexity_reward/mean": 0.5041992664337158, "rewards/code_complexity_reward/std": 0.4194360077381134, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.298828125, "rewards/code_syntax_reward/std": 0.2454250603914261, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 854, "step_time": 45.78190585318953 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.68359375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 480.82421875, "completions/mean_terminated_length": 413.4691467285156, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.38660824252292514, "epoch": 0.9749144811858609, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.017324766144156456, "kl": 0.056165858753956854, "learning_rate": 1.0476298658420036e-08, "loss": 0.0111, "num_tokens": 270500689.0, "reward": 1.2056641578674316, "reward_std": 0.9781553149223328, "rewards/code_complexity_reward/mean": 0.5396484136581421, "rewards/code_complexity_reward/std": 0.40270641446113586, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.326171875, "rewards/code_syntax_reward/std": 0.23834596574306488, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 855, "step_time": 51.4732313612476 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.669921875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 476.240234375, "completions/mean_terminated_length": 403.6627197265625, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.39053900167346, "epoch": 0.976054732041049, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.014498245902359486, "kl": 0.05480702145723626, "learning_rate": 9.58569009521959e-09, "loss": 0.0101, "num_tokens": 270806932.0, "reward": 1.1818358898162842, "reward_std": 1.0099225044250488, "rewards/code_complexity_reward/mean": 0.5236327648162842, "rewards/code_complexity_reward/std": 0.4160056710243225, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.310546875, "rewards/code_syntax_reward/std": 0.24279458820819855, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 856, "step_time": 60.618847442790866 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.689453125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 483.4453125, "completions/mean_terminated_length": 420.05029296875, "completions/min_length": 186.0, "completions/min_terminated_length": 186.0, "entropy": 0.3966613099910319, "epoch": 0.9771949828962372, "frac_reward_zero_std": 0.1875, "grad_norm": 0.01583501324057579, "kl": 0.056559556687716395, "learning_rate": 8.73456513427462e-09, "loss": 0.0113, "num_tokens": 271115480.0, "reward": 1.093408226966858, "reward_std": 1.002724528312683, "rewards/code_complexity_reward/mean": 0.4876953065395355, "rewards/code_complexity_reward/std": 0.41501298546791077, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.294921875, "rewards/code_syntax_reward/std": 0.24617145955562592, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 857, "step_time": 59.17242847662419 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.673828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 478.427734375, "completions/mean_terminated_length": 409.0718688964844, "completions/min_length": 220.0, "completions/min_terminated_length": 220.0, "entropy": 0.37867012713104486, "epoch": 0.9783352337514253, "frac_reward_zero_std": 0.125, "grad_norm": 0.018218878656625748, "kl": 0.05841673689428717, "learning_rate": 7.922937269516096e-09, "loss": 0.0125, "num_tokens": 271417983.0, "reward": 1.208251953125, "reward_std": 0.9840964078903198, "rewards/code_complexity_reward/mean": 0.5419921875, "rewards/code_complexity_reward/std": 0.40595415234565735, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.32421875, "rewards/code_syntax_reward/std": 0.2389625608921051, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 858, "step_time": 53.000407077372074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.642578125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 474.47265625, "completions/mean_terminated_length": 407.0054626464844, "completions/min_length": 180.0, "completions/min_terminated_length": 180.0, "entropy": 0.3819176536053419, "epoch": 0.9794754846066135, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.01472719106823206, "kl": 0.05652491911314428, "learning_rate": 7.150819368679229e-09, "loss": 0.0123, "num_tokens": 271720165.0, "reward": 1.1785156726837158, "reward_std": 0.9926477670669556, "rewards/code_complexity_reward/mean": 0.529589831829071, "rewards/code_complexity_reward/std": 0.4114772379398346, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.31640625, "rewards/code_syntax_reward/std": 0.24125482141971588, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 859, "step_time": 45.400524228811264 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.70703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 482.1953125, "completions/mean_terminated_length": 410.26666259765625, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.38490858068689704, "epoch": 0.9806157354618016, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.01673400215804577, "kl": 0.057910605624783784, "learning_rate": 6.418223673099466e-09, "loss": 0.0142, "num_tokens": 272025925.0, "reward": 1.22998046875, "reward_std": 1.0282230377197266, "rewards/code_complexity_reward/mean": 0.529296875, "rewards/code_complexity_reward/std": 0.41193196177482605, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.3154296875, "rewards/code_syntax_reward/std": 0.24152201414108276, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 860, "step_time": 44.96849089115858 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 477.78515625, "completions/mean_terminated_length": 402.51251220703125, "completions/min_length": 229.0, "completions/min_terminated_length": 229.0, "entropy": 0.392869156319648, "epoch": 0.9817559863169898, "frac_reward_zero_std": 0.1875, "grad_norm": 0.016329888254404068, "kl": 0.057891942153219134, "learning_rate": 5.725161797517087e-09, "loss": 0.0095, "num_tokens": 272329715.0, "reward": 1.1690430641174316, "reward_std": 1.0099183320999146, "rewards/code_complexity_reward/mean": 0.5127929449081421, "rewards/code_complexity_reward/std": 0.41246169805526733, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.30859375, "rewards/code_syntax_reward/std": 0.2432742565870285, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 861, "step_time": 64.01232600025833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 480.66015625, "completions/mean_terminated_length": 406.4342041015625, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "entropy": 0.3866485059261322, "epoch": 0.9828962371721779, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.014793185517191887, "kl": 0.05512247659498826, "learning_rate": 5.071644729895408e-09, "loss": 0.0073, "num_tokens": 272636197.0, "reward": 1.15771484375, "reward_std": 1.0142980813980103, "rewards/code_complexity_reward/mean": 0.50634765625, "rewards/code_complexity_reward/std": 0.4121013879776001, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.3056640625, "rewards/code_syntax_reward/std": 0.2439626157283783, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 862, "step_time": 46.005457701161504 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.662109375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 480.00390625, "completions/mean_terminated_length": 417.30633544921875, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "entropy": 0.3754178681410849, "epoch": 0.984036488027366, "frac_reward_zero_std": 0.1875, "grad_norm": 0.015045646578073502, "kl": 0.05626372044207528, "learning_rate": 4.4576828312442586e-09, "loss": 0.0093, "num_tokens": 272941447.0, "reward": 1.2390624284744263, "reward_std": 0.9902644753456116, "rewards/code_complexity_reward/mean": 0.555468738079071, "rewards/code_complexity_reward/std": 0.40516403317451477, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.330078125, "rewards/code_syntax_reward/std": 0.2370595932006836, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 863, "step_time": 51.67612227983773 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.720703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 485.26171875, "completions/mean_terminated_length": 416.2657165527344, "completions/min_length": 226.0, "completions/min_terminated_length": 226.0, "entropy": 0.39418822387233377, "epoch": 0.9851767388825542, "frac_reward_zero_std": 0.1875, "grad_norm": 0.01514471136033535, "kl": 0.057297655555885285, "learning_rate": 3.8832858354567736e-09, "loss": 0.0152, "num_tokens": 273249641.0, "reward": 0.9962890148162842, "reward_std": 0.9589397311210632, "rewards/code_complexity_reward/mean": 0.46308591961860657, "rewards/code_complexity_reward/std": 0.41263630986213684, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.283203125, "rewards/code_syntax_reward/std": 0.24802762269973755, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 864, "step_time": 53.11571353767067 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.71875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 488.775390625, "completions/mean_terminated_length": 429.4236145019531, "completions/min_length": 286.0, "completions/min_terminated_length": 286.0, "entropy": 0.3811192004941404, "epoch": 0.9863169897377423, "frac_reward_zero_std": 0.203125, "grad_norm": 0.016141505911946297, "kl": 0.05338582134572789, "learning_rate": 3.348462849155909e-09, "loss": 0.0122, "num_tokens": 273559938.0, "reward": 1.1174805164337158, "reward_std": 1.0180379152297974, "rewards/code_complexity_reward/mean": 0.49736326932907104, "rewards/code_complexity_reward/std": 0.4194720983505249, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.2958984375, "rewards/code_syntax_reward/std": 0.24599088728427887, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 865, "step_time": 50.511646039783955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.720703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 483.869140625, "completions/mean_terminated_length": 411.27972412109375, "completions/min_length": 190.0, "completions/min_terminated_length": 190.0, "entropy": 0.39028475200757384, "epoch": 0.9874572405929305, "frac_reward_zero_std": 0.25, "grad_norm": 0.017372073605656624, "kl": 0.05519482138333842, "learning_rate": 2.8532223515481683e-09, "loss": 0.0078, "num_tokens": 273867447.0, "reward": 1.1175780296325684, "reward_std": 0.9894371628761292, "rewards/code_complexity_reward/mean": 0.5062500238418579, "rewards/code_complexity_reward/std": 0.41242918372154236, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.3046875, "rewards/code_syntax_reward/std": 0.24418380856513977, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 866, "step_time": 45.219196829013526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 484.74609375, "completions/mean_terminated_length": 422.5513000488281, "completions/min_length": 223.0, "completions/min_terminated_length": 223.0, "entropy": 0.38755055982619524, "epoch": 0.9885974914481186, "frac_reward_zero_std": 0.109375, "grad_norm": 0.017229294404387474, "kl": 0.05669003649381921, "learning_rate": 2.3975721942903762e-09, "loss": 0.0139, "num_tokens": 274176197.0, "reward": 1.2115235328674316, "reward_std": 0.9863768815994263, "rewards/code_complexity_reward/mean": 0.5484375357627869, "rewards/code_complexity_reward/std": 0.40839704871177673, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.3251953125, "rewards/code_syntax_reward/std": 0.23865646123886108, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 867, "step_time": 77.37497315276414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.671875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 480.845703125, "completions/mean_terminated_length": 417.0535888671875, "completions/min_length": 227.0, "completions/min_terminated_length": 227.0, "entropy": 0.3842049897648394, "epoch": 0.9897377423033067, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.016126127913594246, "kl": 0.05689837038516998, "learning_rate": 1.9815196013650563e-09, "loss": 0.0101, "num_tokens": 274480582.0, "reward": 1.224218726158142, "reward_std": 0.9771906733512878, "rewards/code_complexity_reward/mean": 0.5562499761581421, "rewards/code_complexity_reward/std": 0.406587153673172, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.330078125, "rewards/code_syntax_reward/std": 0.2370595932006836, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 868, "step_time": 63.785021058283746 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 480.470703125, "completions/mean_terminated_length": 411.10626220703125, "completions/min_length": 204.0, "completions/min_terminated_length": 204.0, "entropy": 0.3942759712226689, "epoch": 0.9908779931584949, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.01579565741121769, "kl": 0.0599643659661524, "learning_rate": 1.6050711689663544e-09, "loss": 0.0128, "num_tokens": 274785031.0, "reward": 1.0978515148162842, "reward_std": 0.9855091571807861, "rewards/code_complexity_reward/mean": 0.49726563692092896, "rewards/code_complexity_reward/std": 0.4139121174812317, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.2998046875, "rewards/code_syntax_reward/std": 0.2452283650636673, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 869, "step_time": 51.75064655113965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.693359375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 479.0859375, "completions/mean_terminated_length": 404.66241455078125, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.39121107663959265, "epoch": 0.992018244013683, "frac_reward_zero_std": 0.15625, "grad_norm": 0.015910295769572258, "kl": 0.05874667863827199, "learning_rate": 1.2682328653940146e-09, "loss": 0.0117, "num_tokens": 275087207.0, "reward": 1.156494140625, "reward_std": 0.9523033499717712, "rewards/code_complexity_reward/mean": 0.5380859375, "rewards/code_complexity_reward/std": 0.4030153751373291, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.3251953125, "rewards/code_syntax_reward/std": 0.23865646123886108, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 870, "step_time": 51.634260601364076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6484375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 475.048828125, "completions/mean_terminated_length": 406.8944396972656, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 0.3705503074452281, "epoch": 0.9931584948688712, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.01591453328728676, "kl": 0.05599206825718284, "learning_rate": 9.710100309603954e-10, "loss": 0.0143, "num_tokens": 275390288.0, "reward": 1.31103515625, "reward_std": 1.0255686044692993, "rewards/code_complexity_reward/mean": 0.55517578125, "rewards/code_complexity_reward/std": 0.4040857255458832, "rewards/code_execution_reward/mean": 0.423828125, "rewards/code_execution_reward/std": 0.4946470856666565, "rewards/code_syntax_reward/mean": 0.33203125, "rewards/code_syntax_reward/std": 0.23638953268527985, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 871, "step_time": 59.39597074035555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.705078125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 482.7734375, "completions/mean_terminated_length": 412.9006652832031, "completions/min_length": 210.0, "completions/min_terminated_length": 210.0, "entropy": 0.388024028390646, "epoch": 0.9942987457240593, "frac_reward_zero_std": 0.15625, "grad_norm": 0.015991192311048508, "kl": 0.0564937480376102, "learning_rate": 7.134073779044293e-10, "loss": 0.0133, "num_tokens": 275696192.0, "reward": 1.0510742664337158, "reward_std": 0.9979890584945679, "rewards/code_complexity_reward/mean": 0.47099608182907104, "rewards/code_complexity_reward/std": 0.41909366846084595, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.283203125, "rewards/code_syntax_reward/std": 0.24802762269973755, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 872, "step_time": 54.9336155038327 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.69140625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 481.189453125, "completions/mean_terminated_length": 412.1582336425781, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.3869183002971113, "epoch": 0.9954389965792474, "frac_reward_zero_std": 0.15625, "grad_norm": 0.01652613840997219, "kl": 0.05600163631606847, "learning_rate": 4.954289903180698e-10, "loss": 0.0078, "num_tokens": 276001765.0, "reward": 1.1967284679412842, "reward_std": 0.9905968904495239, "rewards/code_complexity_reward/mean": 0.531445324420929, "rewards/code_complexity_reward/std": 0.40430349111557007, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.3212890625, "rewards/code_syntax_reward/std": 0.2398546040058136, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 873, "step_time": 53.857097887434065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6484375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 481.203125, "completions/mean_terminated_length": 424.4000244140625, "completions/min_length": 204.0, "completions/min_terminated_length": 204.0, "entropy": 0.3744994541630149, "epoch": 0.9965792474344356, "frac_reward_zero_std": 0.1875, "grad_norm": 0.015226436778903008, "kl": 0.055860173946712166, "learning_rate": 3.1707832408134354e-10, "loss": 0.0095, "num_tokens": 276307009.0, "reward": 1.2738280296325684, "reward_std": 1.0013927221298218, "rewards/code_complexity_reward/mean": 0.5516601800918579, "rewards/code_complexity_reward/std": 0.39981913566589355, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.3330078125, "rewards/code_syntax_reward/std": 0.23604771494865417, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.00048828125, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 874, "step_time": 52.13291462417692 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.65625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 476.384765625, "completions/mean_terminated_length": 408.3920593261719, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.37104156240820885, "epoch": 0.9977194982896237, "frac_reward_zero_std": 0.21875, "grad_norm": 0.014924569055438042, "kl": 0.05756498983828351, "learning_rate": 1.7835820680600634e-10, "loss": 0.011, "num_tokens": 276610398.0, "reward": 1.225732445716858, "reward_std": 0.9768495559692383, "rewards/code_complexity_reward/mean": 0.5477539300918579, "rewards/code_complexity_reward/std": 0.4002687931060791, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.33203125, "rewards/code_syntax_reward/std": 0.23638953268527985, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 875, "step_time": 45.91719316598028 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.712890625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 485.697265625, "completions/mean_terminated_length": 420.38775634765625, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.38721349323168397, "epoch": 0.9988597491448119, "frac_reward_zero_std": 0.15625, "grad_norm": 0.01751933991909027, "kl": 0.058727525582071394, "learning_rate": 7.927083779335487e-11, "loss": 0.0117, "num_tokens": 276918731.0, "reward": 1.1282227039337158, "reward_std": 1.0078601837158203, "rewards/code_complexity_reward/mean": 0.49833986163139343, "rewards/code_complexity_reward/std": 0.4143490195274353, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.2998046875, "rewards/code_syntax_reward/std": 0.2452283650636673, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 876, "step_time": 51.92184309847653 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.6875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 476.984375, "completions/mean_terminated_length": 399.95001220703125, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.3824261953122914, "epoch": 1.0, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.017148463055491447, "kl": 0.05499977199360728, "learning_rate": 1.98177879973116e-11, "loss": 0.0114, "num_tokens": 277222079.0, "reward": 1.1712403297424316, "reward_std": 1.0002148151397705, "rewards/code_complexity_reward/mean": 0.5147460699081421, "rewards/code_complexity_reward/std": 0.4084627330303192, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.3125, "rewards/code_syntax_reward/std": 0.2422981858253479, "rewards/reasoning_present_reward_func/mean": 0.0, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.000244140625, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 877, "step_time": 60.59415504988283 } ], "logging_steps": 1, "max_steps": 877, "num_input_tokens_seen": 277222079, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 1, "trial_name": null, "trial_params": null }