{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 100, "global_step": 877, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.056640625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 303.3828125, "completions/mean_terminated_length": 290.8571472167969, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.26919206441380084, "epoch": 0.0011402508551881414, "frac_reward_zero_std": 0.0, "grad_norm": 0.054739631712436676, "kl": 0.0, "learning_rate": 0.0, "loss": -2.4432665668427944e-08, "num_tokens": 224556.0, "reward": 1.913232445716858, "reward_std": 0.8212127089500427, "rewards/code_complexity_reward/mean": 0.6744140982627869, "rewards/code_complexity_reward/std": 0.31203773617744446, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.419921875, "rewards/code_syntax_reward/std": 0.1835547834634781, "rewards/reasoning_present_reward_func/mean": 0.09062500298023224, "rewards/reasoning_present_reward_func/std": 0.029176566749811172, "rewards/xmlcount_reward_func/mean": 0.417724609375, "rewards/xmlcount_reward_func/std": 0.10183995962142944, "step": 1, "step_time": 65.25882855989039 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 306.810546875, "completions/mean_terminated_length": 286.5557861328125, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.2674515324179083, "epoch": 0.002280501710376283, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.044257719069719315, "kl": 0.0, "learning_rate": 5.681818181818182e-08, "loss": 2.4621840566396713e-08, "num_tokens": 451403.0, "reward": 1.8957030773162842, "reward_std": 0.8459895253181458, "rewards/code_complexity_reward/mean": 0.6695312261581421, "rewards/code_complexity_reward/std": 0.32338622212409973, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4111328125, "rewards/code_syntax_reward/std": 0.1913314312696457, "rewards/reasoning_present_reward_func/mean": 0.08847656846046448, "rewards/reasoning_present_reward_func/std": 0.03196168690919876, "rewards/xmlcount_reward_func/mean": 0.41796875, "rewards/xmlcount_reward_func/std": 0.09790803492069244, "step": 2, "step_time": 68.8704728987068 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.083984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 310.265625, "completions/mean_terminated_length": 291.7697448730469, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2657131287269294, "epoch": 0.0034207525655644243, "frac_reward_zero_std": 0.0, "grad_norm": 0.04048004746437073, "kl": 0.001394905884808395, "learning_rate": 1.1363636363636364e-07, "loss": 6.946065695956349e-06, "num_tokens": 680791.0, "reward": 1.9125487804412842, "reward_std": 0.7814038991928101, "rewards/code_complexity_reward/mean": 0.6959960460662842, "rewards/code_complexity_reward/std": 0.3049471080303192, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.42578125, "rewards/code_syntax_reward/std": 0.17794041335582733, "rewards/reasoning_present_reward_func/mean": 0.0908203125, "rewards/reasoning_present_reward_func/std": 0.028902137652039528, "rewards/xmlcount_reward_func/mean": 0.424560546875, "rewards/xmlcount_reward_func/std": 0.09094821661710739, "step": 3, "step_time": 90.492297927849 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.076171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 308.748046875, "completions/mean_terminated_length": 291.9894104003906, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2624019703362137, "epoch": 0.004561003420752566, "frac_reward_zero_std": 0.0, "grad_norm": 0.04836735129356384, "kl": 0.0013854751477992977, "learning_rate": 1.7045454545454545e-07, "loss": 6.8562221713364124e-06, "num_tokens": 908526.0, "reward": 1.9422852993011475, "reward_std": 0.8026689291000366, "rewards/code_complexity_reward/mean": 0.6959960460662842, "rewards/code_complexity_reward/std": 0.308361291885376, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4248046875, "rewards/code_syntax_reward/std": 0.17890173196792603, "rewards/reasoning_present_reward_func/mean": 0.09003905951976776, "rewards/reasoning_present_reward_func/std": 0.029977135360240936, "rewards/xmlcount_reward_func/mean": 0.4228515625, "rewards/xmlcount_reward_func/std": 0.09301838278770447, "step": 4, "step_time": 58.93293912895024 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.041015625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 303.591796875, "completions/mean_terminated_length": 294.67822265625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2658364539965987, "epoch": 0.005701254275940707, "frac_reward_zero_std": 0.0, "grad_norm": 0.05292944982647896, "kl": 0.001422632512912969, "learning_rate": 2.2727272727272729e-07, "loss": 7.0973183028399944e-06, "num_tokens": 1130721.0, "reward": 1.9368653297424316, "reward_std": 0.7618821859359741, "rewards/code_complexity_reward/mean": 0.696972668170929, "rewards/code_complexity_reward/std": 0.2903294861316681, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.435546875, "rewards/code_syntax_reward/std": 0.16771192848682404, "rewards/reasoning_present_reward_func/mean": 0.09218749403953552, "rewards/reasoning_present_reward_func/std": 0.026863064616918564, "rewards/xmlcount_reward_func/mean": 0.427001953125, "rewards/xmlcount_reward_func/std": 0.09226252883672714, "step": 5, "step_time": 57.29660900589079 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 311.880859375, "completions/mean_terminated_length": 298.53961181640625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2695745942182839, "epoch": 0.0068415051311288486, "frac_reward_zero_std": 0.015625, "grad_norm": 0.049079060554504395, "kl": 0.0014245347747419146, "learning_rate": 2.840909090909091e-07, "loss": 7.107970304787159e-06, "num_tokens": 1361068.0, "reward": 1.8453125953674316, "reward_std": 0.8025766015052795, "rewards/code_complexity_reward/mean": 0.6678710579872131, "rewards/code_complexity_reward/std": 0.3178141415119171, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.416015625, "rewards/code_syntax_reward/std": 0.1871020793914795, "rewards/reasoning_present_reward_func/mean": 0.08906249701976776, "rewards/reasoning_present_reward_func/std": 0.031241437420248985, "rewards/xmlcount_reward_func/mean": 0.41845703125, "rewards/xmlcount_reward_func/std": 0.10257764905691147, "step": 6, "step_time": 60.45865281764418 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.072265625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 307.90625, "completions/mean_terminated_length": 292.0084228515625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2670722322072834, "epoch": 0.00798175598631699, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.042982444167137146, "kl": 0.0013887173881812487, "learning_rate": 3.409090909090909e-07, "loss": 6.87985448166728e-06, "num_tokens": 1589252.0, "reward": 1.8611328601837158, "reward_std": 0.8102197647094727, "rewards/code_complexity_reward/mean": 0.673828125, "rewards/code_complexity_reward/std": 0.3213057219982147, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4150390625, "rewards/code_syntax_reward/std": 0.1879657357931137, "rewards/reasoning_present_reward_func/mean": 0.08964844048023224, "rewards/reasoning_present_reward_func/std": 0.030492907389998436, "rewards/xmlcount_reward_func/mean": 0.4228515625, "rewards/xmlcount_reward_func/std": 0.09399937838315964, "step": 7, "step_time": 63.57449244149029 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.083984375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 309.57421875, "completions/mean_terminated_length": 291.0149230957031, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.25342261767946184, "epoch": 0.009122006841505131, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04456686973571777, "kl": 0.0013067017316643614, "learning_rate": 3.9772727272727276e-07, "loss": 6.377929821610451e-06, "num_tokens": 1815566.0, "reward": 1.9292480945587158, "reward_std": 0.8287535309791565, "rewards/code_complexity_reward/mean": 0.6808593273162842, "rewards/code_complexity_reward/std": 0.31824007630348206, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.419921875, "rewards/code_syntax_reward/std": 0.1835547834634781, "rewards/reasoning_present_reward_func/mean": 0.08945313096046448, "rewards/reasoning_present_reward_func/std": 0.03074568696320057, "rewards/xmlcount_reward_func/mean": 0.420654296875, "rewards/xmlcount_reward_func/std": 0.09840663522481918, "step": 8, "step_time": 71.80946519691497 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.083984375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 318.333984375, "completions/mean_terminated_length": 300.57781982421875, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.2624082297552377, "epoch": 0.010262257696693273, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04423991218209267, "kl": 0.0013378892426771927, "learning_rate": 4.5454545454545457e-07, "loss": 6.694215699099004e-06, "num_tokens": 2047029.0, "reward": 1.8522460460662842, "reward_std": 0.7931056022644043, "rewards/code_complexity_reward/mean": 0.6708008050918579, "rewards/code_complexity_reward/std": 0.3193099796772003, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4150390625, "rewards/code_syntax_reward/std": 0.1879657357931137, "rewards/reasoning_present_reward_func/mean": 0.08964844048023224, "rewards/reasoning_present_reward_func/std": 0.030492907389998436, "rewards/xmlcount_reward_func/mean": 0.4169921875, "rewards/xmlcount_reward_func/std": 0.10048481076955795, "step": 9, "step_time": 71.67954256013036 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 310.857421875, "completions/mean_terminated_length": 298.33819580078125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2661251896061003, "epoch": 0.011402508551881414, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04904405027627945, "kl": 0.0013480451698342222, "learning_rate": 5.113636363636364e-07, "loss": 6.753427442163229e-06, "num_tokens": 2275344.0, "reward": 1.8457520008087158, "reward_std": 0.7956724166870117, "rewards/code_complexity_reward/mean": 0.666210949420929, "rewards/code_complexity_reward/std": 0.3184645175933838, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.416015625, "rewards/code_syntax_reward/std": 0.1871020793914795, "rewards/reasoning_present_reward_func/mean": 0.09140625596046448, "rewards/reasoning_present_reward_func/std": 0.028054581955075264, "rewards/xmlcount_reward_func/mean": 0.422119140625, "rewards/xmlcount_reward_func/std": 0.09739890694618225, "step": 10, "step_time": 59.25299417972565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 319.69921875, "completions/mean_terminated_length": 296.0833435058594, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.26384456432424486, "epoch": 0.012542759407069556, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04821447655558586, "kl": 0.0014376538329088362, "learning_rate": 5.681818181818182e-07, "loss": 7.153430487960577e-06, "num_tokens": 2509882.0, "reward": 1.818701148033142, "reward_std": 0.8087571859359741, "rewards/code_complexity_reward/mean": 0.6592773199081421, "rewards/code_complexity_reward/std": 0.3273221254348755, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.408203125, "rewards/code_syntax_reward/std": 0.1937655806541443, "rewards/reasoning_present_reward_func/mean": 0.087890625, "rewards/reasoning_present_reward_func/std": 0.03265552595257759, "rewards/xmlcount_reward_func/mean": 0.419189453125, "rewards/xmlcount_reward_func/std": 0.09999597817659378, "step": 11, "step_time": 97.0877822432667 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 315.490234375, "completions/mean_terminated_length": 295.1616516113281, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.26628425088711083, "epoch": 0.013683010262257697, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04648621007800102, "kl": 0.0013891680027882103, "learning_rate": 6.25e-07, "loss": 6.885093171149492e-06, "num_tokens": 2741193.0, "reward": 1.8626465797424316, "reward_std": 0.8352715969085693, "rewards/code_complexity_reward/mean": 0.6597656011581421, "rewards/code_complexity_reward/std": 0.33046895265579224, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.408203125, "rewards/code_syntax_reward/std": 0.1937655806541443, "rewards/reasoning_present_reward_func/mean": 0.087890625, "rewards/reasoning_present_reward_func/std": 0.03265552595257759, "rewards/xmlcount_reward_func/mean": 0.417724609375, "rewards/xmlcount_reward_func/std": 0.0962839275598526, "step": 12, "step_time": 58.48564247135073 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 308.91796875, "completions/mean_terminated_length": 291.7076416015625, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.2597211448010057, "epoch": 0.014823261117445839, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03931799530982971, "kl": 0.0013798267882521031, "learning_rate": 6.818181818181818e-07, "loss": 6.9084635470062494e-06, "num_tokens": 2966731.0, "reward": 1.8998534679412842, "reward_std": 0.8028135299682617, "rewards/code_complexity_reward/mean": 0.6845703125, "rewards/code_complexity_reward/std": 0.31367436051368713, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4208984375, "rewards/code_syntax_reward/std": 0.18264412879943848, "rewards/reasoning_present_reward_func/mean": 0.08906249701976776, "rewards/reasoning_present_reward_func/std": 0.031241437420248985, "rewards/xmlcount_reward_func/mean": 0.418212890625, "rewards/xmlcount_reward_func/std": 0.10193374752998352, "step": 13, "step_time": 59.53062731400132 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 312.841796875, "completions/mean_terminated_length": 294.1175231933594, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.26285994169302285, "epoch": 0.01596351197263398, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.049488212913274765, "kl": 0.0013794846454402432, "learning_rate": 7.386363636363638e-07, "loss": 6.783287972211838e-06, "num_tokens": 3194014.0, "reward": 1.888427734375, "reward_std": 0.8239588141441345, "rewards/code_complexity_reward/mean": 0.674023449420929, "rewards/code_complexity_reward/std": 0.3223097622394562, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4130859375, "rewards/code_syntax_reward/std": 0.18966612219810486, "rewards/reasoning_present_reward_func/mean": 0.09160156548023224, "rewards/reasoning_present_reward_func/std": 0.02776356413960457, "rewards/xmlcount_reward_func/mean": 0.422607421875, "rewards/xmlcount_reward_func/std": 0.09778810292482376, "step": 14, "step_time": 62.949358792975545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.076171875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 306.654296875, "completions/mean_terminated_length": 289.7230529785156, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.26849856693297625, "epoch": 0.01710376282782212, "frac_reward_zero_std": 0.0, "grad_norm": 0.0515284426510334, "kl": 0.0013463714631143375, "learning_rate": 7.954545454545455e-07, "loss": 6.704271072521806e-06, "num_tokens": 3418373.0, "reward": 1.8484864234924316, "reward_std": 0.8485332727432251, "rewards/code_complexity_reward/mean": 0.6571289300918579, "rewards/code_complexity_reward/std": 0.33999964594841003, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.40234375, "rewards/code_syntax_reward/std": 0.1984144002199173, "rewards/reasoning_present_reward_func/mean": 0.09003905951976776, "rewards/reasoning_present_reward_func/std": 0.029977135360240936, "rewards/xmlcount_reward_func/mean": 0.415771484375, "rewards/xmlcount_reward_func/std": 0.09775634109973907, "step": 15, "step_time": 69.50452680978924 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.072265625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 312.107421875, "completions/mean_terminated_length": 296.5368347167969, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.26059392280876637, "epoch": 0.018244013683010263, "frac_reward_zero_std": 0.0, "grad_norm": 0.04968167096376419, "kl": 0.0013567409687311738, "learning_rate": 8.522727272727273e-07, "loss": 6.775546353310347e-06, "num_tokens": 3649044.0, "reward": 1.9398436546325684, "reward_std": 0.82467120885849, "rewards/code_complexity_reward/mean": 0.6868164539337158, "rewards/code_complexity_reward/std": 0.3099413216114044, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.423828125, "rewards/code_syntax_reward/std": 0.17985260486602783, "rewards/reasoning_present_reward_func/mean": 0.08847656846046448, "rewards/reasoning_present_reward_func/std": 0.03196168690919876, "rewards/xmlcount_reward_func/mean": 0.42236328125, "rewards/xmlcount_reward_func/std": 0.09743722528219223, "step": 16, "step_time": 71.7255785157904 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 304.390625, "completions/mean_terminated_length": 292.3801574707031, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.25844030966982245, "epoch": 0.019384264538198404, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.055274847894907, "kl": 0.001335348169959616, "learning_rate": 9.090909090909091e-07, "loss": 6.6183984017698094e-06, "num_tokens": 3872676.0, "reward": 1.8698241710662842, "reward_std": 0.8310775756835938, "rewards/code_complexity_reward/mean": 0.664355456829071, "rewards/code_complexity_reward/std": 0.32413360476493835, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.412109375, "rewards/code_syntax_reward/std": 0.1905031055212021, "rewards/reasoning_present_reward_func/mean": 0.08925781399011612, "rewards/reasoning_present_reward_func/std": 0.030995169654488564, "rewards/xmlcount_reward_func/mean": 0.4189453125, "rewards/xmlcount_reward_func/std": 0.09871959686279297, "step": 17, "step_time": 68.24613171536475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.068359375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 308.01953125, "completions/mean_terminated_length": 293.0523986816406, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.26469149673357606, "epoch": 0.020524515393386546, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04183091223239899, "kl": 0.001374257837596815, "learning_rate": 9.65909090909091e-07, "loss": 6.845366442576051e-06, "num_tokens": 4100110.0, "reward": 1.8974609375, "reward_std": 0.8110225796699524, "rewards/code_complexity_reward/mean": 0.6827148199081421, "rewards/code_complexity_reward/std": 0.3177696466445923, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4169921875, "rewards/code_syntax_reward/std": 0.18622928857803345, "rewards/reasoning_present_reward_func/mean": 0.08828125149011612, "rewards/reasoning_present_reward_func/std": 0.032195813953876495, "rewards/xmlcount_reward_func/mean": 0.42041015625, "rewards/xmlcount_reward_func/std": 0.09584533423185349, "step": 18, "step_time": 58.89718358591199 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.076171875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 302.115234375, "completions/mean_terminated_length": 284.8097229003906, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.2568984942045063, "epoch": 0.021664766248574687, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.05392146483063698, "kl": 0.0013476061631081393, "learning_rate": 1.0227272727272729e-06, "loss": 6.690737791359425e-06, "num_tokens": 4324037.0, "reward": 1.90283203125, "reward_std": 0.8225951194763184, "rewards/code_complexity_reward/mean": 0.67431640625, "rewards/code_complexity_reward/std": 0.3239608108997345, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4140625, "rewards/code_syntax_reward/std": 0.18882036209106445, "rewards/reasoning_present_reward_func/mean": 0.0898437574505806, "rewards/reasoning_present_reward_func/std": 0.030236752703785896, "rewards/xmlcount_reward_func/mean": 0.42578125, "rewards/xmlcount_reward_func/std": 0.09077765047550201, "step": 19, "step_time": 52.216299642808735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 317.689453125, "completions/mean_terminated_length": 297.5883483886719, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2711745009291917, "epoch": 0.02280501710376283, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04525657370686531, "kl": 0.0013900464764446951, "learning_rate": 1.0795454545454546e-06, "loss": 6.922055035829544e-06, "num_tokens": 4556826.0, "reward": 1.8024413585662842, "reward_std": 0.7883129119873047, "rewards/code_complexity_reward/mean": 0.66455078125, "rewards/code_complexity_reward/std": 0.32362639904022217, "rewards/code_execution_reward/mean": 0.22265625, "rewards/code_execution_reward/std": 0.41643625497817993, "rewards/code_syntax_reward/mean": 0.4111328125, "rewards/code_syntax_reward/std": 0.1913314312696457, "rewards/reasoning_present_reward_func/mean": 0.08613280951976776, "rewards/reasoning_present_reward_func/std": 0.034594181925058365, "rewards/xmlcount_reward_func/mean": 0.41796875, "rewards/xmlcount_reward_func/std": 0.10308060795068741, "step": 20, "step_time": 66.17920078709722 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.064453125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 306.763671875, "completions/mean_terminated_length": 292.6242370605469, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2715784488245845, "epoch": 0.02394526795895097, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.0543094202876091, "kl": 0.0015357610591308912, "learning_rate": 1.1363636363636364e-06, "loss": 7.640595867997035e-06, "num_tokens": 4781021.0, "reward": 1.8386718034744263, "reward_std": 0.8271157145500183, "rewards/code_complexity_reward/mean": 0.6594727039337158, "rewards/code_complexity_reward/std": 0.3311060667037964, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.40625, "rewards/code_syntax_reward/std": 0.19534705579280853, "rewards/reasoning_present_reward_func/mean": 0.0908203125, "rewards/reasoning_present_reward_func/std": 0.028902139514684677, "rewards/xmlcount_reward_func/mean": 0.41259765625, "rewards/xmlcount_reward_func/std": 0.09636235982179642, "step": 21, "step_time": 68.61842386238277 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 307.013671875, "completions/mean_terminated_length": 289.6419372558594, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2625524227041751, "epoch": 0.02508551881413911, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04261329397559166, "kl": 0.0013732399638684, "learning_rate": 1.1931818181818183e-06, "loss": 6.849499186500907e-06, "num_tokens": 5007116.0, "reward": 1.9338867664337158, "reward_std": 0.8215371370315552, "rewards/code_complexity_reward/mean": 0.6887695789337158, "rewards/code_complexity_reward/std": 0.3140478730201721, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4208984375, "rewards/code_syntax_reward/std": 0.18264412879943848, "rewards/reasoning_present_reward_func/mean": 0.087890625, "rewards/reasoning_present_reward_func/std": 0.03265552595257759, "rewards/xmlcount_reward_func/mean": 0.423828125, "rewards/xmlcount_reward_func/std": 0.0988985002040863, "step": 22, "step_time": 67.02576862368733 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.080078125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 310.359375, "completions/mean_terminated_length": 292.80682373046875, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.2666364321485162, "epoch": 0.026225769669327253, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04200086370110512, "kl": 0.0013961029708298156, "learning_rate": 1.25e-06, "loss": 6.9414745667018e-06, "num_tokens": 5234388.0, "reward": 1.816503882408142, "reward_std": 0.7934718132019043, "rewards/code_complexity_reward/mean": 0.666308581829071, "rewards/code_complexity_reward/std": 0.32602787017822266, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.41015625, "rewards/code_syntax_reward/std": 0.19215121865272522, "rewards/reasoning_present_reward_func/mean": 0.09062500298023224, "rewards/reasoning_present_reward_func/std": 0.029176566749811172, "rewards/xmlcount_reward_func/mean": 0.4208984375, "rewards/xmlcount_reward_func/std": 0.09334652870893478, "step": 23, "step_time": 63.90472587943077 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 307.158203125, "completions/mean_terminated_length": 289.7987365722656, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.26780034275725484, "epoch": 0.027366020524515394, "frac_reward_zero_std": 0.0, "grad_norm": 0.04071514680981636, "kl": 0.0014103611965765595, "learning_rate": 1.3068181818181819e-06, "loss": 7.013790309429169e-06, "num_tokens": 5461837.0, "reward": 1.8506836891174316, "reward_std": 0.8070063591003418, "rewards/code_complexity_reward/mean": 0.6668945550918579, "rewards/code_complexity_reward/std": 0.32189008593559265, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4130859375, "rewards/code_syntax_reward/std": 0.18966612219810486, "rewards/reasoning_present_reward_func/mean": 0.08906250447034836, "rewards/reasoning_present_reward_func/std": 0.031241435557603836, "rewards/xmlcount_reward_func/mean": 0.421875, "rewards/xmlcount_reward_func/std": 0.0992264598608017, "step": 24, "step_time": 59.83385683875531 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 312.40625, "completions/mean_terminated_length": 292.703857421875, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.25736940396018326, "epoch": 0.028506271379703536, "frac_reward_zero_std": 0.0, "grad_norm": 0.037963517010211945, "kl": 0.001359398028398573, "learning_rate": 1.3636363636363636e-06, "loss": 6.682093953713775e-06, "num_tokens": 5691737.0, "reward": 1.887109398841858, "reward_std": 0.8352934122085571, "rewards/code_complexity_reward/mean": 0.6572265625, "rewards/code_complexity_reward/std": 0.3305343687534332, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4072265625, "rewards/code_syntax_reward/std": 0.19456037878990173, "rewards/reasoning_present_reward_func/mean": 0.09121093899011612, "rewards/reasoning_present_reward_func/std": 0.028341269120573997, "rewards/xmlcount_reward_func/mean": 0.4326171875, "rewards/xmlcount_reward_func/std": 0.08623788505792618, "step": 25, "step_time": 78.29673322103918 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 314.939453125, "completions/mean_terminated_length": 295.48712158203125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.25884595210663974, "epoch": 0.029646522234891677, "frac_reward_zero_std": 0.015625, "grad_norm": 0.045171983540058136, "kl": 0.0013907230759286904, "learning_rate": 1.4204545454545458e-06, "loss": 7.103139068931341e-06, "num_tokens": 5922962.0, "reward": 1.818505883216858, "reward_std": 0.8197525143623352, "rewards/code_complexity_reward/mean": 0.6549804210662842, "rewards/code_complexity_reward/std": 0.33241334557533264, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4033203125, "rewards/code_syntax_reward/std": 0.19765926897525787, "rewards/reasoning_present_reward_func/mean": 0.09199218451976776, "rewards/reasoning_present_reward_func/std": 0.02716795541346073, "rewards/xmlcount_reward_func/mean": 0.424072265625, "rewards/xmlcount_reward_func/std": 0.09482897073030472, "step": 26, "step_time": 52.6985602742061 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.087890625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 311.865234375, "completions/mean_terminated_length": 292.5802917480469, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2646786405239254, "epoch": 0.03078677309007982, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.044380899518728256, "kl": 0.0013902933351346292, "learning_rate": 1.4772727272727275e-06, "loss": 6.926809874130413e-06, "num_tokens": 6151657.0, "reward": 1.8750977516174316, "reward_std": 0.7703660130500793, "rewards/code_complexity_reward/mean": 0.6996093988418579, "rewards/code_complexity_reward/std": 0.30836328864097595, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.4248046875, "rewards/code_syntax_reward/std": 0.17890173196792603, "rewards/reasoning_present_reward_func/mean": 0.08906250447034836, "rewards/reasoning_present_reward_func/std": 0.031241435557603836, "rewards/xmlcount_reward_func/mean": 0.41943359375, "rewards/xmlcount_reward_func/std": 0.10004045814275742, "step": 27, "step_time": 60.193406091071665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 310.259765625, "completions/mean_terminated_length": 290.3454895019531, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.26925629493780434, "epoch": 0.03192702394526796, "frac_reward_zero_std": 0.0, "grad_norm": 0.048720136284828186, "kl": 0.0014241341032175114, "learning_rate": 1.5340909090909093e-06, "loss": 7.123278919607401e-06, "num_tokens": 6379174.0, "reward": 1.844580054283142, "reward_std": 0.8779778480529785, "rewards/code_complexity_reward/mean": 0.6460937261581421, "rewards/code_complexity_reward/std": 0.3418671786785126, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.3994140625, "rewards/code_syntax_reward/std": 0.2006341516971588, "rewards/reasoning_present_reward_func/mean": 0.0869140625, "rewards/reasoning_present_reward_func/std": 0.03375763073563576, "rewards/xmlcount_reward_func/mean": 0.409423828125, "rewards/xmlcount_reward_func/std": 0.10465117543935776, "step": 28, "step_time": 70.71417840756476 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 323.390625, "completions/mean_terminated_length": 302.9783630371094, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.27348981727845967, "epoch": 0.0330672748004561, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.07304804027080536, "kl": 0.001444786345018656, "learning_rate": 1.590909090909091e-06, "loss": 7.230264600366354e-06, "num_tokens": 6616166.0, "reward": 1.8341796398162842, "reward_std": 0.8461081385612488, "rewards/code_complexity_reward/mean": 0.6415039300918579, "rewards/code_complexity_reward/std": 0.33590003848075867, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4013671875, "rewards/code_syntax_reward/std": 0.1991618573665619, "rewards/reasoning_present_reward_func/mean": 0.09062500298023224, "rewards/reasoning_present_reward_func/std": 0.029176566749811172, "rewards/xmlcount_reward_func/mean": 0.42138671875, "rewards/xmlcount_reward_func/std": 0.09728018939495087, "step": 29, "step_time": 68.40165844280273 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.087890625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 315.12109375, "completions/mean_terminated_length": 296.14990234375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2691686558537185, "epoch": 0.03420752565564424, "frac_reward_zero_std": 0.0, "grad_norm": 0.04698478430509567, "kl": 0.0014094157304498367, "learning_rate": 1.6477272727272728e-06, "loss": 6.9784000515937805e-06, "num_tokens": 6846436.0, "reward": 1.8245117664337158, "reward_std": 0.8107052445411682, "rewards/code_complexity_reward/mean": 0.6591796875, "rewards/code_complexity_reward/std": 0.33039695024490356, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4072265625, "rewards/code_syntax_reward/std": 0.19456037878990173, "rewards/reasoning_present_reward_func/mean": 0.09160156548023224, "rewards/reasoning_present_reward_func/std": 0.02776356227695942, "rewards/xmlcount_reward_func/mean": 0.42236328125, "rewards/xmlcount_reward_func/std": 0.09060775488615036, "step": 30, "step_time": 70.09198270086199 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 307.431640625, "completions/mean_terminated_length": 289.15106201171875, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.26217021560296416, "epoch": 0.03534777651083238, "frac_reward_zero_std": 0.0, "grad_norm": 0.04605013132095337, "kl": 0.0013741788370680297, "learning_rate": 1.7045454545454546e-06, "loss": 6.8006920628249645e-06, "num_tokens": 7071833.0, "reward": 1.926416039466858, "reward_std": 0.7785274386405945, "rewards/code_complexity_reward/mean": 0.6900390386581421, "rewards/code_complexity_reward/std": 0.29968738555908203, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4267578125, "rewards/code_syntax_reward/std": 0.17696848511695862, "rewards/reasoning_present_reward_func/mean": 0.09160155802965164, "rewards/reasoning_present_reward_func/std": 0.02776356413960457, "rewards/xmlcount_reward_func/mean": 0.428955078125, "rewards/xmlcount_reward_func/std": 0.09011968225240707, "step": 31, "step_time": 70.92380745895207 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 305.623046875, "completions/mean_terminated_length": 294.5823059082031, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.25967835797928274, "epoch": 0.036488027366020526, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.05137406289577484, "kl": 0.0013644018908962607, "learning_rate": 1.7613636363636365e-06, "loss": 6.8766530603170395e-06, "num_tokens": 7296020.0, "reward": 1.8807129859924316, "reward_std": 0.7874367237091064, "rewards/code_complexity_reward/mean": 0.681347668170929, "rewards/code_complexity_reward/std": 0.3086469769477844, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4208984375, "rewards/code_syntax_reward/std": 0.18264412879943848, "rewards/reasoning_present_reward_func/mean": 0.09023436903953552, "rewards/reasoning_present_reward_func/std": 0.029713962227106094, "rewards/xmlcount_reward_func/mean": 0.424560546875, "rewards/xmlcount_reward_func/std": 0.09521862864494324, "step": 32, "step_time": 60.0196117348969 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.056640625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 305.8046875, "completions/mean_terminated_length": 293.4244384765625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2619610424153507, "epoch": 0.037628278221208664, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04614606127142906, "kl": 0.0013827839011355536, "learning_rate": 1.8181818181818183e-06, "loss": 6.959540769457817e-06, "num_tokens": 7521620.0, "reward": 1.9722657203674316, "reward_std": 0.7848147749900818, "rewards/code_complexity_reward/mean": 0.7048828601837158, "rewards/code_complexity_reward/std": 0.29476115107536316, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.43359375, "rewards/code_syntax_reward/std": 0.16985194385051727, "rewards/reasoning_present_reward_func/mean": 0.08867187798023224, "rewards/reasoning_present_reward_func/std": 0.03172462433576584, "rewards/xmlcount_reward_func/mean": 0.4248046875, "rewards/xmlcount_reward_func/std": 0.09395871311426163, "step": 33, "step_time": 60.20893394108862 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.044921875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 300.896484375, "completions/mean_terminated_length": 290.96728515625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2535063268151134, "epoch": 0.03876852907639681, "frac_reward_zero_std": 0.0, "grad_norm": 0.042864810675382614, "kl": 0.001302112715166004, "learning_rate": 1.8750000000000003e-06, "loss": 6.503767508547753e-06, "num_tokens": 7744639.0, "reward": 1.944580078125, "reward_std": 0.7851381897926331, "rewards/code_complexity_reward/mean": 0.6919922232627869, "rewards/code_complexity_reward/std": 0.2931613624095917, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.431640625, "rewards/code_syntax_reward/std": 0.1719430834054947, "rewards/reasoning_present_reward_func/mean": 0.09121094644069672, "rewards/reasoning_present_reward_func/std": 0.02834126725792885, "rewards/xmlcount_reward_func/mean": 0.428955078125, "rewards/xmlcount_reward_func/std": 0.09345107525587082, "step": 34, "step_time": 59.31065053213388 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.091796875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 313.86328125, "completions/mean_terminated_length": 293.8365783691406, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2630295988637954, "epoch": 0.039908779931584946, "frac_reward_zero_std": 0.0, "grad_norm": 0.050630342215299606, "kl": 0.0013988411828904646, "learning_rate": 1.931818181818182e-06, "loss": 6.921451131347567e-06, "num_tokens": 7974141.0, "reward": 1.864013671875, "reward_std": 0.8355249762535095, "rewards/code_complexity_reward/mean": 0.6629883050918579, "rewards/code_complexity_reward/std": 0.32956451177597046, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4072265625, "rewards/code_syntax_reward/std": 0.19456037878990173, "rewards/reasoning_present_reward_func/mean": 0.09042969346046448, "rewards/reasoning_present_reward_func/std": 0.02944713830947876, "rewards/xmlcount_reward_func/mean": 0.426025390625, "rewards/xmlcount_reward_func/std": 0.09604545682668686, "step": 35, "step_time": 59.07526582572609 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 315.123046875, "completions/mean_terminated_length": 297.52978515625, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2648513284511864, "epoch": 0.04104903078677309, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04637041315436363, "kl": 0.0014365413480845746, "learning_rate": 1.9886363636363638e-06, "loss": 7.1215181378647685e-06, "num_tokens": 8203360.0, "reward": 1.8756835460662842, "reward_std": 0.8134011030197144, "rewards/code_complexity_reward/mean": 0.6654297113418579, "rewards/code_complexity_reward/std": 0.316082626581192, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4150390625, "rewards/code_syntax_reward/std": 0.1879657357931137, "rewards/reasoning_present_reward_func/mean": 0.08964844048023224, "rewards/reasoning_present_reward_func/std": 0.030492907389998436, "rewards/xmlcount_reward_func/mean": 0.42431640625, "rewards/xmlcount_reward_func/std": 0.09356507658958435, "step": 36, "step_time": 60.68934953678399 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 312.943359375, "completions/mean_terminated_length": 296.9852294921875, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2681258029770106, "epoch": 0.04218928164196123, "frac_reward_zero_std": 0.015625, "grad_norm": 0.0426645427942276, "kl": 0.0014391376644198317, "learning_rate": 2.0454545454545457e-06, "loss": 7.2437687776982784e-06, "num_tokens": 8432231.0, "reward": 1.9017088413238525, "reward_std": 0.8329991698265076, "rewards/code_complexity_reward/mean": 0.6637694835662842, "rewards/code_complexity_reward/std": 0.31875622272491455, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.416015625, "rewards/code_syntax_reward/std": 0.1871020793914795, "rewards/reasoning_present_reward_func/mean": 0.08925781399011612, "rewards/reasoning_present_reward_func/std": 0.030995169654488564, "rewards/xmlcount_reward_func/mean": 0.422119140625, "rewards/xmlcount_reward_func/std": 0.09802477806806564, "step": 37, "step_time": 57.294651188887656 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 298.58984375, "completions/mean_terminated_length": 284.3625183105469, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.26026224065572023, "epoch": 0.043329532497149374, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04473437741398811, "kl": 0.001394702072502696, "learning_rate": 2.1022727272727277e-06, "loss": 6.926929927431047e-06, "num_tokens": 8654377.0, "reward": 1.9297852516174316, "reward_std": 0.8160879611968994, "rewards/code_complexity_reward/mean": 0.68115234375, "rewards/code_complexity_reward/std": 0.3141808807849884, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.419921875, "rewards/code_syntax_reward/std": 0.1835547834634781, "rewards/reasoning_present_reward_func/mean": 0.09042969346046448, "rewards/reasoning_present_reward_func/std": 0.02944713830947876, "rewards/xmlcount_reward_func/mean": 0.427734375, "rewards/xmlcount_reward_func/std": 0.08723488450050354, "step": 38, "step_time": 77.67399378120899 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.064453125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 294.888671875, "completions/mean_terminated_length": 279.9311218261719, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.25896762986667454, "epoch": 0.04446978335233751, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.06297129392623901, "kl": 0.001395698607666418, "learning_rate": 2.1590909090909092e-06, "loss": 6.91309105604887e-06, "num_tokens": 8874008.0, "reward": 1.905859351158142, "reward_std": 0.7768024206161499, "rewards/code_complexity_reward/mean": 0.7010742425918579, "rewards/code_complexity_reward/std": 0.30863916873931885, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4248046875, "rewards/code_syntax_reward/std": 0.17890173196792603, "rewards/reasoning_present_reward_func/mean": 0.09199218451976776, "rewards/reasoning_present_reward_func/std": 0.02716795541346073, "rewards/xmlcount_reward_func/mean": 0.43017578125, "rewards/xmlcount_reward_func/std": 0.0915728434920311, "step": 39, "step_time": 62.60964161809534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.068359375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 304.9140625, "completions/mean_terminated_length": 289.71905517578125, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.26427066745236516, "epoch": 0.04561003420752566, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04646266996860504, "kl": 0.0014445839005929884, "learning_rate": 2.2159090909090912e-06, "loss": 7.2355614975094795e-06, "num_tokens": 9098616.0, "reward": 1.92333984375, "reward_std": 0.8296499848365784, "rewards/code_complexity_reward/mean": 0.66943359375, "rewards/code_complexity_reward/std": 0.31170615553855896, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.419921875, "rewards/code_syntax_reward/std": 0.1835547834634781, "rewards/reasoning_present_reward_func/mean": 0.08984375, "rewards/reasoning_present_reward_func/std": 0.030236756429076195, "rewards/xmlcount_reward_func/mean": 0.416015625, "rewards/xmlcount_reward_func/std": 0.10209327936172485, "step": 40, "step_time": 49.358658974058926 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.060546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 299.328125, "completions/mean_terminated_length": 285.6216125488281, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2519068855326623, "epoch": 0.046750285062713795, "frac_reward_zero_std": 0.015625, "grad_norm": 0.048063166439533234, "kl": 0.0013377421337281703, "learning_rate": 2.2727272727272728e-06, "loss": 6.671005394309759e-06, "num_tokens": 9321584.0, "reward": 1.9680663347244263, "reward_std": 0.828618049621582, "rewards/code_complexity_reward/mean": 0.6892578601837158, "rewards/code_complexity_reward/std": 0.308001309633255, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4228515625, "rewards/code_syntax_reward/std": 0.18079319596290588, "rewards/reasoning_present_reward_func/mean": 0.0908203125, "rewards/reasoning_present_reward_func/std": 0.028902139514684677, "rewards/xmlcount_reward_func/mean": 0.43115234375, "rewards/xmlcount_reward_func/std": 0.0955657809972763, "step": 41, "step_time": 70.57339564803988 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 305.65625, "completions/mean_terminated_length": 286.2564392089844, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.26002482790499926, "epoch": 0.04789053591790194, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.04574716463685036, "kl": 0.0014125971538305748, "learning_rate": 2.3295454545454547e-06, "loss": 7.015391020104289e-06, "num_tokens": 9546272.0, "reward": 1.873388648033142, "reward_std": 0.8522294163703918, "rewards/code_complexity_reward/mean": 0.65283203125, "rewards/code_complexity_reward/std": 0.3368537425994873, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4013671875, "rewards/code_syntax_reward/std": 0.1991618573665619, "rewards/reasoning_present_reward_func/mean": 0.09335937350988388, "rewards/reasoning_present_reward_func/std": 0.02492344006896019, "rewards/xmlcount_reward_func/mean": 0.428955078125, "rewards/xmlcount_reward_func/std": 0.09079574793577194, "step": 42, "step_time": 64.08319152984768 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 300.365234375, "completions/mean_terminated_length": 288.12188720703125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2610872208606452, "epoch": 0.04903078677309008, "frac_reward_zero_std": 0.0, "grad_norm": 0.04443160444498062, "kl": 0.0014137213238427648, "learning_rate": 2.3863636363636367e-06, "loss": 7.1089016273617744e-06, "num_tokens": 9769943.0, "reward": 1.9232909679412842, "reward_std": 0.7138127684593201, "rewards/code_complexity_reward/mean": 0.7144531011581421, "rewards/code_complexity_reward/std": 0.276927649974823, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.443359375, "rewards/code_syntax_reward/std": 0.1586231142282486, "rewards/reasoning_present_reward_func/mean": 0.08945313096046448, "rewards/reasoning_present_reward_func/std": 0.03074568696320057, "rewards/xmlcount_reward_func/mean": 0.431884765625, "rewards/xmlcount_reward_func/std": 0.09496993571519852, "step": 43, "step_time": 70.58858884498477 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 308.953125, "completions/mean_terminated_length": 293.5966491699219, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.25991878216154873, "epoch": 0.05017103762827822, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.0428827740252018, "kl": 0.0014178449209794053, "learning_rate": 2.4431818181818182e-06, "loss": 7.012509740889072e-06, "num_tokens": 9996915.0, "reward": 1.945459008216858, "reward_std": 0.7765491008758545, "rewards/code_complexity_reward/mean": 0.685839831829071, "rewards/code_complexity_reward/std": 0.29465171694755554, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4306640625, "rewards/code_syntax_reward/std": 0.17297089099884033, "rewards/reasoning_present_reward_func/mean": 0.09140625596046448, "rewards/reasoning_present_reward_func/std": 0.028054581955075264, "rewards/xmlcount_reward_func/mean": 0.434814453125, "rewards/xmlcount_reward_func/std": 0.08946522325277328, "step": 44, "step_time": 67.66922506969422 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.095703125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 313.123046875, "completions/mean_terminated_length": 292.0755920410156, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.25606408389285207, "epoch": 0.05131128848346636, "frac_reward_zero_std": 0.0, "grad_norm": 0.053362589329481125, "kl": 0.0016728355367376935, "learning_rate": 2.5e-06, "loss": 8.335162419825792e-06, "num_tokens": 10224994.0, "reward": 1.8743164539337158, "reward_std": 0.838887631893158, "rewards/code_complexity_reward/mean": 0.6581054925918579, "rewards/code_complexity_reward/std": 0.32789090275764465, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4091796875, "rewards/code_syntax_reward/std": 0.19296257197856903, "rewards/reasoning_present_reward_func/mean": 0.09023437649011612, "rewards/reasoning_present_reward_func/std": 0.029713962227106094, "rewards/xmlcount_reward_func/mean": 0.42578125, "rewards/xmlcount_reward_func/std": 0.0950557291507721, "step": 45, "step_time": 69.89271171577275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.080078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 308.865234375, "completions/mean_terminated_length": 291.1826171875, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.25648488267324865, "epoch": 0.052451539338654506, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.040566012263298035, "kl": 0.0014128745378911844, "learning_rate": 2.556818181818182e-06, "loss": 7.075461326166987e-06, "num_tokens": 10453405.0, "reward": 1.951318383216858, "reward_std": 0.7846883535385132, "rewards/code_complexity_reward/mean": 0.70263671875, "rewards/code_complexity_reward/std": 0.29850882291793823, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.431640625, "rewards/code_syntax_reward/std": 0.1719430834054947, "rewards/reasoning_present_reward_func/mean": 0.08828125149011612, "rewards/reasoning_present_reward_func/std": 0.032195813953876495, "rewards/xmlcount_reward_func/mean": 0.427978515625, "rewards/xmlcount_reward_func/std": 0.0926990658044815, "step": 46, "step_time": 81.97499120142311 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.072265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 306.966796875, "completions/mean_terminated_length": 290.99578857421875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2626990987919271, "epoch": 0.053591790193842644, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.04101504012942314, "kl": 0.001423976851583575, "learning_rate": 2.6136363636363637e-06, "loss": 7.0507521741092205e-06, "num_tokens": 10680192.0, "reward": 1.899316430091858, "reward_std": 0.7887305617332458, "rewards/code_complexity_reward/mean": 0.679394543170929, "rewards/code_complexity_reward/std": 0.30667752027511597, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4228515625, "rewards/code_syntax_reward/std": 0.18079319596290588, "rewards/reasoning_present_reward_func/mean": 0.08906250447034836, "rewards/reasoning_present_reward_func/std": 0.031241435557603836, "rewards/xmlcount_reward_func/mean": 0.4248046875, "rewards/xmlcount_reward_func/std": 0.1008644700050354, "step": 47, "step_time": 59.50750030763447 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.087890625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 315.025390625, "completions/mean_terminated_length": 296.0449523925781, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.26436753175221384, "epoch": 0.05473204104903079, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04758133366703987, "kl": 0.0014786357805860462, "learning_rate": 2.6704545454545457e-06, "loss": 7.398892194032669e-06, "num_tokens": 10911637.0, "reward": 1.8613770008087158, "reward_std": 0.7762236595153809, "rewards/code_complexity_reward/mean": 0.6824219226837158, "rewards/code_complexity_reward/std": 0.309066504240036, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.421875, "rewards/code_syntax_reward/std": 0.18172365427017212, "rewards/reasoning_present_reward_func/mean": 0.0908203125, "rewards/reasoning_present_reward_func/std": 0.028902139514684677, "rewards/xmlcount_reward_func/mean": 0.429931640625, "rewards/xmlcount_reward_func/std": 0.09451102465391159, "step": 48, "step_time": 63.36412815190852 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 302.009765625, "completions/mean_terminated_length": 288.01043701171875, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.26318921078927815, "epoch": 0.055872291904218926, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03971395641565323, "kl": 0.001538267764772172, "learning_rate": 2.7272727272727272e-06, "loss": 7.639057002961636e-06, "num_tokens": 11134990.0, "reward": 1.9871094226837158, "reward_std": 0.8132598996162415, "rewards/code_complexity_reward/mean": 0.7005859017372131, "rewards/code_complexity_reward/std": 0.30042311549186707, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.427734375, "rewards/code_syntax_reward/std": 0.17598573863506317, "rewards/reasoning_present_reward_func/mean": 0.09023437649011612, "rewards/reasoning_present_reward_func/std": 0.029713962227106094, "rewards/xmlcount_reward_func/mean": 0.4306640625, "rewards/xmlcount_reward_func/std": 0.09194383025169373, "step": 49, "step_time": 59.36232322733849 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 316.505859375, "completions/mean_terminated_length": 297.2081604003906, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2574364731553942, "epoch": 0.05701254275940707, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.04912746697664261, "kl": 0.001423123741915333, "learning_rate": 2.784090909090909e-06, "loss": 7.2255043050972745e-06, "num_tokens": 11364069.0, "reward": 1.835302710533142, "reward_std": 0.8169258832931519, "rewards/code_complexity_reward/mean": 0.6504882574081421, "rewards/code_complexity_reward/std": 0.3258058726787567, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.408203125, "rewards/code_syntax_reward/std": 0.1937655806541443, "rewards/reasoning_present_reward_func/mean": 0.0908203125, "rewards/reasoning_present_reward_func/std": 0.028902137652039528, "rewards/xmlcount_reward_func/mean": 0.433837890625, "rewards/xmlcount_reward_func/std": 0.0953865721821785, "step": 50, "step_time": 57.16936309821904 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 304.486328125, "completions/mean_terminated_length": 293.384765625, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2662681629881263, "epoch": 0.05815279361459521, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.050673939287662506, "kl": 0.0015064856470416998, "learning_rate": 2.8409090909090916e-06, "loss": 7.524446118623018e-06, "num_tokens": 11589418.0, "reward": 1.957177758216858, "reward_std": 0.7792854309082031, "rewards/code_complexity_reward/mean": 0.7017577886581421, "rewards/code_complexity_reward/std": 0.29953786730766296, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4296875, "rewards/code_syntax_reward/std": 0.17398715019226074, "rewards/reasoning_present_reward_func/mean": 0.09062500298023224, "rewards/reasoning_present_reward_func/std": 0.029176566749811172, "rewards/xmlcount_reward_func/mean": 0.436279296875, "rewards/xmlcount_reward_func/std": 0.08881131559610367, "step": 51, "step_time": 88.586783724837 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 298.416015625, "completions/mean_terminated_length": 284.1770935058594, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.2573248485568911, "epoch": 0.059293044469783354, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04544074460864067, "kl": 0.0014798947458984912, "learning_rate": 2.897727272727273e-06, "loss": 7.368699698417913e-06, "num_tokens": 11811711.0, "reward": 2.069287061691284, "reward_std": 0.7578182220458984, "rewards/code_complexity_reward/mean": 0.7408202886581421, "rewards/code_complexity_reward/std": 0.2675424814224243, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4482421875, "rewards/code_syntax_reward/std": 0.15246453881263733, "rewards/reasoning_present_reward_func/mean": 0.08749999850988388, "rewards/reasoning_present_reward_func/std": 0.03310423716902733, "rewards/xmlcount_reward_func/mean": 0.431396484375, "rewards/xmlcount_reward_func/std": 0.09065619111061096, "step": 52, "step_time": 69.55442185048014 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 297.947265625, "completions/mean_terminated_length": 284.6244812011719, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.25414955150336027, "epoch": 0.06043329532497149, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04714526981115341, "kl": 0.0015905083582765656, "learning_rate": 2.954545454545455e-06, "loss": 7.939990609884262e-06, "num_tokens": 12033988.0, "reward": 1.9868165254592896, "reward_std": 0.8070669770240784, "rewards/code_complexity_reward/mean": 0.6998047232627869, "rewards/code_complexity_reward/std": 0.2987743318080902, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4306640625, "rewards/code_syntax_reward/std": 0.17297089099884033, "rewards/reasoning_present_reward_func/mean": 0.08925781399011612, "rewards/reasoning_present_reward_func/std": 0.030995169654488564, "rewards/xmlcount_reward_func/mean": 0.42919921875, "rewards/xmlcount_reward_func/std": 0.09379967302083969, "step": 53, "step_time": 77.99236116930842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.048828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 302.240234375, "completions/mean_terminated_length": 291.4722900390625, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.25510283722542226, "epoch": 0.06157354618015964, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.03555799648165703, "kl": 0.001545114208056475, "learning_rate": 3.0113636363636366e-06, "loss": 7.671129424124956e-06, "num_tokens": 12257015.0, "reward": 2.0833497047424316, "reward_std": 0.7629972100257874, "rewards/code_complexity_reward/mean": 0.7225586175918579, "rewards/code_complexity_reward/std": 0.26695701479911804, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4462890625, "rewards/code_syntax_reward/std": 0.15497584640979767, "rewards/reasoning_present_reward_func/mean": 0.09101562947034836, "rewards/reasoning_present_reward_func/std": 0.02862374484539032, "rewards/xmlcount_reward_func/mean": 0.438720703125, "rewards/xmlcount_reward_func/std": 0.08312025666236877, "step": 54, "step_time": 70.28648695070297 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 299.1484375, "completions/mean_terminated_length": 284.9583435058594, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.2504849573597312, "epoch": 0.06271379703534778, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.035565558820962906, "kl": 0.001539498902275227, "learning_rate": 3.0681818181818186e-06, "loss": 7.672177162021399e-06, "num_tokens": 12478311.0, "reward": 2.0853028297424316, "reward_std": 0.7487791776657104, "rewards/code_complexity_reward/mean": 0.7379883527755737, "rewards/code_complexity_reward/std": 0.26584291458129883, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4501953125, "rewards/code_syntax_reward/std": 0.14988566935062408, "rewards/reasoning_present_reward_func/mean": 0.09511718899011612, "rewards/reasoning_present_reward_func/std": 0.02157193422317505, "rewards/xmlcount_reward_func/mean": 0.448486328125, "rewards/xmlcount_reward_func/std": 0.08239863067865372, "step": 55, "step_time": 70.5015875119716 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 305.052734375, "completions/mean_terminated_length": 291.2562561035156, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2623715801164508, "epoch": 0.06385404789053592, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04246624559164047, "kl": 0.0016605512792011723, "learning_rate": 3.125e-06, "loss": 8.302507922053337e-06, "num_tokens": 12701482.0, "reward": 1.973046898841858, "reward_std": 0.8149088621139526, "rewards/code_complexity_reward/mean": 0.689648449420929, "rewards/code_complexity_reward/std": 0.302064448595047, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4267578125, "rewards/code_syntax_reward/std": 0.17696848511695862, "rewards/reasoning_present_reward_func/mean": 0.09003905951976776, "rewards/reasoning_present_reward_func/std": 0.029977135360240936, "rewards/xmlcount_reward_func/mean": 0.4423828125, "rewards/xmlcount_reward_func/std": 0.08552580326795578, "step": 56, "step_time": 79.50165251363069 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.076171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 309.380859375, "completions/mean_terminated_length": 292.6744079589844, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.2609192051459104, "epoch": 0.06499429874572406, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04399893805384636, "kl": 0.0016437588728877017, "learning_rate": 3.181818181818182e-06, "loss": 8.177041308954358e-06, "num_tokens": 12928953.0, "reward": 1.905029296875, "reward_std": 0.7849880456924438, "rewards/code_complexity_reward/mean": 0.6826171875, "rewards/code_complexity_reward/std": 0.3074110448360443, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.423828125, "rewards/code_syntax_reward/std": 0.17985260486602783, "rewards/reasoning_present_reward_func/mean": 0.0908203125, "rewards/reasoning_present_reward_func/std": 0.028902137652039528, "rewards/xmlcount_reward_func/mean": 0.436279296875, "rewards/xmlcount_reward_func/std": 0.09219000488519669, "step": 57, "step_time": 59.16589973960072 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.064453125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 307.646484375, "completions/mean_terminated_length": 293.56787109375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.26007164525799453, "epoch": 0.0661345496009122, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.049363572150468826, "kl": 0.0015951256555126747, "learning_rate": 3.2386363636363637e-06, "loss": 8.002098184078932e-06, "num_tokens": 13154356.0, "reward": 1.9721192121505737, "reward_std": 0.8170757293701172, "rewards/code_complexity_reward/mean": 0.6896483898162842, "rewards/code_complexity_reward/std": 0.3058149218559265, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4248046875, "rewards/code_syntax_reward/std": 0.17890173196792603, "rewards/reasoning_present_reward_func/mean": 0.0908203125, "rewards/reasoning_present_reward_func/std": 0.028902137652039528, "rewards/xmlcount_reward_func/mean": 0.440673828125, "rewards/xmlcount_reward_func/std": 0.0901302844285965, "step": 58, "step_time": 60.13473348226398 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.064453125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 301.39453125, "completions/mean_terminated_length": 286.88519287109375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.25600972888059914, "epoch": 0.06727480045610035, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.5530682802200317, "kl": 0.10026626086255419, "learning_rate": 3.2954545454545456e-06, "loss": 0.0005022546392865479, "num_tokens": 13377486.0, "reward": 2.0406737327575684, "reward_std": 0.7318780422210693, "rewards/code_complexity_reward/mean": 0.72607421875, "rewards/code_complexity_reward/std": 0.2602851390838623, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.44921875, "rewards/code_syntax_reward/std": 0.15118376910686493, "rewards/reasoning_present_reward_func/mean": 0.09316406399011612, "rewards/reasoning_present_reward_func/std": 0.025260839611291885, "rewards/xmlcount_reward_func/mean": 0.447998046875, "rewards/xmlcount_reward_func/std": 0.08609059453010559, "step": 59, "step_time": 50.77166089322418 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 299.607421875, "completions/mean_terminated_length": 288.2448425292969, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.2537882497999817, "epoch": 0.06841505131128849, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.04201776906847954, "kl": 0.001808079377951799, "learning_rate": 3.352272727272727e-06, "loss": 9.02448664419353e-06, "num_tokens": 13599725.0, "reward": 2.07958984375, "reward_std": 0.7583865523338318, "rewards/code_complexity_reward/mean": 0.72900390625, "rewards/code_complexity_reward/std": 0.2728475332260132, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4453125, "rewards/code_syntax_reward/std": 0.15620718896389008, "rewards/reasoning_present_reward_func/mean": 0.0927734375, "rewards/reasoning_present_reward_func/std": 0.02591804414987564, "rewards/xmlcount_reward_func/mean": 0.451171875, "rewards/xmlcount_reward_func/std": 0.08236709237098694, "step": 60, "step_time": 68.29186902940273 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 316.984375, "completions/mean_terminated_length": 302.2353210449219, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.24933632975444198, "epoch": 0.06955530216647662, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.038240231573581696, "kl": 0.0017303608383372193, "learning_rate": 3.409090909090909e-06, "loss": 8.625853297417052e-06, "num_tokens": 13831025.0, "reward": 1.9245117902755737, "reward_std": 0.7180966138839722, "rewards/code_complexity_reward/mean": 0.7034180164337158, "rewards/code_complexity_reward/std": 0.2815650999546051, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4384765625, "rewards/code_syntax_reward/std": 0.16440613567829132, "rewards/reasoning_present_reward_func/mean": 0.09218750149011612, "rewards/reasoning_present_reward_func/std": 0.026863066479563713, "rewards/xmlcount_reward_func/mean": 0.4462890625, "rewards/xmlcount_reward_func/std": 0.08041233569383621, "step": 61, "step_time": 67.1145661668852 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 299.44921875, "completions/mean_terminated_length": 287.15289306640625, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.2594315765891224, "epoch": 0.07069555302166476, "frac_reward_zero_std": 0.03125, "grad_norm": 0.02918263152241707, "kl": 0.0018398085376247764, "learning_rate": 3.4659090909090915e-06, "loss": 9.138952009379864e-06, "num_tokens": 14054603.0, "reward": 2.011181592941284, "reward_std": 0.7300716042518616, "rewards/code_complexity_reward/mean": 0.7223632335662842, "rewards/code_complexity_reward/std": 0.2662394642829895, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.447265625, "rewards/code_syntax_reward/std": 0.1537284255027771, "rewards/reasoning_present_reward_func/mean": 0.0927734375, "rewards/reasoning_present_reward_func/std": 0.02591804414987564, "rewards/xmlcount_reward_func/mean": 0.451904296875, "rewards/xmlcount_reward_func/std": 0.08036144077777863, "step": 62, "step_time": 68.44585939217359 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 298.376953125, "completions/mean_terminated_length": 282.2206115722656, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.24792794766835868, "epoch": 0.07183580387685291, "frac_reward_zero_std": 0.015625, "grad_norm": 0.05171387270092964, "kl": 0.0018919954400189454, "learning_rate": 3.522727272727273e-06, "loss": 9.423238225281239e-06, "num_tokens": 14276744.0, "reward": 2.0127928256988525, "reward_std": 0.7620010375976562, "rewards/code_complexity_reward/mean": 0.7149413824081421, "rewards/code_complexity_reward/std": 0.28311312198638916, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.439453125, "rewards/code_syntax_reward/std": 0.16327762603759766, "rewards/reasoning_present_reward_func/mean": 0.09375, "rewards/reasoning_present_reward_func/std": 0.02422981895506382, "rewards/xmlcount_reward_func/mean": 0.4521484375, "rewards/xmlcount_reward_func/std": 0.0818258672952652, "step": 63, "step_time": 76.76648098230362 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 323.35546875, "completions/mean_terminated_length": 305.6196594238281, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.26504696859046817, "epoch": 0.07297605473204105, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03926514461636543, "kl": 0.0019349248286744114, "learning_rate": 3.579545454545455e-06, "loss": 9.625509846955538e-06, "num_tokens": 14513262.0, "reward": 1.9025391340255737, "reward_std": 0.7566184997558594, "rewards/code_complexity_reward/mean": 0.6814453601837158, "rewards/code_complexity_reward/std": 0.29758891463279724, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4306640625, "rewards/code_syntax_reward/std": 0.17297089099884033, "rewards/reasoning_present_reward_func/mean": 0.09316406399011612, "rewards/reasoning_present_reward_func/std": 0.025260839611291885, "rewards/xmlcount_reward_func/mean": 0.453125, "rewards/xmlcount_reward_func/std": 0.08013259619474411, "step": 64, "step_time": 75.1137988101691 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 302.73046875, "completions/mean_terminated_length": 287.8451843261719, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.2571107887197286, "epoch": 0.07411630558722919, "frac_reward_zero_std": 0.03125, "grad_norm": 0.04152867570519447, "kl": 0.0021813245402881876, "learning_rate": 3.6363636363636366e-06, "loss": 1.0861494956770912e-05, "num_tokens": 14738632.0, "reward": 2.001171827316284, "reward_std": 0.7908423542976379, "rewards/code_complexity_reward/mean": 0.7109375, "rewards/code_complexity_reward/std": 0.2962873578071594, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.43359375, "rewards/code_syntax_reward/std": 0.16985194385051727, "rewards/reasoning_present_reward_func/mean": 0.09199218451976776, "rewards/reasoning_present_reward_func/std": 0.02716795541346073, "rewards/xmlcount_reward_func/mean": 0.4521484375, "rewards/xmlcount_reward_func/std": 0.08031721413135529, "step": 65, "step_time": 63.35976053215563 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 315.015625, "completions/mean_terminated_length": 298.322021484375, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.25123084150254726, "epoch": 0.07525655644241733, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.037399083375930786, "kl": 0.002097144426443265, "learning_rate": 3.6931818181818186e-06, "loss": 1.0420684702694416e-05, "num_tokens": 14969636.0, "reward": 1.9875975847244263, "reward_std": 0.744015634059906, "rewards/code_complexity_reward/mean": 0.7184571027755737, "rewards/code_complexity_reward/std": 0.2795127332210541, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4404296875, "rewards/code_syntax_reward/std": 0.16213536262512207, "rewards/reasoning_present_reward_func/mean": 0.09335937350988388, "rewards/reasoning_present_reward_func/std": 0.02492344006896019, "rewards/xmlcount_reward_func/mean": 0.4599609375, "rewards/xmlcount_reward_func/std": 0.07240891456604004, "step": 66, "step_time": 80.84491949994117 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.072265625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 298.755859375, "completions/mean_terminated_length": 282.145263671875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.26207552826963365, "epoch": 0.07639680729760548, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.039158888161182404, "kl": 0.0023165040875028353, "learning_rate": 3.7500000000000005e-06, "loss": 1.1562646250240505e-05, "num_tokens": 15191647.0, "reward": 2.005126953125, "reward_std": 0.736138641834259, "rewards/code_complexity_reward/mean": 0.73583984375, "rewards/code_complexity_reward/std": 0.27675729990005493, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4462890625, "rewards/code_syntax_reward/std": 0.15497584640979767, "rewards/reasoning_present_reward_func/mean": 0.0908203125, "rewards/reasoning_present_reward_func/std": 0.028902139514684677, "rewards/xmlcount_reward_func/mean": 0.452880859375, "rewards/xmlcount_reward_func/std": 0.0860656201839447, "step": 67, "step_time": 60.49427730683237 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 307.76953125, "completions/mean_terminated_length": 287.60943603515625, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.2506056863348931, "epoch": 0.07753705815279362, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.03670023754239082, "kl": 0.0022464333869720576, "learning_rate": 3.806818181818182e-06, "loss": 1.1289899703115225e-05, "num_tokens": 15419729.0, "reward": 2.0394043922424316, "reward_std": 0.7197580933570862, "rewards/code_complexity_reward/mean": 0.7271484136581421, "rewards/code_complexity_reward/std": 0.26104801893234253, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.451171875, "rewards/code_syntax_reward/std": 0.14856980741024017, "rewards/reasoning_present_reward_func/mean": 0.0947265625, "rewards/reasoning_present_reward_func/std": 0.022372130304574966, "rewards/xmlcount_reward_func/mean": 0.463623046875, "rewards/xmlcount_reward_func/std": 0.07159565389156342, "step": 68, "step_time": 67.17821385152638 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.087890625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 316.34375, "completions/mean_terminated_length": 297.4903564453125, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.25531757064163685, "epoch": 0.07867730900798175, "frac_reward_zero_std": 0.046875, "grad_norm": 0.03260109946131706, "kl": 0.0024271969650726533, "learning_rate": 3.863636363636364e-06, "loss": 1.2142360901634675e-05, "num_tokens": 15649865.0, "reward": 2.0330567359924316, "reward_std": 0.7489521503448486, "rewards/code_complexity_reward/mean": 0.7197265625, "rewards/code_complexity_reward/std": 0.27445289492607117, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4423828125, "rewards/code_syntax_reward/std": 0.1598084270954132, "rewards/reasoning_present_reward_func/mean": 0.09433594346046448, "rewards/reasoning_present_reward_func/std": 0.023138070479035378, "rewards/xmlcount_reward_func/mean": 0.460205078125, "rewards/xmlcount_reward_func/std": 0.07482600212097168, "step": 69, "step_time": 58.52154109440744 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 308.091796875, "completions/mean_terminated_length": 292.6701965332031, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.2556555806659162, "epoch": 0.07981755986316989, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.03537225350737572, "kl": 0.002371629132539965, "learning_rate": 3.9204545454545456e-06, "loss": 1.1831172741949558e-05, "num_tokens": 15877752.0, "reward": 2.0426268577575684, "reward_std": 0.6998047232627869, "rewards/code_complexity_reward/mean": 0.72265625, "rewards/code_complexity_reward/std": 0.24840858578681946, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4580078125, "rewards/code_syntax_reward/std": 0.13881781697273254, "rewards/reasoning_present_reward_func/mean": 0.09316405653953552, "rewards/reasoning_present_reward_func/std": 0.025260839611291885, "rewards/xmlcount_reward_func/mean": 0.464111328125, "rewards/xmlcount_reward_func/std": 0.07393960654735565, "step": 70, "step_time": 69.32909472566098 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.060546875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 305.255859375, "completions/mean_terminated_length": 291.931396484375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24091640580445528, "epoch": 0.08095781071835804, "frac_reward_zero_std": 0.03125, "grad_norm": 0.042964108288288116, "kl": 0.002792473746012547, "learning_rate": 3.9772727272727275e-06, "loss": 1.3969081919640303e-05, "num_tokens": 16103335.0, "reward": 2.069628953933716, "reward_std": 0.7515430450439453, "rewards/code_complexity_reward/mean": 0.7213866710662842, "rewards/code_complexity_reward/std": 0.2634570598602295, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4482421875, "rewards/code_syntax_reward/std": 0.15246453881263733, "rewards/reasoning_present_reward_func/mean": 0.09335937350988388, "rewards/reasoning_present_reward_func/std": 0.02492344006896019, "rewards/xmlcount_reward_func/mean": 0.46484375, "rewards/xmlcount_reward_func/std": 0.07283654063940048, "step": 71, "step_time": 69.20507015939802 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.095703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 312.48046875, "completions/mean_terminated_length": 291.364990234375, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.25656614638864994, "epoch": 0.08209806157354618, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.030906936153769493, "kl": 0.002625626349981758, "learning_rate": 4.0340909090909095e-06, "loss": 1.3172542821848765e-05, "num_tokens": 16330925.0, "reward": 2.027050733566284, "reward_std": 0.7649253606796265, "rewards/code_complexity_reward/mean": 0.7069336175918579, "rewards/code_complexity_reward/std": 0.28241994976997375, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4404296875, "rewards/code_syntax_reward/std": 0.16213536262512207, "rewards/reasoning_present_reward_func/mean": 0.09453125298023224, "rewards/reasoning_present_reward_func/std": 0.02275916188955307, "rewards/xmlcount_reward_func/mean": 0.4609375, "rewards/xmlcount_reward_func/std": 0.076621413230896, "step": 72, "step_time": 58.241065255366266 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 287.87109375, "completions/mean_terminated_length": 281.5702819824219, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.24900115840137005, "epoch": 0.08323831242873432, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.032976601272821426, "kl": 0.002775821534669376, "learning_rate": 4.0909090909090915e-06, "loss": 1.3880722690373659e-05, "num_tokens": 16545851.0, "reward": 2.1424803733825684, "reward_std": 0.6464732885360718, "rewards/code_complexity_reward/mean": 0.778124988079071, "rewards/code_complexity_reward/std": 0.20507267117500305, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4736328125, "rewards/code_syntax_reward/std": 0.11186064779758453, "rewards/reasoning_present_reward_func/mean": 0.09531250596046448, "rewards/reasoning_present_reward_func/std": 0.021157780662178993, "rewards/xmlcount_reward_func/mean": 0.47119140625, "rewards/xmlcount_reward_func/std": 0.07012113183736801, "step": 73, "step_time": 62.14113441668451 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.052734375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 299.302734375, "completions/mean_terminated_length": 287.46185302734375, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.2537171063013375, "epoch": 0.08437856328392246, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03467102348804474, "kl": 0.003153982455842197, "learning_rate": 4.1477272727272734e-06, "loss": 1.5684752725064754e-05, "num_tokens": 16766666.0, "reward": 2.089599609375, "reward_std": 0.7370011210441589, "rewards/code_complexity_reward/mean": 0.7318359613418579, "rewards/code_complexity_reward/std": 0.2567334771156311, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4521484375, "rewards/code_syntax_reward/std": 0.1472356915473938, "rewards/reasoning_present_reward_func/mean": 0.09628906846046448, "rewards/reasoning_present_reward_func/std": 0.018921468406915665, "rewards/xmlcount_reward_func/mean": 0.471435546875, "rewards/xmlcount_reward_func/std": 0.06502099335193634, "step": 74, "step_time": 56.603501352481544 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 311.09375, "completions/mean_terminated_length": 295.899169921875, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.24706104211509228, "epoch": 0.08551881413911061, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.029938098043203354, "kl": 0.0032576788908045273, "learning_rate": 4.204545454545455e-06, "loss": 1.625613367650658e-05, "num_tokens": 16996118.0, "reward": 2.0149903297424316, "reward_std": 0.6844449639320374, "rewards/code_complexity_reward/mean": 0.7201171517372131, "rewards/code_complexity_reward/std": 0.24779967963695526, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4560546875, "rewards/code_syntax_reward/std": 0.14170633256435394, "rewards/reasoning_present_reward_func/mean": 0.09492187201976776, "rewards/reasoning_present_reward_func/std": 0.021976543590426445, "rewards/xmlcount_reward_func/mean": 0.466552734375, "rewards/xmlcount_reward_func/std": 0.0734306201338768, "step": 75, "step_time": 79.75564846768975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 293.1328125, "completions/mean_terminated_length": 280.4710693359375, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.2561690118163824, "epoch": 0.08665906499429875, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03523845970630646, "kl": 0.003292939094535541, "learning_rate": 4.2613636363636365e-06, "loss": 1.6272300854325294e-05, "num_tokens": 17213694.0, "reward": 2.1136231422424316, "reward_std": 0.6996862888336182, "rewards/code_complexity_reward/mean": 0.7544921636581421, "rewards/code_complexity_reward/std": 0.24102142453193665, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4619140625, "rewards/code_syntax_reward/std": 0.13276617228984833, "rewards/reasoning_present_reward_func/mean": 0.095703125, "rewards/reasoning_present_reward_func/std": 0.02029850147664547, "rewards/xmlcount_reward_func/mean": 0.471435546875, "rewards/xmlcount_reward_func/std": 0.06868017464876175, "step": 76, "step_time": 67.36736439727247 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 287.5234375, "completions/mean_terminated_length": 274.53717041015625, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.24673541076481342, "epoch": 0.08779931584948689, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.03363867104053497, "kl": 0.0033493212613393553, "learning_rate": 4.3181818181818185e-06, "loss": 1.6851106920512393e-05, "num_tokens": 17428646.0, "reward": 2.152587890625, "reward_std": 0.7056934833526611, "rewards/code_complexity_reward/mean": 0.75537109375, "rewards/code_complexity_reward/std": 0.23317232728004456, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4638671875, "rewards/code_syntax_reward/std": 0.1295902281999588, "rewards/reasoning_present_reward_func/mean": 0.0947265625, "rewards/reasoning_present_reward_func/std": 0.022372130304574966, "rewards/xmlcount_reward_func/mean": 0.473388671875, "rewards/xmlcount_reward_func/std": 0.06676837056875229, "step": 77, "step_time": 66.74031392205507 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 293.751953125, "completions/mean_terminated_length": 283.0184326171875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.2530967155471444, "epoch": 0.08893956670467502, "frac_reward_zero_std": 0.078125, "grad_norm": 0.029970921576023102, "kl": 0.0033483381048426963, "learning_rate": 4.3750000000000005e-06, "loss": 1.662288559600711e-05, "num_tokens": 17647583.0, "reward": 2.0857911109924316, "reward_std": 0.6454166173934937, "rewards/code_complexity_reward/mean": 0.7596679925918579, "rewards/code_complexity_reward/std": 0.22441332042217255, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4677734375, "rewards/code_syntax_reward/std": 0.12289927154779434, "rewards/reasoning_present_reward_func/mean": 0.09589843451976776, "rewards/reasoning_present_reward_func/std": 0.019852032884955406, "rewards/xmlcount_reward_func/mean": 0.475341796875, "rewards/xmlcount_reward_func/std": 0.062334951013326645, "step": 78, "step_time": 68.29603808838874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 299.63671875, "completions/mean_terminated_length": 286.4190979003906, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.24491690145805478, "epoch": 0.09007981755986318, "frac_reward_zero_std": 0.046875, "grad_norm": 0.03652897849678993, "kl": 0.003918045173122664, "learning_rate": 4.4318181818181824e-06, "loss": 1.954213439603336e-05, "num_tokens": 17871677.0, "reward": 2.132519483566284, "reward_std": 0.6908275485038757, "rewards/code_complexity_reward/mean": 0.7471679449081421, "rewards/code_complexity_reward/std": 0.2377050817012787, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4619140625, "rewards/code_syntax_reward/std": 0.13276617228984833, "rewards/reasoning_present_reward_func/mean": 0.09531250596046448, "rewards/reasoning_present_reward_func/std": 0.021157780662178993, "rewards/xmlcount_reward_func/mean": 0.478515625, "rewards/xmlcount_reward_func/std": 0.057698383927345276, "step": 79, "step_time": 71.50319295655936 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.056640625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 290.23046875, "completions/mean_terminated_length": 276.9151306152344, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.2392619384918362, "epoch": 0.09122006841505131, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.0306667722761631, "kl": 0.0037698509877373, "learning_rate": 4.4886363636363636e-06, "loss": 1.879059709608555e-05, "num_tokens": 18089131.0, "reward": 2.172656297683716, "reward_std": 0.691567063331604, "rewards/code_complexity_reward/mean": 0.7625000476837158, "rewards/code_complexity_reward/std": 0.22319069504737854, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.46875, "rewards/code_syntax_reward/std": 0.12114909291267395, "rewards/reasoning_present_reward_func/mean": 0.0966796875, "rewards/reasoning_present_reward_func/std": 0.01793418452143669, "rewards/xmlcount_reward_func/mean": 0.4794921875, "rewards/xmlcount_reward_func/std": 0.06062973290681839, "step": 80, "step_time": 60.3383459970355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 286.900390625, "completions/mean_terminated_length": 277.75, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.24593450408428907, "epoch": 0.09236031927023945, "frac_reward_zero_std": 0.046875, "grad_norm": 0.02557935006916523, "kl": 0.003745507125131553, "learning_rate": 4.5454545454545455e-06, "loss": 1.8747028661891818e-05, "num_tokens": 18305976.0, "reward": 2.1293458938598633, "reward_std": 0.6304735541343689, "rewards/code_complexity_reward/mean": 0.7752929925918579, "rewards/code_complexity_reward/std": 0.21183179318904877, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4736328125, "rewards/code_syntax_reward/std": 0.11186064779758453, "rewards/reasoning_present_reward_func/mean": 0.09355469048023224, "rewards/reasoning_present_reward_func/std": 0.024579854682087898, "rewards/xmlcount_reward_func/mean": 0.476318359375, "rewards/xmlcount_reward_func/std": 0.06463407725095749, "step": 81, "step_time": 66.92190232872963 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.048828125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 285.6953125, "completions/mean_terminated_length": 274.0780334472656, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.23955502128228545, "epoch": 0.09350057012542759, "frac_reward_zero_std": 0.046875, "grad_norm": 0.037397708743810654, "kl": 0.003863460920911166, "learning_rate": 4.6022727272727275e-06, "loss": 1.936085755005479e-05, "num_tokens": 18520652.0, "reward": 2.2155275344848633, "reward_std": 0.6944394707679749, "rewards/code_complexity_reward/mean": 0.7606445550918579, "rewards/code_complexity_reward/std": 0.21983131766319275, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.4697265625, "rewards/code_syntax_reward/std": 0.11936526000499725, "rewards/reasoning_present_reward_func/mean": 0.09648437798023224, "rewards/reasoning_present_reward_func/std": 0.01843547262251377, "rewards/xmlcount_reward_func/mean": 0.48046875, "rewards/xmlcount_reward_func/std": 0.061947163194417953, "step": 82, "step_time": 68.02021906059235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 293.0546875, "completions/mean_terminated_length": 281.341552734375, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.24364518793299794, "epoch": 0.09464082098061574, "frac_reward_zero_std": 0.03125, "grad_norm": 0.0298225749284029, "kl": 0.00468868951247714, "learning_rate": 4.6590909090909095e-06, "loss": 2.329860581085086e-05, "num_tokens": 18739036.0, "reward": 2.1504883766174316, "reward_std": 0.6761159896850586, "rewards/code_complexity_reward/mean": 0.7557617425918579, "rewards/code_complexity_reward/std": 0.2230394035577774, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4677734375, "rewards/code_syntax_reward/std": 0.12289927154779434, "rewards/reasoning_present_reward_func/mean": 0.09687499701976776, "rewards/reasoning_present_reward_func/std": 0.01741628162562847, "rewards/xmlcount_reward_func/mean": 0.478515625, "rewards/xmlcount_reward_func/std": 0.05978061258792877, "step": 83, "step_time": 60.728172823786736 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 292.1796875, "completions/mean_terminated_length": 279.4627990722656, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.243406951893121, "epoch": 0.09578107183580388, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.0282623078674078, "kl": 0.004087860233994434, "learning_rate": 4.715909090909091e-06, "loss": 2.0426377886906266e-05, "num_tokens": 18956544.0, "reward": 2.107128858566284, "reward_std": 0.6719666123390198, "rewards/code_complexity_reward/mean": 0.757617175579071, "rewards/code_complexity_reward/std": 0.2307576835155487, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.46484375, "rewards/code_syntax_reward/std": 0.12796148657798767, "rewards/reasoning_present_reward_func/mean": 0.09609374403953552, "rewards/reasoning_present_reward_func/std": 0.01939331740140915, "rewards/xmlcount_reward_func/mean": 0.47998046875, "rewards/xmlcount_reward_func/std": 0.053858909755945206, "step": 84, "step_time": 98.707549857907 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.044921875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 290.341796875, "completions/mean_terminated_length": 279.9161376953125, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.24405558337457478, "epoch": 0.09692132269099202, "frac_reward_zero_std": 0.046875, "grad_norm": 0.032620497047901154, "kl": 0.004997704785637325, "learning_rate": 4.772727272727273e-06, "loss": 2.4892207875382155e-05, "num_tokens": 19174655.0, "reward": 2.1285643577575684, "reward_std": 0.6515979766845703, "rewards/code_complexity_reward/mean": 0.763671875, "rewards/code_complexity_reward/std": 0.21960780024528503, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4697265625, "rewards/code_syntax_reward/std": 0.11936526000499725, "rewards/reasoning_present_reward_func/mean": 0.09609374403953552, "rewards/reasoning_present_reward_func/std": 0.01939331740140915, "rewards/xmlcount_reward_func/mean": 0.482666015625, "rewards/xmlcount_reward_func/std": 0.05394035205245018, "step": 85, "step_time": 58.03035189025104 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.033203125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 289.650390625, "completions/mean_terminated_length": 282.01416015625, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.24560853326693177, "epoch": 0.09806157354618016, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.025918900966644287, "kl": 0.00437140250869561, "learning_rate": 4.829545454545455e-06, "loss": 2.1847838070243597e-05, "num_tokens": 19392484.0, "reward": 2.2373533248901367, "reward_std": 0.6441631317138672, "rewards/code_complexity_reward/mean": 0.775585949420929, "rewards/code_complexity_reward/std": 0.19234275817871094, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09726562350988388, "rewards/reasoning_present_reward_func/std": 0.016324250027537346, "rewards/xmlcount_reward_func/mean": 0.487548828125, "rewards/xmlcount_reward_func/std": 0.04489602521061897, "step": 86, "step_time": 59.053016224876046 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 301.5625, "completions/mean_terminated_length": 283.72882080078125, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.241712830029428, "epoch": 0.09920182440136831, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.02751206047832966, "kl": 0.004633384716726141, "learning_rate": 4.8863636363636365e-06, "loss": 2.3124857762013562e-05, "num_tokens": 19616468.0, "reward": 2.125244140625, "reward_std": 0.7362744212150574, "rewards/code_complexity_reward/mean": 0.7444336414337158, "rewards/code_complexity_reward/std": 0.2535525858402252, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4560546875, "rewards/code_syntax_reward/std": 0.14170633256435394, "rewards/reasoning_present_reward_func/mean": 0.09492187947034836, "rewards/reasoning_present_reward_func/std": 0.021976543590426445, "rewards/xmlcount_reward_func/mean": 0.478271484375, "rewards/xmlcount_reward_func/std": 0.060455627739429474, "step": 87, "step_time": 68.1332252593711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.041015625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 291.982421875, "completions/mean_terminated_length": 282.57232666015625, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.23736231727525592, "epoch": 0.10034207525655645, "frac_reward_zero_std": 0.0546875, "grad_norm": 0.027488717809319496, "kl": 0.004462690292712068, "learning_rate": 4.9431818181818184e-06, "loss": 2.237400258309208e-05, "num_tokens": 19835755.0, "reward": 2.1534180641174316, "reward_std": 0.6255471706390381, "rewards/code_complexity_reward/mean": 0.7646484375, "rewards/code_complexity_reward/std": 0.19201341271400452, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.09628906100988388, "rewards/reasoning_present_reward_func/std": 0.018921468406915665, "rewards/xmlcount_reward_func/mean": 0.48291015625, "rewards/xmlcount_reward_func/std": 0.05810887739062309, "step": 88, "step_time": 70.03870182018727 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 291.525390625, "completions/mean_terminated_length": 279.7304382324219, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.23949270858429372, "epoch": 0.10148232611174458, "frac_reward_zero_std": 0.046875, "grad_norm": 0.028046511113643646, "kl": 0.0038938892157602822, "learning_rate": 5e-06, "loss": 1.944121868291404e-05, "num_tokens": 20054600.0, "reward": 2.1038575172424316, "reward_std": 0.6011189222335815, "rewards/code_complexity_reward/mean": 0.7706055045127869, "rewards/code_complexity_reward/std": 0.20078180730342865, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4755859375, "rewards/code_syntax_reward/std": 0.10785966366529465, "rewards/reasoning_present_reward_func/mean": 0.0966796875, "rewards/reasoning_present_reward_func/std": 0.017934182658791542, "rewards/xmlcount_reward_func/mean": 0.485595703125, "rewards/xmlcount_reward_func/std": 0.04430686682462692, "step": 89, "step_time": 62.41021960321814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 285.771484375, "completions/mean_terminated_length": 278.4737854003906, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.24651256971992552, "epoch": 0.10262257696693272, "frac_reward_zero_std": 0.0625, "grad_norm": 0.03142227604985237, "kl": 0.006389342117472552, "learning_rate": 4.999980182212003e-06, "loss": 3.191456926288083e-05, "num_tokens": 20270815.0, "reward": 2.121337890625, "reward_std": 0.6218167543411255, "rewards/code_complexity_reward/mean": 0.7676757574081421, "rewards/code_complexity_reward/std": 0.20684251189231873, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4755859375, "rewards/code_syntax_reward/std": 0.10785966366529465, "rewards/reasoning_present_reward_func/mean": 0.09707030653953552, "rewards/reasoning_present_reward_func/std": 0.016880230978131294, "rewards/xmlcount_reward_func/mean": 0.488037109375, "rewards/xmlcount_reward_func/std": 0.04364960640668869, "step": 90, "step_time": 62.195642608217895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 290.705078125, "completions/mean_terminated_length": 279.82171630859375, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.2332771390210837, "epoch": 0.10376282782212087, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.0264880508184433, "kl": 0.006602885012398474, "learning_rate": 4.999920729162207e-06, "loss": 3.2983112760121e-05, "num_tokens": 20488988.0, "reward": 2.191699266433716, "reward_std": 0.6591775417327881, "rewards/code_complexity_reward/mean": 0.7628905773162842, "rewards/code_complexity_reward/std": 0.21030181646347046, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4736328125, "rewards/code_syntax_reward/std": 0.11186064779758453, "rewards/reasoning_present_reward_func/mean": 0.09726563096046448, "rewards/reasoning_present_reward_func/std": 0.016324250027537346, "rewards/xmlcount_reward_func/mean": 0.48486328125, "rewards/xmlcount_reward_func/std": 0.052017804235219955, "step": 91, "step_time": 58.32539480086416 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 278.859375, "completions/mean_terminated_length": 271.3387145996094, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.2343874415382743, "epoch": 0.10490307867730901, "frac_reward_zero_std": 0.03125, "grad_norm": 0.026998214423656464, "kl": 0.006127606400696095, "learning_rate": 4.999821641793195e-06, "loss": 3.058073343709111e-05, "num_tokens": 20699200.0, "reward": 2.24853515625, "reward_std": 0.6134322881698608, "rewards/code_complexity_reward/mean": 0.79052734375, "rewards/code_complexity_reward/std": 0.17651528120040894, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.09765625, "rewards/reasoning_present_reward_func/std": 0.015143636614084244, "rewards/xmlcount_reward_func/mean": 0.490234375, "rewards/xmlcount_reward_func/std": 0.03704262897372246, "step": 92, "step_time": 70.8439681455493 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 282.603515625, "completions/mean_terminated_length": 274.24493408203125, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.2387225478887558, "epoch": 0.10604332953249715, "frac_reward_zero_std": 0.046875, "grad_norm": 0.030447738245129585, "kl": 0.007436025975039229, "learning_rate": 4.999682921675919e-06, "loss": 3.7202087696641684e-05, "num_tokens": 20911305.0, "reward": 2.218554735183716, "reward_std": 0.6384223699569702, "rewards/code_complexity_reward/mean": 0.7759765982627869, "rewards/code_complexity_reward/std": 0.19360879063606262, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.09785155951976776, "rewards/reasoning_present_reward_func/std": 0.014513419941067696, "rewards/xmlcount_reward_func/mean": 0.48828125, "rewards/xmlcount_reward_func/std": 0.04475279897451401, "step": 93, "step_time": 69.87169802933931 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 276.548828125, "completions/mean_terminated_length": 269.9297180175781, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.23696576547808945, "epoch": 0.10718358038768529, "frac_reward_zero_std": 0.0625, "grad_norm": 0.02845817059278488, "kl": 0.0047280235994549, "learning_rate": 4.999504571009682e-06, "loss": 2.3525848519057035e-05, "num_tokens": 21122374.0, "reward": 2.248242139816284, "reward_std": 0.6138839721679688, "rewards/code_complexity_reward/mean": 0.7939453125, "rewards/code_complexity_reward/std": 0.16886422038078308, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09785155951976776, "rewards/reasoning_present_reward_func/std": 0.014513419941067696, "rewards/xmlcount_reward_func/mean": 0.490234375, "rewards/xmlcount_reward_func/std": 0.04784845933318138, "step": 94, "step_time": 74.32432336732745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 281.23046875, "completions/mean_terminated_length": 271.8495788574219, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.24174663075245917, "epoch": 0.10832383124287344, "frac_reward_zero_std": 0.0546875, "grad_norm": 0.024731500074267387, "kl": 0.006605824892176315, "learning_rate": 4.999286592622096e-06, "loss": 3.286229184595868e-05, "num_tokens": 21336668.0, "reward": 2.1590332984924316, "reward_std": 0.621860146522522, "rewards/code_complexity_reward/mean": 0.781542956829071, "rewards/code_complexity_reward/std": 0.2012343555688858, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.09746094048023224, "rewards/reasoning_present_reward_func/std": 0.015746228396892548, "rewards/xmlcount_reward_func/mean": 0.488037109375, "rewards/xmlcount_reward_func/std": 0.045702867209911346, "step": 95, "step_time": 68.71504017151892 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.056640625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 286.140625, "completions/mean_terminated_length": 272.5797119140625, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.2349567641504109, "epoch": 0.10946408209806158, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.02726534940302372, "kl": 0.006675793258182239, "learning_rate": 4.99902898996904e-06, "loss": 3.3313423045910895e-05, "num_tokens": 21553832.0, "reward": 2.154980421066284, "reward_std": 0.6829071044921875, "rewards/code_complexity_reward/mean": 0.7587890625, "rewards/code_complexity_reward/std": 0.22428598999977112, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4677734375, "rewards/code_syntax_reward/std": 0.12289927154779434, "rewards/reasoning_present_reward_func/mean": 0.09785156697034836, "rewards/reasoning_present_reward_func/std": 0.01451342087239027, "rewards/xmlcount_reward_func/mean": 0.48486328125, "rewards/xmlcount_reward_func/std": 0.0537523552775383, "step": 96, "step_time": 63.98590052127838 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.048828125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 280.892578125, "completions/mean_terminated_length": 269.02874755859375, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.23751316708512604, "epoch": 0.11060433295324971, "frac_reward_zero_std": 0.046875, "grad_norm": 0.02768808975815773, "kl": 0.005592901376076043, "learning_rate": 4.998731767134606e-06, "loss": 2.7817150112241507e-05, "num_tokens": 21764437.0, "reward": 2.128368854522705, "reward_std": 0.6273660659790039, "rewards/code_complexity_reward/mean": 0.7699218988418579, "rewards/code_complexity_reward/std": 0.2069808840751648, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4755859375, "rewards/code_syntax_reward/std": 0.10785966366529465, "rewards/reasoning_present_reward_func/mean": 0.09843750298023224, "rewards/reasoning_present_reward_func/std": 0.012414088472723961, "rewards/xmlcount_reward_func/mean": 0.487548828125, "rewards/xmlcount_reward_func/std": 0.04818117991089821, "step": 97, "step_time": 59.96561582572758 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.041015625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 287.21875, "completions/mean_terminated_length": 277.6048889160156, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.23948721284978092, "epoch": 0.11174458380843785, "frac_reward_zero_std": 0.03125, "grad_norm": 0.027711493894457817, "kl": 0.0051925360785389785, "learning_rate": 4.998394928831034e-06, "loss": 2.591370139271021e-05, "num_tokens": 21979181.0, "reward": 2.1728515625, "reward_std": 0.6210906505584717, "rewards/code_complexity_reward/mean": 0.772656261920929, "rewards/code_complexity_reward/std": 0.19683000445365906, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.4921875, "rewards/xmlcount_reward_func/std": 0.034087203443050385, "step": 98, "step_time": 67.17210439220071 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.029296875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 280.88671875, "completions/mean_terminated_length": 273.9114685058594, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.24290773016400635, "epoch": 0.11288483466362599, "frac_reward_zero_std": 0.0625, "grad_norm": 0.029658164829015732, "kl": 0.00574053919990547, "learning_rate": 4.998018480398635e-06, "loss": 2.8637121431529522e-05, "num_tokens": 22191931.0, "reward": 2.2022461891174316, "reward_std": 0.5835312008857727, "rewards/code_complexity_reward/mean": 0.7875000238418579, "rewards/code_complexity_reward/std": 0.17368459701538086, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09882812201976776, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.49267578125, "rewards/xmlcount_reward_func/std": 0.030409282073378563, "step": 99, "step_time": 61.06361153256148 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 272.6015625, "completions/mean_terminated_length": 266.85601806640625, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.23527400917373598, "epoch": 0.11402508551881414, "frac_reward_zero_std": 0.046875, "grad_norm": 0.027682935819029808, "kl": 0.004265145671524806, "learning_rate": 4.99760242780571e-06, "loss": 2.1238054614514112e-05, "num_tokens": 22399811.0, "reward": 2.190478563308716, "reward_std": 0.5953410267829895, "rewards/code_complexity_reward/mean": 0.785449206829071, "rewards/code_complexity_reward/std": 0.175640270113945, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09765625, "rewards/reasoning_present_reward_func/std": 0.015143636614084244, "rewards/xmlcount_reward_func/mean": 0.492919921875, "rewards/xmlcount_reward_func/std": 0.028922587633132935, "step": 100, "step_time": 60.68683035578579 }, { "epoch": 0.11402508551881414, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.045, "eval_completions/max_length": 405.26, "eval_completions/max_terminated_length": 380.08, "eval_completions/mean_length": 276.665, "eval_completions/mean_terminated_length": 266.34190826416017, "eval_completions/min_length": 175.58, "eval_completions/min_terminated_length": 175.58, "eval_entropy": 0.24157741904258728, "eval_frac_reward_zero_std": 0.06, "eval_kl": 0.005511236232705414, "eval_loss": 2.7553000109037384e-05, "eval_num_tokens": 22399811.0, "eval_reward": 2.111875011920929, "eval_reward_std": 0.49019926600158215, "eval_rewards/code_complexity_reward/mean": 0.7692500007152557, "eval_rewards/code_complexity_reward/std": 0.14989317081868647, "eval_rewards/code_execution_reward/mean": 0.275, "eval_rewards/code_execution_reward/std": 0.3325467395782471, "eval_rewards/code_syntax_reward/mean": 0.47625, "eval_rewards/code_syntax_reward/std": 0.05984924167394638, "eval_rewards/reasoning_present_reward_func/mean": 0.09950000166893005, "eval_rewards/reasoning_present_reward_func/std": 0.001414213627576828, "eval_rewards/xmlcount_reward_func/mean": 0.491875, "eval_rewards/xmlcount_reward_func/std": 0.017915936782956124, "eval_runtime": 932.2711, "eval_samples_per_second": 0.107, "eval_steps_per_second": 0.014, "step": 100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.044921875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 281.359375, "completions/mean_terminated_length": 270.51123046875, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.2339237171690911, "epoch": 0.11516533637400228, "frac_reward_zero_std": 0.046875, "grad_norm": 0.025964384898543358, "kl": 0.005305602622684091, "learning_rate": 4.9971467776484526e-06, "loss": 2.6532623451203108e-05, "num_tokens": 22614755.0, "reward": 2.1583008766174316, "reward_std": 0.6398782134056091, "rewards/code_complexity_reward/mean": 0.75341796875, "rewards/code_complexity_reward/std": 0.21079720556735992, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.474609375, "rewards/code_syntax_reward/std": 0.10988271236419678, "rewards/reasoning_present_reward_func/mean": 0.09824219346046448, "rewards/reasoning_present_reward_func/std": 0.013154060579836369, "rewards/xmlcount_reward_func/mean": 0.490234375, "rewards/xmlcount_reward_func/std": 0.03785909339785576, "step": 101, "step_time": 62.10478555969894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.041015625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 291.28125, "completions/mean_terminated_length": 281.8411560058594, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.23522402881644666, "epoch": 0.11630558722919042, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.027507588267326355, "kl": 0.006600369870284339, "learning_rate": 4.9966515371508445e-06, "loss": 3.3002350392052904e-05, "num_tokens": 22833543.0, "reward": 2.1637208461761475, "reward_std": 0.6210617423057556, "rewards/code_complexity_reward/mean": 0.7618163824081421, "rewards/code_complexity_reward/std": 0.20320817828178406, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.490966796875, "rewards/xmlcount_reward_func/std": 0.03922802954912186, "step": 102, "step_time": 70.92139722965658 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 270.03125, "completions/mean_terminated_length": 264.2239990234375, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.22721580299548805, "epoch": 0.11744583808437856, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.03251733258366585, "kl": 0.005359108730772277, "learning_rate": 4.9961167141645435e-06, "loss": 2.6765745133161545e-05, "num_tokens": 23041351.0, "reward": 2.255810499191284, "reward_std": 0.6237348914146423, "rewards/code_complexity_reward/mean": 0.788769543170929, "rewards/code_complexity_reward/std": 0.17894518375396729, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.0986328125, "rewards/reasoning_present_reward_func/std": 0.011623830534517765, "rewards/xmlcount_reward_func/mean": 0.491455078125, "rewards/xmlcount_reward_func/std": 0.04160411283373833, "step": 103, "step_time": 63.81842847261578 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 270.56640625, "completions/mean_terminated_length": 264.7720031738281, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.2307599065825343, "epoch": 0.11858608893956671, "frac_reward_zero_std": 0.09375, "grad_norm": 0.02594231627881527, "kl": 0.005676496079104254, "learning_rate": 4.995542317168756e-06, "loss": 2.8385264158714563e-05, "num_tokens": 23248637.0, "reward": 2.2635743618011475, "reward_std": 0.6079354286193848, "rewards/code_complexity_reward/mean": 0.7850586175918579, "rewards/code_complexity_reward/std": 0.1727989763021469, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.0986328125, "rewards/reasoning_present_reward_func/std": 0.011623830534517765, "rewards/xmlcount_reward_func/mean": 0.4921875, "rewards/xmlcount_reward_func/std": 0.03750407695770264, "step": 104, "step_time": 70.31647734064609 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.033203125, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 284.201171875, "completions/mean_terminated_length": 276.3777770996094, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.23564442922361195, "epoch": 0.11972633979475485, "frac_reward_zero_std": 0.0546875, "grad_norm": 0.026194678619503975, "kl": 0.008150979425408877, "learning_rate": 4.994928355270105e-06, "loss": 4.08193445764482e-05, "num_tokens": 23463048.0, "reward": 2.1629881858825684, "reward_std": 0.6085594892501831, "rewards/code_complexity_reward/mean": 0.7718750238418579, "rewards/code_complexity_reward/std": 0.18869724869728088, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.0986328125, "rewards/reasoning_present_reward_func/std": 0.011623830534517765, "rewards/xmlcount_reward_func/mean": 0.48779296875, "rewards/xmlcount_reward_func/std": 0.04918526113033295, "step": 105, "step_time": 79.39715717267245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 279.857421875, "completions/mean_terminated_length": 277.1047668457031, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.22830879944376647, "epoch": 0.12086659064994298, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.03755500167608261, "kl": 0.005519333892152645, "learning_rate": 4.994274838202483e-06, "loss": 2.7596252039074898e-05, "num_tokens": 23675631.0, "reward": 2.241015672683716, "reward_std": 0.5660116672515869, "rewards/code_complexity_reward/mean": 0.7928710579872131, "rewards/code_complexity_reward/std": 0.15035170316696167, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843363426625729, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.028927749022841454, "step": 106, "step_time": 81.01581315975636 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 275.83203125, "completions/mean_terminated_length": 270.6466979980469, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.22931298008188605, "epoch": 0.12200684150513112, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.02913593128323555, "kl": 0.005977226643153699, "learning_rate": 4.993581776326901e-06, "loss": 2.9893995815655217e-05, "num_tokens": 23884761.0, "reward": 2.2207517623901367, "reward_std": 0.576655924320221, "rewards/code_complexity_reward/mean": 0.780957043170929, "rewards/code_complexity_reward/std": 0.16857774555683136, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812851272523403, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.033988069742918015, "step": 107, "step_time": 67.1685042893514 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 267.974609375, "completions/mean_terminated_length": 262.61676025390625, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.23858800204470754, "epoch": 0.12314709236031927, "frac_reward_zero_std": 0.0625, "grad_norm": 0.03522571548819542, "kl": 0.005869289154361468, "learning_rate": 4.9928491806313216e-06, "loss": 2.9505346901714802e-05, "num_tokens": 24090624.0, "reward": 2.2154297828674316, "reward_std": 0.5965726375579834, "rewards/code_complexity_reward/mean": 0.7907226085662842, "rewards/code_complexity_reward/std": 0.18147273361682892, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09902344644069672, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.49462890625, "rewards/xmlcount_reward_func/std": 0.031791869550943375, "step": 108, "step_time": 77.07581944111735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 266.08984375, "completions/mean_terminated_length": 261.1912536621094, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.22960085608065128, "epoch": 0.12428734321550741, "frac_reward_zero_std": 0.0546875, "grad_norm": 0.025218123570084572, "kl": 0.006322263143374585, "learning_rate": 4.992077062730485e-06, "loss": 3.156973980367184e-05, "num_tokens": 24294778.0, "reward": 2.3087403774261475, "reward_std": 0.59367835521698, "rewards/code_complexity_reward/mean": 0.8064453601837158, "rewards/code_complexity_reward/std": 0.15583078563213348, "rewards/code_execution_reward/mean": 0.419921875, "rewards/code_execution_reward/std": 0.4940285086631775, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.02716791070997715, "step": 109, "step_time": 59.88844134192914 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 273.115234375, "completions/mean_terminated_length": 267.8702697753906, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.22691780584864318, "epoch": 0.12542759407069556, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.027802111580967903, "kl": 0.005626559744996484, "learning_rate": 4.991265434865726e-06, "loss": 2.8131413273513317e-05, "num_tokens": 24502569.0, "reward": 2.1922364234924316, "reward_std": 0.5738167762756348, "rewards/code_complexity_reward/mean": 0.7789062261581421, "rewards/code_complexity_reward/std": 0.1668001413345337, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.027255699038505554, "step": 110, "step_time": 57.05219815764576 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 273.720703125, "completions/mean_terminated_length": 268.489013671875, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.22575042117387056, "epoch": 0.1265678449258837, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.026442285627126694, "kl": 0.006111047961894656, "learning_rate": 4.990414309904781e-06, "loss": 3.063183612539433e-05, "num_tokens": 24713578.0, "reward": 2.1617674827575684, "reward_std": 0.6052077412605286, "rewards/code_complexity_reward/mean": 0.7727539539337158, "rewards/code_complexity_reward/std": 0.19321876764297485, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.492919921875, "rewards/xmlcount_reward_func/std": 0.03805340826511383, "step": 111, "step_time": 58.467571541666985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 256.830078125, "completions/mean_terminated_length": 253.80435180664062, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.22735812375321984, "epoch": 0.12770809578107184, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.027796076610684395, "kl": 0.008198462572181597, "learning_rate": 4.98952370134158e-06, "loss": 4.091400478500873e-05, "num_tokens": 24912563.0, "reward": 2.2790040969848633, "reward_std": 0.574246346950531, "rewards/code_complexity_reward/mean": 0.8084961175918579, "rewards/code_complexity_reward/std": 0.14642204344272614, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.032946839928627014, "step": 112, "step_time": 58.158880236558616 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 268.7265625, "completions/mean_terminated_length": 260.8790283203125, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.23159212316386402, "epoch": 0.12884834663625996, "frac_reward_zero_std": 0.046875, "grad_norm": 0.031140228733420372, "kl": 0.006972101789870067, "learning_rate": 4.988593623296038e-06, "loss": 3.4786557080224156e-05, "num_tokens": 25118839.0, "reward": 2.164501905441284, "reward_std": 0.6105349063873291, "rewards/code_complexity_reward/mean": 0.773730456829071, "rewards/code_complexity_reward/std": 0.19646671414375305, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.09902344644069672, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.492919921875, "rewards/xmlcount_reward_func/std": 0.03556118905544281, "step": 113, "step_time": 62.36340954899788 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 272.193359375, "completions/mean_terminated_length": 266.9281311035156, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.22853899956680834, "epoch": 0.12998859749144812, "frac_reward_zero_std": 0.0625, "grad_norm": 0.02656293846666813, "kl": 0.00593985623709159, "learning_rate": 4.987624090513825e-06, "loss": 2.955706077045761e-05, "num_tokens": 25327682.0, "reward": 2.1878418922424316, "reward_std": 0.5833153128623962, "rewards/code_complexity_reward/mean": 0.782910168170929, "rewards/code_complexity_reward/std": 0.17738017439842224, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09902344644069672, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.494384765625, "rewards/xmlcount_reward_func/std": 0.032227516174316406, "step": 114, "step_time": 97.96527441125363 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 283.87890625, "completions/mean_terminated_length": 275.5668029785156, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.23400648590177298, "epoch": 0.13112884834663627, "frac_reward_zero_std": 0.078125, "grad_norm": 0.026790285483002663, "kl": 0.00647415419734898, "learning_rate": 4.986615118366138e-06, "loss": 3.233309689676389e-05, "num_tokens": 25542140.0, "reward": 2.1043457984924316, "reward_std": 0.5986587405204773, "rewards/code_complexity_reward/mean": 0.7655273675918579, "rewards/code_complexity_reward/std": 0.19921104609966278, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.09882812947034836, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.490966796875, "rewards/xmlcount_reward_func/std": 0.040757179260253906, "step": 115, "step_time": 58.83592155110091 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 275.03515625, "completions/mean_terminated_length": 270.79522705078125, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.23050048691220582, "epoch": 0.1322690992018244, "frac_reward_zero_std": 0.0546875, "grad_norm": 0.026964211836457253, "kl": 0.006776059180992888, "learning_rate": 4.985566722849454e-06, "loss": 3.4020107705146074e-05, "num_tokens": 25751974.0, "reward": 2.185546875, "reward_std": 0.5804808139801025, "rewards/code_complexity_reward/mean": 0.778027355670929, "rewards/code_complexity_reward/std": 0.1734326332807541, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.026730235666036606, "step": 116, "step_time": 88.03418215736747 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 270.251953125, "completions/mean_terminated_length": 262.45361328125, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.2352555268444121, "epoch": 0.13340935005701254, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.02848079241812229, "kl": 0.008900973185518524, "learning_rate": 4.984478920585277e-06, "loss": 4.441101918928325e-05, "num_tokens": 25958183.0, "reward": 2.190722703933716, "reward_std": 0.5939431190490723, "rewards/code_complexity_reward/mean": 0.7780272960662842, "rewards/code_complexity_reward/std": 0.17988528311252594, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4921875, "rewards/xmlcount_reward_func/std": 0.0366797111928463, "step": 117, "step_time": 61.58066969551146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025390625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 262.587890625, "completions/mean_terminated_length": 256.0901794433594, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.22968924953602254, "epoch": 0.1345496009122007, "frac_reward_zero_std": 0.046875, "grad_norm": 0.02829144150018692, "kl": 0.0065240537842328195, "learning_rate": 4.983351728819874e-06, "loss": 3.2649921195115894e-05, "num_tokens": 26161308.0, "reward": 2.2337403297424316, "reward_std": 0.5836344361305237, "rewards/code_complexity_reward/mean": 0.8021484613418579, "rewards/code_complexity_reward/std": 0.16255442798137665, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09882812947034836, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.033988069742918015, "step": 118, "step_time": 60.34283592924476 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 258.763671875, "completions/mean_terminated_length": 256.2662658691406, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.22525634430348873, "epoch": 0.13568985176738882, "frac_reward_zero_std": 0.09375, "grad_norm": 0.029568390920758247, "kl": 0.006566897925949888, "learning_rate": 4.9821851654240025e-06, "loss": 3.267908323323354e-05, "num_tokens": 26361103.0, "reward": 2.3109374046325684, "reward_std": 0.5793642401695251, "rewards/code_complexity_reward/mean": 0.802734375, "rewards/code_complexity_reward/std": 0.14682665467262268, "rewards/code_execution_reward/mean": 0.421875, "rewards/code_execution_reward/std": 0.49434176087379456, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843363426625729, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.03281606733798981, "step": 119, "step_time": 69.30490815546364 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 262.087890625, "completions/mean_terminated_length": 259.12451171875, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.22594658588059247, "epoch": 0.13683010262257697, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.029063096269965172, "kl": 0.0063633235622546636, "learning_rate": 4.98097924889263e-06, "loss": 3.1658681109547615e-05, "num_tokens": 26565188.0, "reward": 2.2579102516174316, "reward_std": 0.5727426409721375, "rewards/code_complexity_reward/mean": 0.8052734732627869, "rewards/code_complexity_reward/std": 0.1508343517780304, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49462890625, "rewards/xmlcount_reward_func/std": 0.0327395424246788, "step": 120, "step_time": 78.28971132542938 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 268.365234375, "completions/mean_terminated_length": 265.9625244140625, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.23046807083301246, "epoch": 0.1379703534777651, "frac_reward_zero_std": 0.078125, "grad_norm": 0.026411088183522224, "kl": 0.006560861209436553, "learning_rate": 4.979733998344632e-06, "loss": 3.257319986005314e-05, "num_tokens": 26771751.0, "reward": 2.18408203125, "reward_std": 0.5360949635505676, "rewards/code_complexity_reward/mean": 0.79052734375, "rewards/code_complexity_reward/std": 0.15203483402729034, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.028997857123613358, "step": 121, "step_time": 69.37030460685492 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 245.373046875, "completions/mean_terminated_length": 244.8512725830078, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.2308702147565782, "epoch": 0.13911060433295325, "frac_reward_zero_std": 0.078125, "grad_norm": 0.02745538204908371, "kl": 0.007122264287318103, "learning_rate": 4.9784494335225e-06, "loss": 3.552134148776531e-05, "num_tokens": 26963806.0, "reward": 2.3135743141174316, "reward_std": 0.5390874147415161, "rewards/code_complexity_reward/mean": 0.8205077648162842, "rewards/code_complexity_reward/std": 0.1174139752984047, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 122, "step_time": 66.17671398911625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 271.4375, "completions/mean_terminated_length": 262.67205810546875, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.2329344362951815, "epoch": 0.1402508551881414, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.02794659696519375, "kl": 0.00678294878161978, "learning_rate": 4.977125574792018e-06, "loss": 3.392782309674658e-05, "num_tokens": 27172674.0, "reward": 2.2255859375, "reward_std": 0.6452271938323975, "rewards/code_complexity_reward/mean": 0.772656261920929, "rewards/code_complexity_reward/std": 0.20417456328868866, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.4931640625, "rewards/xmlcount_reward_func/std": 0.03339334949851036, "step": 123, "step_time": 76.58651342615485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 259.888671875, "completions/mean_terminated_length": 251.75604248046875, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.23157053557224572, "epoch": 0.14139110604332952, "frac_reward_zero_std": 0.078125, "grad_norm": 0.031542882323265076, "kl": 0.006605697450140724, "learning_rate": 4.975762443141949e-06, "loss": 3.295089118182659e-05, "num_tokens": 27374353.0, "reward": 2.229443311691284, "reward_std": 0.6152923703193665, "rewards/code_complexity_reward/mean": 0.7912108898162842, "rewards/code_complexity_reward/std": 0.18250511586666107, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.03307618945837021, "step": 124, "step_time": 57.69025210477412 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 264.212890625, "completions/mean_terminated_length": 259.7793273925781, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.22547683399170637, "epoch": 0.14253135689851767, "frac_reward_zero_std": 0.0546875, "grad_norm": 0.049909573048353195, "kl": 0.006771458156435983, "learning_rate": 4.9743600601836974e-06, "loss": 3.3756390621419996e-05, "num_tokens": 27579342.0, "reward": 2.221923828125, "reward_std": 0.587867796421051, "rewards/code_complexity_reward/mean": 0.7803710699081421, "rewards/code_complexity_reward/std": 0.17433254420757294, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.03213844448328018, "step": 125, "step_time": 60.12967143021524 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 261.234375, "completions/mean_terminated_length": 258.7613525390625, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.23067352059297264, "epoch": 0.14367160775370583, "frac_reward_zero_std": 0.0625, "grad_norm": 0.03591250628232956, "kl": 0.006555946136359125, "learning_rate": 4.9729184481509644e-06, "loss": 3.2815187296364456e-05, "num_tokens": 27782466.0, "reward": 2.2311034202575684, "reward_std": 0.5527157187461853, "rewards/code_complexity_reward/mean": 0.7962890267372131, "rewards/code_complexity_reward/std": 0.14244131743907928, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.02955172397196293, "step": 126, "step_time": 81.58105740044266 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 255.466796875, "completions/mean_terminated_length": 250.35658264160156, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.23029351723380387, "epoch": 0.14481185860889395, "frac_reward_zero_std": 0.0625, "grad_norm": 0.0322989895939827, "kl": 0.0069717376318294555, "learning_rate": 4.971437629899399e-06, "loss": 3.4893571864813566e-05, "num_tokens": 27980037.0, "reward": 2.2426271438598633, "reward_std": 0.6007664799690247, "rewards/code_complexity_reward/mean": 0.7953124642372131, "rewards/code_complexity_reward/std": 0.1756383329629898, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.027255699038505554, "step": 127, "step_time": 78.89367237873375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 259.9765625, "completions/mean_terminated_length": 256.9881591796875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2319807941094041, "epoch": 0.1459521094640821, "frac_reward_zero_std": 0.0546875, "grad_norm": 0.030474143102765083, "kl": 0.00878206225024769, "learning_rate": 4.969917628906234e-06, "loss": 4.382542101666331e-05, "num_tokens": 28183785.0, "reward": 2.1976561546325684, "reward_std": 0.5377908945083618, "rewards/code_complexity_reward/mean": 0.7974609136581421, "rewards/code_complexity_reward/std": 0.150255486369133, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.02564004622399807, "step": 128, "step_time": 81.41957739274949 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 259.556640625, "completions/mean_terminated_length": 255.0397491455078, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2271695111412555, "epoch": 0.14709236031927023, "frac_reward_zero_std": 0.0546875, "grad_norm": 0.034119199961423874, "kl": 0.007615655020345002, "learning_rate": 4.968358469269917e-06, "loss": 3.8276833947747946e-05, "num_tokens": 28387010.0, "reward": 2.229931592941284, "reward_std": 0.5743765234947205, "rewards/code_complexity_reward/mean": 0.7930664420127869, "rewards/code_complexity_reward/std": 0.15525931119918823, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812851272523403, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.032308951020240784, "step": 129, "step_time": 61.56261794175953 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 265.287109375, "completions/mean_terminated_length": 260.3725280761719, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.23064173851162195, "epoch": 0.14823261117445838, "frac_reward_zero_std": 0.09375, "grad_norm": 0.0268456619232893, "kl": 0.00789166243703221, "learning_rate": 4.966760175709725e-06, "loss": 3.9275430026464164e-05, "num_tokens": 28592333.0, "reward": 2.177978515625, "reward_std": 0.5797501802444458, "rewards/code_complexity_reward/mean": 0.7830078601837158, "rewards/code_complexity_reward/std": 0.1697942167520523, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.03331366926431656, "step": 130, "step_time": 57.58666274789721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 266.169921875, "completions/mean_terminated_length": 261.2729187011719, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.22137373965233564, "epoch": 0.14937286202964653, "frac_reward_zero_std": 0.0625, "grad_norm": 0.025777636095881462, "kl": 0.007671776576898992, "learning_rate": 4.96512277356537e-06, "loss": 3.833469236269593e-05, "num_tokens": 28797964.0, "reward": 2.1607909202575684, "reward_std": 0.5626085996627808, "rewards/code_complexity_reward/mean": 0.78564453125, "rewards/code_complexity_reward/std": 0.17059804499149323, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.02948698401451111, "step": 131, "step_time": 68.47260410618037 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 245.40625, "completions/mean_terminated_length": 244.88453674316406, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.22647977340966463, "epoch": 0.15051311288483465, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.02816055156290531, "kl": 0.009122651426878292, "learning_rate": 4.963446288796605e-06, "loss": 4.5630891690962017e-05, "num_tokens": 28992944.0, "reward": 2.3144044876098633, "reward_std": 0.5497230887413025, "rewards/code_complexity_reward/mean": 0.8158203363418579, "rewards/code_complexity_reward/std": 0.12498027086257935, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660499989986, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 132, "step_time": 61.899257962591946 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 264.626953125, "completions/mean_terminated_length": 257.67266845703125, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.2300021171104163, "epoch": 0.1516533637400228, "frac_reward_zero_std": 0.0625, "grad_norm": 0.03328491374850273, "kl": 0.00858757003879873, "learning_rate": 4.961730747982804e-06, "loss": 4.2808773287106305e-05, "num_tokens": 29198045.0, "reward": 2.1203126907348633, "reward_std": 0.5577662587165833, "rewards/code_complexity_reward/mean": 0.779980480670929, "rewards/code_complexity_reward/std": 0.1701403707265854, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49365234375, "rewards/xmlcount_reward_func/std": 0.03439070284366608, "step": 133, "step_time": 69.2066400963813 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 251.0859375, "completions/mean_terminated_length": 247.9921112060547, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.22473033657297492, "epoch": 0.15279361459521096, "frac_reward_zero_std": 0.0546875, "grad_norm": 0.028299592435359955, "kl": 0.0077180253065307625, "learning_rate": 4.9599761783225465e-06, "loss": 3.867878331220709e-05, "num_tokens": 29394209.0, "reward": 2.240966796875, "reward_std": 0.549862265586853, "rewards/code_complexity_reward/mean": 0.80712890625, "rewards/code_complexity_reward/std": 0.1439686268568039, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.02382291667163372, "step": 134, "step_time": 78.33477069810033 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 249.771484375, "completions/mean_terminated_length": 247.18540954589844, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.22272364841774106, "epoch": 0.15393386545039908, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.029376711696386337, "kl": 0.008092054307780927, "learning_rate": 4.9581826076331854e-06, "loss": 4.0202110540121794e-05, "num_tokens": 29590308.0, "reward": 2.30859375, "reward_std": 0.5743151903152466, "rewards/code_complexity_reward/mean": 0.806640625, "rewards/code_complexity_reward/std": 0.14475446939468384, "rewards/code_execution_reward/mean": 0.4140625, "rewards/code_execution_reward/std": 0.49304109811782837, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.015517610125243664, "step": 135, "step_time": 81.065200955607 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 255.927734375, "completions/mean_terminated_length": 250.30538940429688, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.22651391383260489, "epoch": 0.15507411630558723, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.028610562905669212, "kl": 0.009237838843546342, "learning_rate": 4.956350064350403e-06, "loss": 4.616937803803012e-05, "num_tokens": 29790539.0, "reward": 2.205322265625, "reward_std": 0.5656440854072571, "rewards/code_complexity_reward/mean": 0.795214831829071, "rewards/code_complexity_reward/std": 0.16230228543281555, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.019682783633470535, "step": 136, "step_time": 77.31548765674233 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 252.6796875, "completions/mean_terminated_length": 248.0397491455078, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.23140525282360613, "epoch": 0.15621436716077536, "frac_reward_zero_std": 0.0625, "grad_norm": 0.03594258427619934, "kl": 0.008424683212069795, "learning_rate": 4.954478577527761e-06, "loss": 4.206603625789285e-05, "num_tokens": 29987667.0, "reward": 2.255126953125, "reward_std": 0.5769562721252441, "rewards/code_complexity_reward/mean": 0.8057616949081421, "rewards/code_complexity_reward/std": 0.15854957699775696, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.02619195356965065, "step": 137, "step_time": 64.55246205255389 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 242.77734375, "completions/mean_terminated_length": 239.58499145507812, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.22415748704224825, "epoch": 0.1573546180159635, "frac_reward_zero_std": 0.078125, "grad_norm": 0.030936995521187782, "kl": 0.013478503708029166, "learning_rate": 4.952568176836246e-06, "loss": 6.7436762037687e-05, "num_tokens": 30182629.0, "reward": 2.3011717796325684, "reward_std": 0.5636828541755676, "rewards/code_complexity_reward/mean": 0.817089855670929, "rewards/code_complexity_reward/std": 0.14175282418727875, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.021983280777931213, "step": 138, "step_time": 72.83094310946763 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 258.431640625, "completions/mean_terminated_length": 256.4350280761719, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.23413574276492, "epoch": 0.15849486887115166, "frac_reward_zero_std": 0.03125, "grad_norm": 0.029939210042357445, "kl": 0.013061451696557924, "learning_rate": 4.9506188925637885e-06, "loss": 6.530352402478456e-05, "num_tokens": 30385610.0, "reward": 2.213134765625, "reward_std": 0.5620648860931396, "rewards/code_complexity_reward/mean": 0.7920898199081421, "rewards/code_complexity_reward/std": 0.15793099999427795, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03244912996888161, "step": 139, "step_time": 70.72713880147785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 248.08984375, "completions/mean_terminated_length": 245.4871826171875, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.23344311118125916, "epoch": 0.15963511972633979, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.029522504657506943, "kl": 0.009533877258945722, "learning_rate": 4.948630755614792e-06, "loss": 4.773087130161002e-05, "num_tokens": 30579540.0, "reward": 2.2265138626098633, "reward_std": 0.5714517831802368, "rewards/code_complexity_reward/mean": 0.798144519329071, "rewards/code_complexity_reward/std": 0.1613345593214035, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.022640403360128403, "step": 140, "step_time": 61.53861210681498 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 257.265625, "completions/mean_terminated_length": 253.2222442626953, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.22970815375447273, "epoch": 0.16077537058152794, "frac_reward_zero_std": 0.09375, "grad_norm": 0.02655455283820629, "kl": 0.013915062452724669, "learning_rate": 4.946603797509635e-06, "loss": 6.95659255143255e-05, "num_tokens": 30780008.0, "reward": 2.192431926727295, "reward_std": 0.5631154179573059, "rewards/code_complexity_reward/mean": 0.789355456829071, "rewards/code_complexity_reward/std": 0.16331622004508972, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.023893006145954132, "step": 141, "step_time": 59.209952684119344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 258.759765625, "completions/mean_terminated_length": 256.2623291015625, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.23760195495560765, "epoch": 0.1619156214367161, "frac_reward_zero_std": 0.046875, "grad_norm": 0.028442617505788803, "kl": 0.008590756813646294, "learning_rate": 4.944538050384181e-06, "loss": 4.281627479940653e-05, "num_tokens": 30982445.0, "reward": 2.1623048782348633, "reward_std": 0.5045996308326721, "rewards/code_complexity_reward/mean": 0.802539050579071, "rewards/code_complexity_reward/std": 0.1361096203327179, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02059757336974144, "step": 142, "step_time": 60.36151958536357 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 243.37109375, "completions/mean_terminated_length": 240.18577575683594, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.22587269358336926, "epoch": 0.1630558722919042, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.038347311317920685, "kl": 0.0152064622452599, "learning_rate": 4.94243354698926e-06, "loss": 7.608404848724604e-05, "num_tokens": 31175691.0, "reward": 2.2594237327575684, "reward_std": 0.5474927425384521, "rewards/code_complexity_reward/mean": 0.8150390982627869, "rewards/code_complexity_reward/std": 0.13192981481552124, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.028498241677880287, "step": 143, "step_time": 57.01981243956834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 244.92578125, "completions/mean_terminated_length": 241.22377014160156, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.230778347235173, "epoch": 0.16419612314709237, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.03932662308216095, "kl": 0.009391314364620484, "learning_rate": 4.9402903206901535e-06, "loss": 4.67579229734838e-05, "num_tokens": 31371065.0, "reward": 2.2688965797424316, "reward_std": 0.569214403629303, "rewards/code_complexity_reward/mean": 0.8172851800918579, "rewards/code_complexity_reward/std": 0.14932677149772644, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.018141774460673332, "step": 144, "step_time": 76.77159021515399 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 239.7578125, "completions/mean_terminated_length": 236.52964782714844, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.22692382824607193, "epoch": 0.1653363740022805, "frac_reward_zero_std": 0.09375, "grad_norm": 0.031234126538038254, "kl": 0.009785782895050943, "learning_rate": 4.938108405466065e-06, "loss": 4.893419099971652e-05, "num_tokens": 31562773.0, "reward": 2.3017578125, "reward_std": 0.5779369473457336, "rewards/code_complexity_reward/mean": 0.8179687857627869, "rewards/code_complexity_reward/std": 0.15387803316116333, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 145, "step_time": 60.599133423529565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 246.513671875, "completions/mean_terminated_length": 241.76341247558594, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2205955779645592, "epoch": 0.16647662485746864, "frac_reward_zero_std": 0.09375, "grad_norm": 0.03465723246335983, "kl": 0.011090171108662616, "learning_rate": 4.935887835909581e-06, "loss": 5.5478361900895834e-05, "num_tokens": 31758896.0, "reward": 2.2681641578674316, "reward_std": 0.5692607760429382, "rewards/code_complexity_reward/mean": 0.8040039539337158, "rewards/code_complexity_reward/std": 0.1549527645111084, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.021852491423487663, "step": 146, "step_time": 58.37059367354959 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 238.921875, "completions/mean_terminated_length": 237.31239318847656, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.22370563284493983, "epoch": 0.1676168757126568, "frac_reward_zero_std": 0.078125, "grad_norm": 0.03342772647738457, "kl": 0.011745964642614126, "learning_rate": 4.933628647226123e-06, "loss": 5.8756733778864145e-05, "num_tokens": 31951484.0, "reward": 2.2504396438598633, "reward_std": 0.5302110910415649, "rewards/code_complexity_reward/mean": 0.81640625, "rewards/code_complexity_reward/std": 0.1255296915769577, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.026382790878415108, "step": 147, "step_time": 62.7755225840956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 245.802734375, "completions/mean_terminated_length": 241.577392578125, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.2316304196137935, "epoch": 0.16875712656784492, "frac_reward_zero_std": 0.0625, "grad_norm": 0.031039372086524963, "kl": 0.01549780803907197, "learning_rate": 4.93133087523339e-06, "loss": 7.746287155896425e-05, "num_tokens": 32144671.0, "reward": 2.2459473609924316, "reward_std": 0.5712595582008362, "rewards/code_complexity_reward/mean": 0.7931640148162842, "rewards/code_complexity_reward/std": 0.15842963755130768, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.02382291667163372, "step": 148, "step_time": 77.05459019728005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 242.884765625, "completions/mean_terminated_length": 239.6936798095703, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.22758719557896256, "epoch": 0.16989737742303307, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.030708497390151024, "kl": 0.010860662849154323, "learning_rate": 4.928994556360787e-06, "loss": 5.4110125347506255e-05, "num_tokens": 32336184.0, "reward": 2.261035203933716, "reward_std": 0.5442303419113159, "rewards/code_complexity_reward/mean": 0.8153320550918579, "rewards/code_complexity_reward/std": 0.1403249204158783, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 149, "step_time": 58.14823040459305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 243.162109375, "completions/mean_terminated_length": 239.43565368652344, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.23352679796516895, "epoch": 0.17103762827822122, "frac_reward_zero_std": 0.09375, "grad_norm": 0.03014707937836647, "kl": 0.016750152768508997, "learning_rate": 4.926619727648852e-06, "loss": 8.357228944078088e-05, "num_tokens": 32528911.0, "reward": 2.249267578125, "reward_std": 0.5600733757019043, "rewards/code_complexity_reward/mean": 0.8069336414337158, "rewards/code_complexity_reward/std": 0.15023063123226166, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.028556859120726585, "step": 150, "step_time": 61.00817208271474 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 255.412109375, "completions/mean_terminated_length": 251.85545349121094, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.22919196891598403, "epoch": 0.17217787913340935, "frac_reward_zero_std": 0.0546875, "grad_norm": 0.03096112236380577, "kl": 0.010788759391289204, "learning_rate": 4.924206426748668e-06, "loss": 5.388067802414298e-05, "num_tokens": 32729422.0, "reward": 2.183349609375, "reward_std": 0.5620215535163879, "rewards/code_complexity_reward/mean": 0.7966797351837158, "rewards/code_complexity_reward/std": 0.16208255290985107, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03516262024641037, "step": 151, "step_time": 79.29040466807783 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 243.404296875, "completions/mean_terminated_length": 241.28936767578125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.22916759504005313, "epoch": 0.1733181299885975, "frac_reward_zero_std": 0.078125, "grad_norm": 0.03117053210735321, "kl": 0.01120224485930521, "learning_rate": 4.921754691921262e-06, "loss": 5.596363916993141e-05, "num_tokens": 32921525.0, "reward": 2.193310499191284, "reward_std": 0.5762975215911865, "rewards/code_complexity_reward/mean": 0.80810546875, "rewards/code_complexity_reward/std": 0.1784849315881729, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.027517380192875862, "step": 152, "step_time": 77.67282583285123 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 239.2578125, "completions/mean_terminated_length": 237.1102294921875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.22902117134071887, "epoch": 0.17445838084378562, "frac_reward_zero_std": 0.046875, "grad_norm": 0.031007081270217896, "kl": 0.013320015015779063, "learning_rate": 4.919264562037003e-06, "loss": 6.667034904239699e-05, "num_tokens": 33111161.0, "reward": 2.2563962936401367, "reward_std": 0.5549848079681396, "rewards/code_complexity_reward/mean": 0.80810546875, "rewards/code_complexity_reward/std": 0.1435765027999878, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.03510142117738724, "step": 153, "step_time": 77.53116395883262 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 505.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 233.96875, "completions/mean_terminated_length": 233.96875, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.22067751549184322, "epoch": 0.17559863169897377, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.029529916122555733, "kl": 0.01616298424778506, "learning_rate": 4.9167360765749845e-06, "loss": 8.073715434875339e-05, "num_tokens": 33300185.0, "reward": 2.2553224563598633, "reward_std": 0.5472889542579651, "rewards/code_complexity_reward/mean": 0.802050769329071, "rewards/code_complexity_reward/std": 0.1404169499874115, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 154, "step_time": 58.60316814575344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 247.87109375, "completions/mean_terminated_length": 243.6785888671875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22383178235031664, "epoch": 0.17673888255416192, "frac_reward_zero_std": 0.09375, "grad_norm": 0.030694402754306793, "kl": 0.011418063601013273, "learning_rate": 4.914169275622397e-06, "loss": 5.7023771660169587e-05, "num_tokens": 33496347.0, "reward": 2.22119140625, "reward_std": 0.5761996507644653, "rewards/code_complexity_reward/mean": 0.7867187261581421, "rewards/code_complexity_reward/std": 0.16349704563617706, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.021852491423487663, "step": 155, "step_time": 65.88493523746729 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 243.87109375, "completions/mean_terminated_length": 241.7598419189453, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.23363410867750645, "epoch": 0.17787913340935005, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.03192905709147453, "kl": 0.012163304083514959, "learning_rate": 4.911564199873894e-06, "loss": 6.087432848289609e-05, "num_tokens": 33690273.0, "reward": 2.173828125, "reward_std": 0.5038657784461975, "rewards/code_complexity_reward/mean": 0.816699206829071, "rewards/code_complexity_reward/std": 0.12476766109466553, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03206123411655426, "step": 156, "step_time": 62.47756314929575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 233.58984375, "completions/mean_terminated_length": 232.498046875, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.22186969849281013, "epoch": 0.1790193842645382, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.03944079205393791, "kl": 0.013233961239166092, "learning_rate": 4.908920890630947e-06, "loss": 6.61211452097632e-05, "num_tokens": 33876195.0, "reward": 2.324951171875, "reward_std": 0.5464963316917419, "rewards/code_complexity_reward/mean": 0.8173828125, "rewards/code_complexity_reward/std": 0.12717995047569275, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 157, "step_time": 68.32974316459149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 236.876953125, "completions/mean_terminated_length": 233.61463928222656, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.23736028163693845, "epoch": 0.18015963511972635, "frac_reward_zero_std": 0.078125, "grad_norm": 0.035462599247694016, "kl": 0.013432941283099353, "learning_rate": 4.906239389801191e-06, "loss": 6.71118323225528e-05, "num_tokens": 34066820.0, "reward": 2.21875, "reward_std": 0.557178258895874, "rewards/code_complexity_reward/mean": 0.8095703125, "rewards/code_complexity_reward/std": 0.15220560133457184, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.02048126794397831, "step": 158, "step_time": 81.89866472408175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 237.751953125, "completions/mean_terminated_length": 235.59251403808594, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2242794211488217, "epoch": 0.18129988597491448, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.031157292425632477, "kl": 0.015477960550924763, "learning_rate": 4.903519739897755e-06, "loss": 7.729524077149108e-05, "num_tokens": 34255905.0, "reward": 2.3184571266174316, "reward_std": 0.5467557907104492, "rewards/code_complexity_reward/mean": 0.8070312738418579, "rewards/code_complexity_reward/std": 0.12057986855506897, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 159, "step_time": 71.75457703415304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 242.740234375, "completions/mean_terminated_length": 239.0079345703125, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.22647906746715307, "epoch": 0.18244013683010263, "frac_reward_zero_std": 0.109375, "grad_norm": 0.03430405259132385, "kl": 0.013514900383597706, "learning_rate": 4.9007619840385975e-06, "loss": 6.749638123437762e-05, "num_tokens": 34449288.0, "reward": 2.265625, "reward_std": 0.5680006742477417, "rewards/code_complexity_reward/mean": 0.80908203125, "rewards/code_complexity_reward/std": 0.14293722808361053, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.02570982649922371, "step": 160, "step_time": 70.90224775951356 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 235.76953125, "completions/mean_terminated_length": 233.594482421875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.2315020803362131, "epoch": 0.18358038768529075, "frac_reward_zero_std": 0.0625, "grad_norm": 0.03138595446944237, "kl": 0.013230334276158828, "learning_rate": 4.897966165945815e-06, "loss": 6.608365220017731e-05, "num_tokens": 34638262.0, "reward": 2.2428221702575684, "reward_std": 0.5562586784362793, "rewards/code_complexity_reward/mean": 0.8017578125, "rewards/code_complexity_reward/std": 0.15421931445598602, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.014529787935316563, "step": 161, "step_time": 95.21343524567783 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 235.955078125, "completions/mean_terminated_length": 234.32810974121094, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.227990732062608, "epoch": 0.1847206385404789, "frac_reward_zero_std": 0.09375, "grad_norm": 0.03056161478161812, "kl": 0.012201590230688453, "learning_rate": 4.8951323299449514e-06, "loss": 6.093451884225942e-05, "num_tokens": 34826663.0, "reward": 2.2710938453674316, "reward_std": 0.5330691933631897, "rewards/code_complexity_reward/mean": 0.8097655773162842, "rewards/code_complexity_reward/std": 0.12228397279977798, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02465205453336239, "step": 162, "step_time": 67.0880550807342 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 236.9921875, "completions/mean_terminated_length": 234.2800750732422, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.22606653673574328, "epoch": 0.18586088939566706, "frac_reward_zero_std": 0.109375, "grad_norm": 0.03261985257267952, "kl": 0.01908813667250797, "learning_rate": 4.892260520964295e-06, "loss": 9.558700548950583e-05, "num_tokens": 35017067.0, "reward": 2.206494092941284, "reward_std": 0.5195451378822327, "rewards/code_complexity_reward/mean": 0.81640625, "rewards/code_complexity_reward/std": 0.12978370487689972, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.021246958523988724, "step": 163, "step_time": 61.953845573589206 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 232.845703125, "completions/mean_terminated_length": 230.64764404296875, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2262307540513575, "epoch": 0.18700114025085518, "frac_reward_zero_std": 0.109375, "grad_norm": 0.031208723783493042, "kl": 0.016135880468937103, "learning_rate": 4.889350784534168e-06, "loss": 8.06963216746226e-05, "num_tokens": 35204480.0, "reward": 2.2935547828674316, "reward_std": 0.5372657775878906, "rewards/code_complexity_reward/mean": 0.825878918170929, "rewards/code_complexity_reward/std": 0.12437400221824646, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 164, "step_time": 60.58621198218316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 234.904296875, "completions/mean_terminated_length": 233.27113342285156, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.23032439802773297, "epoch": 0.18814139110604333, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.03161782771348953, "kl": 0.013424207194475457, "learning_rate": 4.886403166786203e-06, "loss": 6.721513636875898e-05, "num_tokens": 35394483.0, "reward": 2.2835936546325684, "reward_std": 0.5711482167243958, "rewards/code_complexity_reward/mean": 0.801953136920929, "rewards/code_complexity_reward/std": 0.15110841393470764, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 165, "step_time": 59.51647731103003 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 231.921875, "completions/mean_terminated_length": 229.15975952148438, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.22528509702533484, "epoch": 0.18928164196123148, "frac_reward_zero_std": 0.140625, "grad_norm": 0.03262278437614441, "kl": 0.01299568928516237, "learning_rate": 4.883417714452607e-06, "loss": 6.497603317257017e-05, "num_tokens": 35580343.0, "reward": 2.304931640625, "reward_std": 0.5727366209030151, "rewards/code_complexity_reward/mean": 0.8125, "rewards/code_complexity_reward/std": 0.14272987842559814, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03250797092914581, "step": 166, "step_time": 68.32194846589118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 231.2109375, "completions/mean_terminated_length": 227.31881713867188, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23049082676880062, "epoch": 0.1904218928164196, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.03420884907245636, "kl": 0.014371444085554685, "learning_rate": 4.880394474865433e-06, "loss": 7.162413385231048e-05, "num_tokens": 35767527.0, "reward": 2.2112793922424316, "reward_std": 0.5711624026298523, "rewards/code_complexity_reward/mean": 0.801464855670929, "rewards/code_complexity_reward/std": 0.1619439274072647, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.03238280490040779, "step": 167, "step_time": 66.33089871611446 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 236.529296875, "completions/mean_terminated_length": 231.04183959960938, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.23060583928599954, "epoch": 0.19156214367160776, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.031051669269800186, "kl": 0.013716530862438958, "learning_rate": 4.8773334959558165e-06, "loss": 6.856447726022452e-05, "num_tokens": 35958206.0, "reward": 2.2652344703674316, "reward_std": 0.5893356204032898, "rewards/code_complexity_reward/mean": 0.8026367425918579, "rewards/code_complexity_reward/std": 0.16044177114963531, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.02449164353311062, "step": 168, "step_time": 58.85621795151383 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 228.677734375, "completions/mean_terminated_length": 227.00787353515625, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.21822956670075655, "epoch": 0.19270239452679588, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.034671299159526825, "kl": 0.021940541686490178, "learning_rate": 4.874234826253223e-06, "loss": 0.00010957811173284426, "num_tokens": 36145849.0, "reward": 2.332324266433716, "reward_std": 0.5514192581176758, "rewards/code_complexity_reward/mean": 0.8167968988418579, "rewards/code_complexity_reward/std": 0.13026051223278046, "rewards/code_execution_reward/mean": 0.421875, "rewards/code_execution_reward/std": 0.49434176087379456, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 169, "step_time": 70.13853682484478 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 228.984375, "completions/mean_terminated_length": 227.31631469726562, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2300945781171322, "epoch": 0.19384264538198404, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.035833001136779785, "kl": 0.01461858612310607, "learning_rate": 4.871098514884675e-06, "loss": 7.301082951016724e-05, "num_tokens": 36330349.0, "reward": 2.248974561691284, "reward_std": 0.5441133975982666, "rewards/code_complexity_reward/mean": 0.80615234375, "rewards/code_complexity_reward/std": 0.14151209592819214, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 170, "step_time": 58.125130045227706 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 225.462890625, "completions/mean_terminated_length": 223.20669555664062, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.23235267959535122, "epoch": 0.1949828962371722, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.032203324139118195, "kl": 0.016939231063588522, "learning_rate": 4.867924611573977e-06, "loss": 8.469141903333366e-05, "num_tokens": 36512734.0, "reward": 2.2913575172424316, "reward_std": 0.5802858471870422, "rewards/code_complexity_reward/mean": 0.8148437738418579, "rewards/code_complexity_reward/std": 0.141609326004982, "rewards/code_execution_reward/mean": 0.390625, "rewards/code_execution_reward/std": 0.48836761713027954, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09882812947034836, "rewards/reasoning_present_reward_func/std": 0.010772225446999073, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.03992818295955658, "step": 171, "step_time": 90.348836989142 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 217.328125, "completions/mean_terminated_length": 216.75146484375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2276985126081854, "epoch": 0.1961231470923603, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.037126362323760986, "kl": 0.03599644462519791, "learning_rate": 4.864713166640921e-06, "loss": 0.00018016202375292778, "num_tokens": 36692538.0, "reward": 2.3021483421325684, "reward_std": 0.5104368329048157, "rewards/code_complexity_reward/mean": 0.84130859375, "rewards/code_complexity_reward/std": 0.10606668889522552, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.025821086019277573, "step": 172, "step_time": 60.69802646711469 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 223.337890625, "completions/mean_terminated_length": 222.20590209960938, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.22422935999929905, "epoch": 0.19726339794754846, "frac_reward_zero_std": 0.078125, "grad_norm": 0.03233739361166954, "kl": 0.01630443891917821, "learning_rate": 4.8614642310004975e-06, "loss": 8.15275197965093e-05, "num_tokens": 36876179.0, "reward": 2.296630859375, "reward_std": 0.5577265024185181, "rewards/code_complexity_reward/mean": 0.80859375, "rewards/code_complexity_reward/std": 0.13715152442455292, "rewards/code_execution_reward/mean": 0.396484375, "rewards/code_execution_reward/std": 0.4896455705165863, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.021303100511431694, "step": 173, "step_time": 87.13025365676731 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 218.244140625, "completions/mean_terminated_length": 217.0921630859375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.22046875674277544, "epoch": 0.19840364880273662, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.032009780406951904, "kl": 0.018633599625900388, "learning_rate": 4.8581778561620785e-06, "loss": 9.327253792434931e-05, "num_tokens": 37056116.0, "reward": 2.3068361282348633, "reward_std": 0.555749237537384, "rewards/code_complexity_reward/mean": 0.818359375, "rewards/code_complexity_reward/std": 0.13191664218902588, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 174, "step_time": 67.8535048160702 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 222.078125, "completions/mean_terminated_length": 220.36935424804688, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.22691516019403934, "epoch": 0.19954389965792474, "frac_reward_zero_std": 0.125, "grad_norm": 0.0317218117415905, "kl": 0.016241397461271845, "learning_rate": 4.8548540942286095e-06, "loss": 8.134225208777934e-05, "num_tokens": 37237600.0, "reward": 2.27978515625, "reward_std": 0.5570501089096069, "rewards/code_complexity_reward/mean": 0.80810546875, "rewards/code_complexity_reward/std": 0.1492568850517273, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02465205453336239, "step": 175, "step_time": 58.37607354670763 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 465.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 212.880859375, "completions/mean_terminated_length": 212.880859375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.23382976814173162, "epoch": 0.2006841505131129, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.03435453027486801, "kl": 0.01736430183518678, "learning_rate": 4.851492997895777e-06, "loss": 8.674681157572195e-05, "num_tokens": 37415695.0, "reward": 2.30322265625, "reward_std": 0.5410830974578857, "rewards/code_complexity_reward/mean": 0.8328124284744263, "rewards/code_complexity_reward/std": 0.12322118878364563, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 176, "step_time": 55.57473448291421 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 222.240234375, "completions/mean_terminated_length": 221.1039276123047, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.23887269292026758, "epoch": 0.20182440136830102, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.03353308141231537, "kl": 0.01809447405685205, "learning_rate": 4.848094620451177e-06, "loss": 9.039115684572607e-05, "num_tokens": 37598486.0, "reward": 2.2666990756988525, "reward_std": 0.5429290533065796, "rewards/code_complexity_reward/mean": 0.8175780773162842, "rewards/code_complexity_reward/std": 0.1449529081583023, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 177, "step_time": 56.50905712507665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 215.19921875, "completions/mean_terminated_length": 213.4499053955078, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.22798765334300697, "epoch": 0.20296465222348917, "frac_reward_zero_std": 0.1875, "grad_norm": 0.032448507845401764, "kl": 0.01886511359771248, "learning_rate": 4.844659015773468e-06, "loss": 9.444072929909453e-05, "num_tokens": 37775416.0, "reward": 2.2577147483825684, "reward_std": 0.5340683460235596, "rewards/code_complexity_reward/mean": 0.8208984136581421, "rewards/code_complexity_reward/std": 0.13614331185817719, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.021983280777931213, "step": 178, "step_time": 57.8787400200963 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 230.361328125, "completions/mean_terminated_length": 226.45742797851562, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.229193045059219, "epoch": 0.20410490307867732, "frac_reward_zero_std": 0.078125, "grad_norm": 0.034433092921972275, "kl": 0.023657366036786698, "learning_rate": 4.841186238331519e-06, "loss": 0.00011824601097032428, "num_tokens": 37962917.0, "reward": 2.2428712844848633, "reward_std": 0.5651753544807434, "rewards/code_complexity_reward/mean": 0.80419921875, "rewards/code_complexity_reward/std": 0.15329690277576447, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.02048126794397831, "step": 179, "step_time": 71.19289907813072 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 227.484375, "completions/mean_terminated_length": 223.5406036376953, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22732703387737274, "epoch": 0.20524515393386544, "frac_reward_zero_std": 0.09375, "grad_norm": 0.03832489252090454, "kl": 0.01811535229353467, "learning_rate": 4.8376763431835424e-06, "loss": 9.057654824573547e-05, "num_tokens": 38150253.0, "reward": 2.2515134811401367, "reward_std": 0.585307240486145, "rewards/code_complexity_reward/mean": 0.8047851324081421, "rewards/code_complexity_reward/std": 0.1675819456577301, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.026382790878415108, "step": 180, "step_time": 71.4945640815422 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 212.470703125, "completions/mean_terminated_length": 211.2960968017578, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.242474329425022, "epoch": 0.2063854047890536, "frac_reward_zero_std": 0.109375, "grad_norm": 0.038104940205812454, "kl": 0.01848360290750861, "learning_rate": 4.834129385976227e-06, "loss": 9.235413745045662e-05, "num_tokens": 38326438.0, "reward": 2.27880859375, "reward_std": 0.5295593738555908, "rewards/code_complexity_reward/mean": 0.83837890625, "rewards/code_complexity_reward/std": 0.12424609065055847, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 181, "step_time": 61.711217171512544 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 214.48046875, "completions/mean_terminated_length": 212.7269287109375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.22957795578986406, "epoch": 0.20752565564424175, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.0376751683652401, "kl": 0.0183871381013887, "learning_rate": 4.830545422943847e-06, "loss": 9.186401439365e-05, "num_tokens": 38502344.0, "reward": 2.274951219558716, "reward_std": 0.538174569606781, "rewards/code_complexity_reward/mean": 0.8236328363418579, "rewards/code_complexity_reward/std": 0.13155730068683624, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02960825525224209, "step": 182, "step_time": 85.59663418494165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 223.01171875, "completions/mean_terminated_length": 221.30845642089844, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2323534672614187, "epoch": 0.20866590649942987, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.04052748903632164, "kl": 0.018107669602613896, "learning_rate": 4.8269245109073795e-06, "loss": 9.05353226698935e-05, "num_tokens": 38684590.0, "reward": 2.2803711891174316, "reward_std": 0.5354853272438049, "rewards/code_complexity_reward/mean": 0.8182617425918579, "rewards/code_complexity_reward/std": 0.12707574665546417, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 183, "step_time": 68.53842923976481 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 222.365234375, "completions/mean_terminated_length": 219.50888061523438, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2314030339475721, "epoch": 0.20980615735461802, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.03435293585062027, "kl": 0.022622255215537734, "learning_rate": 4.823266707273596e-06, "loss": 0.00011295190779492259, "num_tokens": 38867861.0, "reward": 2.2579588890075684, "reward_std": 0.5688066482543945, "rewards/code_complexity_reward/mean": 0.8182617425918579, "rewards/code_complexity_reward/std": 0.15232543647289276, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.026382790878415108, "step": 184, "step_time": 60.19242612738162 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 220.609375, "completions/mean_terminated_length": 218.89195251464844, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.226814117282629, "epoch": 0.21094640820980615, "frac_reward_zero_std": 0.109375, "grad_norm": 0.03249761462211609, "kl": 0.019330273906234652, "learning_rate": 4.819572070034162e-06, "loss": 9.65806539170444e-05, "num_tokens": 39049577.0, "reward": 2.275341749191284, "reward_std": 0.550714373588562, "rewards/code_complexity_reward/mean": 0.81640625, "rewards/code_complexity_reward/std": 0.14091667532920837, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.022693097591400146, "step": 185, "step_time": 59.6764058386907 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 209.1796875, "completions/mean_terminated_length": 208.5870819091797, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.23827095772139728, "epoch": 0.2120866590649943, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.0378442257642746, "kl": 0.021244356932584196, "learning_rate": 4.815840657764704e-06, "loss": 0.00010617317457217723, "num_tokens": 39225657.0, "reward": 2.2774901390075684, "reward_std": 0.5308855175971985, "rewards/code_complexity_reward/mean": 0.8310546875, "rewards/code_complexity_reward/std": 0.12395338714122772, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 186, "step_time": 60.87364842556417 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 217.97265625, "completions/mean_terminated_length": 217.3972625732422, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.23113767616450787, "epoch": 0.21322690992018245, "frac_reward_zero_std": 0.125, "grad_norm": 0.035749778151512146, "kl": 0.019716519906069152, "learning_rate": 4.812072529623894e-06, "loss": 9.860606223810464e-05, "num_tokens": 39403939.0, "reward": 2.2179198265075684, "reward_std": 0.5322545766830444, "rewards/code_complexity_reward/mean": 0.8173828125, "rewards/code_complexity_reward/std": 0.12995778024196625, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.025244520977139473, "step": 187, "step_time": 76.63637692015618 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 224.802734375, "completions/mean_terminated_length": 221.9704132080078, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2353772297501564, "epoch": 0.21436716077537057, "frac_reward_zero_std": 0.078125, "grad_norm": 0.033190056681632996, "kl": 0.018661645968677476, "learning_rate": 4.808267745352502e-06, "loss": 9.329491876997054e-05, "num_tokens": 39587222.0, "reward": 2.2123045921325684, "reward_std": 0.5107527375221252, "rewards/code_complexity_reward/mean": 0.808789074420929, "rewards/code_complexity_reward/std": 0.13848812878131866, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 188, "step_time": 59.52973727323115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 214.29296875, "completions/mean_terminated_length": 213.7103729248047, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2299796703737229, "epoch": 0.21550741163055873, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.03683067858219147, "kl": 0.02026335705886595, "learning_rate": 4.804426365272455e-06, "loss": 0.00010126392589882016, "num_tokens": 39765816.0, "reward": 2.294628858566284, "reward_std": 0.5245424509048462, "rewards/code_complexity_reward/mean": 0.832812488079071, "rewards/code_complexity_reward/std": 0.10734976083040237, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 189, "step_time": 76.03002768103033 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 218.966796875, "completions/mean_terminated_length": 216.65945434570312, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.2333189006894827, "epoch": 0.21664766248574688, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.03596603125333786, "kl": 0.020331192630692385, "learning_rate": 4.800548450285878e-06, "loss": 0.00010160945384996012, "num_tokens": 39947467.0, "reward": 2.286816358566284, "reward_std": 0.5706706047058105, "rewards/code_complexity_reward/mean": 0.816699206829071, "rewards/code_complexity_reward/std": 0.15593114495277405, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 190, "step_time": 59.4872083440423 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 221.494140625, "completions/mean_terminated_length": 218.6291961669922, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.22630588337779045, "epoch": 0.217787913340935, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.03367050364613533, "kl": 0.020804526197025552, "learning_rate": 4.79663406187413e-06, "loss": 0.00010410259710624814, "num_tokens": 40129396.0, "reward": 2.2568359375, "reward_std": 0.5705077648162842, "rewards/code_complexity_reward/mean": 0.803027331829071, "rewards/code_complexity_reward/std": 0.15924113988876343, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03206123411655426, "step": 191, "step_time": 61.79894382227212 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 218.671875, "completions/mean_terminated_length": 215.7790985107422, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23267662292346358, "epoch": 0.21892816419612315, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.042886633425951004, "kl": 0.020816737262066454, "learning_rate": 4.792683262096825e-06, "loss": 0.00010389916133135557, "num_tokens": 40310012.0, "reward": 2.2362794876098633, "reward_std": 0.5411862134933472, "rewards/code_complexity_reward/mean": 0.8264648914337158, "rewards/code_complexity_reward/std": 0.1461769938468933, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.014529787935316563, "step": 192, "step_time": 70.94756127241999 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 211.9765625, "completions/mean_terminated_length": 210.208251953125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.23325867857784033, "epoch": 0.22006841505131128, "frac_reward_zero_std": 0.109375, "grad_norm": 0.03495795652270317, "kl": 0.021539896144531667, "learning_rate": 4.788696113590853e-06, "loss": 0.00010751622176030651, "num_tokens": 40488752.0, "reward": 2.2652831077575684, "reward_std": 0.5252034068107605, "rewards/code_complexity_reward/mean": 0.826171875, "rewards/code_complexity_reward/std": 0.12057669460773468, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.019864002242684364, "step": 193, "step_time": 69.32776970788836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 214.8671875, "completions/mean_terminated_length": 213.7019805908203, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23321853601373732, "epoch": 0.22120866590649943, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.035626545548439026, "kl": 0.022754153877031058, "learning_rate": 4.7846726795693855e-06, "loss": 0.0001137083163484931, "num_tokens": 40668880.0, "reward": 2.3198728561401367, "reward_std": 0.5445125699043274, "rewards/code_complexity_reward/mean": 0.823535144329071, "rewards/code_complexity_reward/std": 0.12858480215072632, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.025197163224220276, "step": 194, "step_time": 60.830832998268306 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 213.765625, "completions/mean_terminated_length": 210.824462890625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.23163259448483586, "epoch": 0.22234891676168758, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.0434439592063427, "kl": 0.02388784842332825, "learning_rate": 4.780613023820872e-06, "loss": 0.0001193916323245503, "num_tokens": 40846656.0, "reward": 2.1951169967651367, "reward_std": 0.5306580066680908, "rewards/code_complexity_reward/mean": 0.8125976920127869, "rewards/code_complexity_reward/std": 0.15858502686023712, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.020545316860079765, "step": 195, "step_time": 60.76048994716257 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 226.037109375, "completions/mean_terminated_length": 223.78543090820312, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2352539519779384, "epoch": 0.2234891676168757, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.04705392196774483, "kl": 0.030102136995992623, "learning_rate": 4.776517210708032e-06, "loss": 0.00015037565026432276, "num_tokens": 41032679.0, "reward": 2.230224609375, "reward_std": 0.5509694218635559, "rewards/code_complexity_reward/mean": 0.8023437261581421, "rewards/code_complexity_reward/std": 0.15379852056503296, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 196, "step_time": 60.06983934249729 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 212.263671875, "completions/mean_terminated_length": 211.08824157714844, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2330706703942269, "epoch": 0.22462941847206386, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.03772176429629326, "kl": 0.02328692510491237, "learning_rate": 4.772385305166828e-06, "loss": 0.00011645213817246258, "num_tokens": 41210882.0, "reward": 2.309521436691284, "reward_std": 0.5538655519485474, "rewards/code_complexity_reward/mean": 0.82080078125, "rewards/code_complexity_reward/std": 0.12379336357116699, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 197, "step_time": 61.35083664301783 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 205.67578125, "completions/mean_terminated_length": 203.26377868652344, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.22802978032268584, "epoch": 0.22576966932725198, "frac_reward_zero_std": 0.125, "grad_norm": 0.042607005685567856, "kl": 0.026483860579901375, "learning_rate": 4.768217372705442e-06, "loss": 0.0001323914621025324, "num_tokens": 41385032.0, "reward": 2.336181640625, "reward_std": 0.5453218817710876, "rewards/code_complexity_reward/mean": 0.83056640625, "rewards/code_complexity_reward/std": 0.1294965147972107, "rewards/code_execution_reward/mean": 0.4140625, "rewards/code_execution_reward/std": 0.49304109811782837, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.026328405365347862, "step": 198, "step_time": 58.8287007054314 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 207.275390625, "completions/mean_terminated_length": 206.67906188964844, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22784129693172872, "epoch": 0.22690992018244013, "frac_reward_zero_std": 0.125, "grad_norm": 0.036647163331508636, "kl": 0.026057390205096453, "learning_rate": 4.764013479403239e-06, "loss": 0.0001304536999668926, "num_tokens": 41560165.0, "reward": 2.3232421875, "reward_std": 0.5376839637756348, "rewards/code_complexity_reward/mean": 0.8267577886581421, "rewards/code_complexity_reward/std": 0.11715326458215714, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 199, "step_time": 57.66938034724444 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 482.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 201.068359375, "completions/mean_terminated_length": 201.068359375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23759997892193496, "epoch": 0.22805017103762829, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.036619797348976135, "kl": 0.02828551546554081, "learning_rate": 4.759773691909708e-06, "loss": 0.00014145506429485977, "num_tokens": 41730572.0, "reward": 2.322558641433716, "reward_std": 0.5395455956459045, "rewards/code_complexity_reward/mean": 0.8389648199081421, "rewards/code_complexity_reward/std": 0.11252440512180328, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 200, "step_time": 67.60555088426918 }, { "epoch": 0.22805017103762829, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.005, "eval_completions/max_length": 319.48, "eval_completions/max_terminated_length": 314.04, "eval_completions/mean_length": 211.3925, "eval_completions/mean_terminated_length": 210.08250030517578, "eval_completions/min_length": 127.76, "eval_completions/min_terminated_length": 127.76, "eval_entropy": 0.23059711426496507, "eval_frac_reward_zero_std": 0.08, "eval_kl": 0.02595198791474104, "eval_loss": 0.00012997922021895647, "eval_num_tokens": 41730572.0, "eval_reward": 2.2301875066757204, "eval_reward_std": 0.4365191526710987, "eval_rewards/code_complexity_reward/mean": 0.8226250004768372, "eval_rewards/code_complexity_reward/std": 0.09869407951831817, "eval_rewards/code_execution_reward/mean": 0.315, "eval_rewards/code_execution_reward/std": 0.36563742280006406, "eval_rewards/code_syntax_reward/mean": 0.49375, "eval_rewards/code_syntax_reward/std": 0.01767766922712326, "eval_rewards/reasoning_present_reward_func/mean": 0.09975000157952309, "eval_rewards/reasoning_present_reward_func/std": 0.000707106813788414, "eval_rewards/xmlcount_reward_func/mean": 0.4990625, "eval_rewards/xmlcount_reward_func/std": 0.002651650384068489, "eval_runtime": 739.5624, "eval_samples_per_second": 0.135, "eval_steps_per_second": 0.018, "step": 200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 211.8046875, "completions/mean_terminated_length": 211.21722412109375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23785272054374218, "epoch": 0.2291904218928164, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.03583535552024841, "kl": 0.027811395324533805, "learning_rate": 4.755498077443419e-06, "loss": 0.00013882704661227763, "num_tokens": 41908292.0, "reward": 2.2037110328674316, "reward_std": 0.5126180648803711, "rewards/code_complexity_reward/mean": 0.8190429210662842, "rewards/code_complexity_reward/std": 0.12917645275592804, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 201, "step_time": 99.7959735635668 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 212.943359375, "completions/mean_terminated_length": 211.18075561523438, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.23902156599797308, "epoch": 0.23033067274800456, "frac_reward_zero_std": 0.0625, "grad_norm": 0.043039076030254364, "kl": 0.02704663840995636, "learning_rate": 4.7511867037909484e-06, "loss": 0.00013539381325244904, "num_tokens": 42086287.0, "reward": 2.263671875, "reward_std": 0.5570212006568909, "rewards/code_complexity_reward/mean": 0.8140624761581421, "rewards/code_complexity_reward/std": 0.14378003776073456, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 202, "step_time": 67.11243926081806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 211.150390625, "completions/mean_terminated_length": 210.5616455078125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.23317479016259313, "epoch": 0.2314709236031927, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.036075521260499954, "kl": 0.0290682567137992, "learning_rate": 4.746839639305808e-06, "loss": 0.00014521364937536418, "num_tokens": 42263788.0, "reward": 2.2977051734924316, "reward_std": 0.5386580228805542, "rewards/code_complexity_reward/mean": 0.8297851085662842, "rewards/code_complexity_reward/std": 0.123296819627285, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 203, "step_time": 70.29117907676846 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 200.90625, "completions/mean_terminated_length": 198.45669555664062, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2272886224091053, "epoch": 0.23261117445838084, "frac_reward_zero_std": 0.15625, "grad_norm": 0.04092281684279442, "kl": 0.0311253470426891, "learning_rate": 4.742456952907358e-06, "loss": 0.0001556060160510242, "num_tokens": 42435152.0, "reward": 2.324462890625, "reward_std": 0.5822626948356628, "rewards/code_complexity_reward/mean": 0.8250000476837158, "rewards/code_complexity_reward/std": 0.1606919914484024, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.022640403360128403, "step": 204, "step_time": 77.69877587538213 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 195.8828125, "completions/mean_terminated_length": 195.26419067382812, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23741767462342978, "epoch": 0.233751425313569, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.037869393825531006, "kl": 0.028658105031354353, "learning_rate": 4.73803871407972e-06, "loss": 0.00014322035713121295, "num_tokens": 42602772.0, "reward": 2.2423338890075684, "reward_std": 0.503013014793396, "rewards/code_complexity_reward/mean": 0.841015636920929, "rewards/code_complexity_reward/std": 0.11598022282123566, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.024002734571695328, "step": 205, "step_time": 57.22325189691037 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 209.216796875, "completions/mean_terminated_length": 208.624267578125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23783527524210513, "epoch": 0.2348916761687571, "frac_reward_zero_std": 0.15625, "grad_norm": 0.044152308255434036, "kl": 0.029140953221940435, "learning_rate": 4.733584992870669e-06, "loss": 0.00014566653408110142, "num_tokens": 42777499.0, "reward": 2.249267578125, "reward_std": 0.5179325342178345, "rewards/code_complexity_reward/mean": 0.826367199420929, "rewards/code_complexity_reward/std": 0.11898167431354523, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.027517380192875862, "step": 206, "step_time": 57.2258691303432 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 503.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 200.0234375, "completions/mean_terminated_length": 200.0234375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.23073120904155076, "epoch": 0.23603192702394526, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.03451969474554062, "kl": 0.033970679563935846, "learning_rate": 4.729095859890529e-06, "loss": 0.00016972383309621364, "num_tokens": 42947375.0, "reward": 2.331738233566284, "reward_std": 0.5533773899078369, "rewards/code_complexity_reward/mean": 0.836230456829071, "rewards/code_complexity_reward/std": 0.12430176138877869, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 207, "step_time": 58.68887387868017 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 506.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 201.5078125, "completions/mean_terminated_length": 201.5078125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2220796279143542, "epoch": 0.23717217787913342, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.03937068209052086, "kl": 0.03450481776962988, "learning_rate": 4.724571386311046e-06, "loss": 0.0001725789625197649, "num_tokens": 43118095.0, "reward": 2.299316644668579, "reward_std": 0.5472753643989563, "rewards/code_complexity_reward/mean": 0.8243163824081421, "rewards/code_complexity_reward/std": 0.13362886011600494, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 208, "step_time": 74.00884594302624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 204.435546875, "completions/mean_terminated_length": 203.22943115234375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23076316132210195, "epoch": 0.23831242873432154, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.03879963606595993, "kl": 0.037612792395520955, "learning_rate": 4.720011643864268e-06, "loss": 0.00018801106489263475, "num_tokens": 43290598.0, "reward": 2.2828125953674316, "reward_std": 0.516496479511261, "rewards/code_complexity_reward/mean": 0.8346679210662842, "rewards/code_complexity_reward/std": 0.11340783536434174, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 209, "step_time": 56.91263633593917 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 193.52734375, "completions/mean_terminated_length": 191.65029907226562, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23570575006306171, "epoch": 0.2394526795895097, "frac_reward_zero_std": 0.203125, "grad_norm": 0.0413094237446785, "kl": 0.031596134198480286, "learning_rate": 4.715416704841404e-06, "loss": 0.00015804677968844771, "num_tokens": 43459240.0, "reward": 2.239208936691284, "reward_std": 0.5350640416145325, "rewards/code_complexity_reward/mean": 0.837890625, "rewards/code_complexity_reward/std": 0.14514262974262238, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 210, "step_time": 60.86115307826549 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 213.521484375, "completions/mean_terminated_length": 211.1712646484375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2342066988348961, "epoch": 0.24059293044469784, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.037212517112493515, "kl": 0.030744784817215987, "learning_rate": 4.710786642091673e-06, "loss": 0.00015363175771199167, "num_tokens": 43636187.0, "reward": 2.2633790969848633, "reward_std": 0.5492337942123413, "rewards/code_complexity_reward/mean": 0.8251953125, "rewards/code_complexity_reward/std": 0.13566520810127258, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 211, "step_time": 76.75793597660959 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 197.591796875, "completions/mean_terminated_length": 195.73870849609375, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.23786866082809865, "epoch": 0.24173318129988597, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.04189534857869148, "kl": 0.03164388450386468, "learning_rate": 4.706121529021158e-06, "loss": 0.00015821994747966528, "num_tokens": 43807310.0, "reward": 2.2112793922424316, "reward_std": 0.5402566194534302, "rewards/code_complexity_reward/mean": 0.820507824420929, "rewards/code_complexity_reward/std": 0.15346403419971466, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.021303100511431694, "step": 212, "step_time": 68.19994723238051 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 198.00390625, "completions/mean_terminated_length": 197.38943481445312, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23368645505979657, "epoch": 0.24287343215507412, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.037429966032505035, "kl": 0.045510528958402574, "learning_rate": 4.7014214395916355e-06, "loss": 0.00022742462169844657, "num_tokens": 43977832.0, "reward": 2.2956056594848633, "reward_std": 0.5234218239784241, "rewards/code_complexity_reward/mean": 0.838671863079071, "rewards/code_complexity_reward/std": 0.11076408624649048, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 213, "step_time": 49.18100745882839 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 204.51171875, "completions/mean_terminated_length": 203.30589294433594, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23783705825917423, "epoch": 0.24401368301026224, "frac_reward_zero_std": 0.109375, "grad_norm": 0.045366182923316956, "kl": 0.03733677105628885, "learning_rate": 4.696686448319408e-06, "loss": 0.00018665591778699309, "num_tokens": 44152438.0, "reward": 2.2398438453674316, "reward_std": 0.5369464755058289, "rewards/code_complexity_reward/mean": 0.8189452886581421, "rewards/code_complexity_reward/std": 0.14434026181697845, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.022032126784324646, "step": 214, "step_time": 76.10972288064659 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 195.791015625, "completions/mean_terminated_length": 194.55099487304688, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2302374462597072, "epoch": 0.2451539338654504, "frac_reward_zero_std": 0.1875, "grad_norm": 0.03717475011944771, "kl": 0.03411706618499011, "learning_rate": 4.691916630274117e-06, "loss": 0.0001706102048046887, "num_tokens": 44319071.0, "reward": 2.2816896438598633, "reward_std": 0.5709884762763977, "rewards/code_complexity_reward/mean": 0.8147460222244263, "rewards/code_complexity_reward/std": 0.15901584923267365, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 215, "step_time": 65.48106732126325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 196.923828125, "completions/mean_terminated_length": 196.30723571777344, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2428469150327146, "epoch": 0.24629418472063855, "frac_reward_zero_std": 0.109375, "grad_norm": 0.04119021072983742, "kl": 0.03829990077065304, "learning_rate": 4.687112061077556e-06, "loss": 0.00019149252329953015, "num_tokens": 44487820.0, "reward": 2.27880859375, "reward_std": 0.5341128706932068, "rewards/code_complexity_reward/mean": 0.8264648914337158, "rewards/code_complexity_reward/std": 0.12377699464559555, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 216, "step_time": 66.12473729252815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 197.208984375, "completions/mean_terminated_length": 195.3536376953125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23761512339115143, "epoch": 0.24743443557582667, "frac_reward_zero_std": 0.109375, "grad_norm": 0.04635133966803551, "kl": 0.03521931279101409, "learning_rate": 4.6822728169024735e-06, "loss": 0.0001760537998052314, "num_tokens": 44660307.0, "reward": 2.2748537063598633, "reward_std": 0.5589196085929871, "rewards/code_complexity_reward/mean": 0.828125, "rewards/code_complexity_reward/std": 0.14926433563232422, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 217, "step_time": 59.92157367058098 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 186.58203125, "completions/mean_terminated_length": 185.30589294433594, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23166302568279207, "epoch": 0.24857468643101482, "frac_reward_zero_std": 0.125, "grad_norm": 0.04211258515715599, "kl": 0.04168310045497492, "learning_rate": 4.67739897447136e-06, "loss": 0.00020850409055128694, "num_tokens": 44824513.0, "reward": 2.3174805641174316, "reward_std": 0.5512635111808777, "rewards/code_complexity_reward/mean": 0.8395507335662842, "rewards/code_complexity_reward/std": 0.1366111785173416, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 218, "step_time": 68.30889949481934 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 197.03125, "completions/mean_terminated_length": 195.17486572265625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23420076444745064, "epoch": 0.24971493728620298, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.042487286031246185, "kl": 0.04084058667649515, "learning_rate": 4.672490611055238e-06, "loss": 0.000204132724320516, "num_tokens": 44995849.0, "reward": 2.228271484375, "reward_std": 0.5460745692253113, "rewards/code_complexity_reward/mean": 0.8162109851837158, "rewards/code_complexity_reward/std": 0.15008443593978882, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 219, "step_time": 78.76008952502161 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 195.189453125, "completions/mean_terminated_length": 193.32220458984375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2380617205053568, "epoch": 0.2508551881413911, "frac_reward_zero_std": 0.15625, "grad_norm": 0.042174600064754486, "kl": 0.04403989436104894, "learning_rate": 4.667547804472431e-06, "loss": 0.0002201078605139628, "num_tokens": 45164578.0, "reward": 2.287402391433716, "reward_std": 0.5328081846237183, "rewards/code_complexity_reward/mean": 0.8343750238418579, "rewards/code_complexity_reward/std": 0.12390419095754623, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 220, "step_time": 66.65109257772565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 198.029296875, "completions/mean_terminated_length": 196.1787872314453, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2306073484942317, "epoch": 0.2519954389965792, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.0377323180437088, "kl": 0.037250170891638845, "learning_rate": 4.662570633087333e-06, "loss": 0.00018629920668900013, "num_tokens": 45332909.0, "reward": 2.275830030441284, "reward_std": 0.5206198692321777, "rewards/code_complexity_reward/mean": 0.834277331829071, "rewards/code_complexity_reward/std": 0.12119699269533157, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.022693097591400146, "step": 221, "step_time": 67.35092446394265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 187.18359375, "completions/mean_terminated_length": 186.54794311523438, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23655583895742893, "epoch": 0.2531356898517674, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.041517920792102814, "kl": 0.043362150725442916, "learning_rate": 4.657559175809168e-06, "loss": 0.00021669379202648997, "num_tokens": 45497819.0, "reward": 2.276611328125, "reward_std": 0.5336142778396606, "rewards/code_complexity_reward/mean": 0.8316406011581421, "rewards/code_complexity_reward/std": 0.12157174199819565, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 222, "step_time": 58.36354533303529 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 186.921875, "completions/mean_terminated_length": 186.28570556640625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2338154213503003, "epoch": 0.2542759407069555, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.050085995346307755, "kl": 0.04080951045034453, "learning_rate": 4.6525135120907314e-06, "loss": 0.00020383152877911925, "num_tokens": 45662991.0, "reward": 2.300048828125, "reward_std": 0.5315245389938354, "rewards/code_complexity_reward/mean": 0.8418945074081421, "rewards/code_complexity_reward/std": 0.1188979223370552, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 223, "step_time": 71.97516251727939 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 189.607421875, "completions/mean_terminated_length": 187.707275390625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23369509959593415, "epoch": 0.2554161915621437, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.046487271785736084, "kl": 0.04183450271375477, "learning_rate": 4.647433721927139e-06, "loss": 0.00020908871374558657, "num_tokens": 45828066.0, "reward": 2.261914014816284, "reward_std": 0.5316243171691895, "rewards/code_complexity_reward/mean": 0.8357422351837158, "rewards/code_complexity_reward/std": 0.13223887979984283, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 224, "step_time": 56.93646091129631 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 193.375, "completions/mean_terminated_length": 192.75146484375, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2309370213188231, "epoch": 0.25655644241733183, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.04064740613102913, "kl": 0.03971458875457756, "learning_rate": 4.642319885854557e-06, "loss": 0.00019845081260427833, "num_tokens": 45993606.0, "reward": 2.31884765625, "reward_std": 0.5437095761299133, "rewards/code_complexity_reward/mean": 0.828906238079071, "rewards/code_complexity_reward/std": 0.12478699535131454, "rewards/code_execution_reward/mean": 0.396484375, "rewards/code_execution_reward/std": 0.4896455705165863, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 225, "step_time": 56.69329619128257 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 177.22265625, "completions/mean_terminated_length": 175.90982055664062, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23521008878014982, "epoch": 0.2576966932725199, "frac_reward_zero_std": 0.140625, "grad_norm": 0.04589587450027466, "kl": 0.048746186977950856, "learning_rate": 4.637172084948917e-06, "loss": 0.00024363654665648937, "num_tokens": 46151632.0, "reward": 2.323046922683716, "reward_std": 0.5338916182518005, "rewards/code_complexity_reward/mean": 0.851855456829071, "rewards/code_complexity_reward/std": 0.11455466598272324, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.026872845366597176, "step": 226, "step_time": 51.27671730238944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 183.728515625, "completions/mean_terminated_length": 183.0861053466797, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23985323985107243, "epoch": 0.2588369441277081, "frac_reward_zero_std": 0.21875, "grad_norm": 0.043313056230545044, "kl": 0.04214111901819706, "learning_rate": 4.631990400824643e-06, "loss": 0.0002105945022776723, "num_tokens": 46314453.0, "reward": 2.2838377952575684, "reward_std": 0.5313473343849182, "rewards/code_complexity_reward/mean": 0.8349609375, "rewards/code_complexity_reward/std": 0.12310559302568436, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 227, "step_time": 74.8421209603548 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 190.49609375, "completions/mean_terminated_length": 189.2353057861328, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.240688614314422, "epoch": 0.25997719498289623, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.044013410806655884, "kl": 0.043754950631409883, "learning_rate": 4.626774915633349e-06, "loss": 0.00021880415442865342, "num_tokens": 46480991.0, "reward": 2.217090129852295, "reward_std": 0.5469658374786377, "rewards/code_complexity_reward/mean": 0.8255859613418579, "rewards/code_complexity_reward/std": 0.15758588910102844, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 228, "step_time": 60.877811565063894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 185.294921875, "completions/mean_terminated_length": 184.65557861328125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2421895524021238, "epoch": 0.2611174458380844, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.04380982369184494, "kl": 0.04786631546448916, "learning_rate": 4.621525712062537e-06, "loss": 0.00023920706007629633, "num_tokens": 46645750.0, "reward": 2.2772462368011475, "reward_std": 0.5322315096855164, "rewards/code_complexity_reward/mean": 0.8447265625, "rewards/code_complexity_reward/std": 0.1265665888786316, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 229, "step_time": 78.7948769601062 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 167.083984375, "completions/mean_terminated_length": 165.73138427734375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2461465378291905, "epoch": 0.26225769669327254, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.047584764659404755, "kl": 0.052171258168527856, "learning_rate": 4.616242873334292e-06, "loss": 0.0002607781789265573, "num_tokens": 46800577.0, "reward": 2.3816895484924316, "reward_std": 0.5515992045402527, "rewards/code_complexity_reward/mean": 0.8587890863418579, "rewards/code_complexity_reward/std": 0.1288246363401413, "rewards/code_execution_reward/mean": 0.431640625, "rewards/code_execution_reward/std": 0.4957893490791321, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.021246958523988724, "step": 230, "step_time": 49.94130008388311 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 180.4609375, "completions/mean_terminated_length": 179.8121337890625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23520631692372262, "epoch": 0.2633979475484607, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.051082249730825424, "kl": 0.07565359977888875, "learning_rate": 4.610926483203954e-06, "loss": 0.0003778087848331779, "num_tokens": 46959961.0, "reward": 2.3560545444488525, "reward_std": 0.5388434529304504, "rewards/code_complexity_reward/mean": 0.8483397960662842, "rewards/code_complexity_reward/std": 0.11604277789592743, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 231, "step_time": 58.19391745142639 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 178.396484375, "completions/mean_terminated_length": 177.08824157714844, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23847284680232406, "epoch": 0.2645381984036488, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.04413970932364464, "kl": 0.04799745965283364, "learning_rate": 4.6055766259588004e-06, "loss": 0.0002399940276518464, "num_tokens": 47118964.0, "reward": 2.30712890625, "reward_std": 0.5374096632003784, "rewards/code_complexity_reward/mean": 0.839160144329071, "rewards/code_complexity_reward/std": 0.12348762899637222, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.024554960429668427, "step": 232, "step_time": 66.28471945691854 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 181.60546875, "completions/mean_terminated_length": 180.9589080810547, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23284510127268732, "epoch": 0.26567844925883694, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.05391375720500946, "kl": 0.047265800851164386, "learning_rate": 4.600193386416697e-06, "loss": 0.00023612676886841655, "num_tokens": 47282978.0, "reward": 2.3037595748901367, "reward_std": 0.5471968650817871, "rewards/code_complexity_reward/mean": 0.8335937261581421, "rewards/code_complexity_reward/std": 0.1448880136013031, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 233, "step_time": 65.8684801235795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 180.890625, "completions/mean_terminated_length": 178.93910217285156, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23989270394667983, "epoch": 0.2668187001140251, "frac_reward_zero_std": 0.171875, "grad_norm": 0.04853050038218498, "kl": 0.056854036112781614, "learning_rate": 4.594776849924766e-06, "loss": 0.00028397407731972635, "num_tokens": 47447342.0, "reward": 2.281982421875, "reward_std": 0.5405081510543823, "rewards/code_complexity_reward/mean": 0.840136706829071, "rewards/code_complexity_reward/std": 0.13611674308776855, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.024042511358857155, "step": 234, "step_time": 59.65402271412313 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 181.4453125, "completions/mean_terminated_length": 179.49705505371094, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.23466755892150104, "epoch": 0.26795895096921324, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.04226365685462952, "kl": 0.044561543327290565, "learning_rate": 4.589327102358024e-06, "loss": 0.0002228060329798609, "num_tokens": 47607942.0, "reward": 2.2987794876098633, "reward_std": 0.5524560809135437, "rewards/code_complexity_reward/mean": 0.8388671875, "rewards/code_complexity_reward/std": 0.14139743149280548, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.021303100511431694, "step": 235, "step_time": 57.881591581739485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 447.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 179.462890625, "completions/mean_terminated_length": 179.462890625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24214614531956613, "epoch": 0.2690992018244014, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.15927723050117493, "kl": 0.04938998309080489, "learning_rate": 4.5838442301180245e-06, "loss": 0.00024692100123502314, "num_tokens": 47768611.0, "reward": 2.362060546875, "reward_std": 0.5244811773300171, "rewards/code_complexity_reward/mean": 0.8575195074081421, "rewards/code_complexity_reward/std": 0.09660123288631439, "rewards/code_execution_reward/mean": 0.40625, "rewards/code_execution_reward/std": 0.49161264300346375, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 236, "step_time": 44.319613698869944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 171.88671875, "completions/mean_terminated_length": 171.22113037109375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2468171964865178, "epoch": 0.2702394526795895, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.043096210807561874, "kl": 0.052329408936202526, "learning_rate": 4.5783283201314876e-06, "loss": 0.0002613875549286604, "num_tokens": 47924413.0, "reward": 2.2750978469848633, "reward_std": 0.5311965346336365, "rewards/code_complexity_reward/mean": 0.85498046875, "rewards/code_complexity_reward/std": 0.12843938171863556, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.030144967138767242, "step": 237, "step_time": 50.57674630731344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 192.14453125, "completions/mean_terminated_length": 191.51858520507812, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.24652767763473094, "epoch": 0.27137970353477764, "frac_reward_zero_std": 0.15625, "grad_norm": 0.04329901933670044, "kl": 0.044166253035655245, "learning_rate": 4.572779459848922e-06, "loss": 0.00022083320072852075, "num_tokens": 48092019.0, "reward": 2.1728515625, "reward_std": 0.5160703063011169, "rewards/code_complexity_reward/mean": 0.8155273199081421, "rewards/code_complexity_reward/std": 0.1566781997680664, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 238, "step_time": 59.842264810577035 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 182.791015625, "completions/mean_terminated_length": 182.1467742919922, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2369159199297428, "epoch": 0.2725199543899658, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.04553395137190819, "kl": 0.0833966436330229, "learning_rate": 4.5671977372432355e-06, "loss": 0.00041721720481291413, "num_tokens": 48253804.0, "reward": 2.2972168922424316, "reward_std": 0.5312848091125488, "rewards/code_complexity_reward/mean": 0.8441406488418579, "rewards/code_complexity_reward/std": 0.11714494973421097, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 239, "step_time": 49.966251730918884 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 183.0078125, "completions/mean_terminated_length": 182.36399841308594, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.23436674335971475, "epoch": 0.27366020524515394, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.045713379979133606, "kl": 0.04797527630580589, "learning_rate": 4.561583240808344e-06, "loss": 0.0002398307842668146, "num_tokens": 48414820.0, "reward": 2.288623332977295, "reward_std": 0.5249161124229431, "rewards/code_complexity_reward/mean": 0.8441405892372131, "rewards/code_complexity_reward/std": 0.12145096063613892, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 240, "step_time": 58.083670338615775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 469.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 186.095703125, "completions/mean_terminated_length": 186.095703125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2330819945782423, "epoch": 0.2748004561003421, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.058921147137880325, "kl": 0.04493933680350892, "learning_rate": 4.555936059557768e-06, "loss": 0.00022450453252531588, "num_tokens": 48579781.0, "reward": 2.290087938308716, "reward_std": 0.5327812433242798, "rewards/code_complexity_reward/mean": 0.8294922113418579, "rewards/code_complexity_reward/std": 0.12991543114185333, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 241, "step_time": 56.911069138906896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 492.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 170.306640625, "completions/mean_terminated_length": 170.306640625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24832015135325491, "epoch": 0.2759407069555302, "frac_reward_zero_std": 0.234375, "grad_norm": 0.041717611253261566, "kl": 0.05591752796317451, "learning_rate": 4.5502562830232225e-06, "loss": 0.00027950212825089693, "num_tokens": 48737886.0, "reward": 2.3008790016174316, "reward_std": 0.5165051817893982, "rewards/code_complexity_reward/mean": 0.8499999642372131, "rewards/code_complexity_reward/std": 0.1150597557425499, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.020638125017285347, "step": 242, "step_time": 65.31590379867703 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 177.943359375, "completions/mean_terminated_length": 177.943359375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24205771670676768, "epoch": 0.27708095781071834, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.047840166836977005, "kl": 0.04865769992466085, "learning_rate": 4.544544001253189e-06, "loss": 0.0002431773900752887, "num_tokens": 48899541.0, "reward": 2.2959961891174316, "reward_std": 0.5266188383102417, "rewards/code_complexity_reward/mean": 0.849316418170929, "rewards/code_complexity_reward/std": 0.1297842562198639, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 243, "step_time": 53.84770690090954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 487.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 177.548828125, "completions/mean_terminated_length": 177.548828125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.24072758294641972, "epoch": 0.2782212086659065, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.0511808842420578, "kl": 0.04952059857896529, "learning_rate": 4.538799304811503e-06, "loss": 0.00024743564426898956, "num_tokens": 49059146.0, "reward": 2.2855467796325684, "reward_std": 0.5165022611618042, "rewards/code_complexity_reward/mean": 0.85009765625, "rewards/code_complexity_reward/std": 0.10779314488172531, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 244, "step_time": 79.9970513433218 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 176.62109375, "completions/mean_terminated_length": 175.9647674560547, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.24136393563821912, "epoch": 0.27936145952109465, "frac_reward_zero_std": 0.171875, "grad_norm": 0.04554453492164612, "kl": 0.05921634641708806, "learning_rate": 4.533022284775903e-06, "loss": 0.0002961310092359781, "num_tokens": 49218968.0, "reward": 2.3238770961761475, "reward_std": 0.5321673154830933, "rewards/code_complexity_reward/mean": 0.84912109375, "rewards/code_complexity_reward/std": 0.10842317342758179, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 245, "step_time": 58.41779040545225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 165.375, "completions/mean_terminated_length": 164.69667053222656, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24515019543468952, "epoch": 0.2805017103762828, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.053368788212537766, "kl": 0.056023489858489484, "learning_rate": 4.527213032736596e-06, "loss": 0.0002799980575218797, "num_tokens": 49371848.0, "reward": 2.348095655441284, "reward_std": 0.5371840000152588, "rewards/code_complexity_reward/mean": 0.8544921875, "rewards/code_complexity_reward/std": 0.11164722591638565, "rewards/code_execution_reward/mean": 0.396484375, "rewards/code_execution_reward/std": 0.4896455705165863, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 246, "step_time": 56.57246588263661 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 175.787109375, "completions/mean_terminated_length": 174.46864318847656, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24591078725643456, "epoch": 0.28164196123147095, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.042095500975847244, "kl": 0.05103164448519237, "learning_rate": 4.521371640794802e-06, "loss": 0.00025514073786325753, "num_tokens": 49531291.0, "reward": 2.2890138626098633, "reward_std": 0.5301364660263062, "rewards/code_complexity_reward/mean": 0.8466796875, "rewards/code_complexity_reward/std": 0.12562450766563416, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.03160629794001579, "step": 247, "step_time": 80.4810008527711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 508.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 170.318359375, "completions/mean_terminated_length": 170.318359375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24286843207664788, "epoch": 0.28278221208665905, "frac_reward_zero_std": 0.171875, "grad_norm": 0.044400542974472046, "kl": 0.05114845480420627, "learning_rate": 4.5154982015612965e-06, "loss": 0.0002556513645686209, "num_tokens": 49686242.0, "reward": 2.2968263626098633, "reward_std": 0.5062002539634705, "rewards/code_complexity_reward/mean": 0.848925769329071, "rewards/code_complexity_reward/std": 0.1008467972278595, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 248, "step_time": 57.038485878147185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 171.5390625, "completions/mean_terminated_length": 170.872802734375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23651901143603027, "epoch": 0.2839224629418472, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.048233289271593094, "kl": 0.050308987381868064, "learning_rate": 4.509592808154936e-06, "loss": 0.000251420249696821, "num_tokens": 49840750.0, "reward": 2.3504881858825684, "reward_std": 0.5492520928382874, "rewards/code_complexity_reward/mean": 0.8463866710662842, "rewards/code_complexity_reward/std": 0.12695296108722687, "rewards/code_execution_reward/mean": 0.41015625, "rewards/code_execution_reward/std": 0.49234291911125183, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 249, "step_time": 75.2924263747409 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 510.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 172.388671875, "completions/mean_terminated_length": 172.388671875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2430347348563373, "epoch": 0.28506271379703535, "frac_reward_zero_std": 0.21875, "grad_norm": 0.039310671389102936, "kl": 0.055998808646108955, "learning_rate": 4.50365555420119e-06, "loss": 0.0002800488146021962, "num_tokens": 49997801.0, "reward": 2.2747559547424316, "reward_std": 0.507539689540863, "rewards/code_complexity_reward/mean": 0.8558593988418579, "rewards/code_complexity_reward/std": 0.1056840643286705, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 250, "step_time": 66.67052242159843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 174.30078125, "completions/mean_terminated_length": 172.9764862060547, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23304122756235301, "epoch": 0.2862029646522235, "frac_reward_zero_std": 0.1875, "grad_norm": 0.04262685403227806, "kl": 0.057304706919239834, "learning_rate": 4.497686533830648e-06, "loss": 0.00028656714130192995, "num_tokens": 50155531.0, "reward": 2.33740234375, "reward_std": 0.5420750975608826, "rewards/code_complexity_reward/mean": 0.8467773199081421, "rewards/code_complexity_reward/std": 0.125685453414917, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 251, "step_time": 67.68699688464403 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 174.39453125, "completions/mean_terminated_length": 174.39453125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24047417589463294, "epoch": 0.28734321550741165, "frac_reward_zero_std": 0.15625, "grad_norm": 0.04656779021024704, "kl": 0.053832661476917565, "learning_rate": 4.491685841677538e-06, "loss": 0.0002690237306524068, "num_tokens": 50313353.0, "reward": 2.2963380813598633, "reward_std": 0.5174050331115723, "rewards/code_complexity_reward/mean": 0.838183581829071, "rewards/code_complexity_reward/std": 0.11557742953300476, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 252, "step_time": 41.27367976028472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 177.822265625, "completions/mean_terminated_length": 177.1682891845703, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24023852916434407, "epoch": 0.28848346636259975, "frac_reward_zero_std": 0.171875, "grad_norm": 0.04301512986421585, "kl": 0.04880163649795577, "learning_rate": 4.485653572878213e-06, "loss": 0.00024393595231231302, "num_tokens": 50474298.0, "reward": 2.3051271438598633, "reward_std": 0.539793074131012, "rewards/code_complexity_reward/mean": 0.84228515625, "rewards/code_complexity_reward/std": 0.12007030844688416, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 253, "step_time": 87.6287597650662 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 182.8046875, "completions/mean_terminated_length": 182.16046142578125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23717236635275185, "epoch": 0.2896237172177879, "frac_reward_zero_std": 0.1875, "grad_norm": 0.045548126101493835, "kl": 0.065700929320883, "learning_rate": 4.4795898230696535e-06, "loss": 0.0003283400146756321, "num_tokens": 50636590.0, "reward": 2.254638671875, "reward_std": 0.5186933279037476, "rewards/code_complexity_reward/mean": 0.8379882574081421, "rewards/code_complexity_reward/std": 0.134095698595047, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.024042511358857155, "step": 254, "step_time": 56.097156565636396 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 160.306640625, "completions/mean_terminated_length": 159.61839294433594, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2431432786397636, "epoch": 0.29076396807297605, "frac_reward_zero_std": 0.1875, "grad_norm": 0.04350922629237175, "kl": 0.06873816909501329, "learning_rate": 4.473494688387945e-06, "loss": 0.0003436318365857005, "num_tokens": 50788143.0, "reward": 2.387011766433716, "reward_std": 0.5602232217788696, "rewards/code_complexity_reward/mean": 0.8638671636581421, "rewards/code_complexity_reward/std": 0.12637922167778015, "rewards/code_execution_reward/mean": 0.4296875, "rewards/code_execution_reward/std": 0.4955156147480011, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 255, "step_time": 57.360355980694294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 486.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 163.1796875, "completions/mean_terminated_length": 163.1796875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.24325177050195634, "epoch": 0.2919042189281642, "frac_reward_zero_std": 0.171875, "grad_norm": 0.053656525909900665, "kl": 0.06532431914820336, "learning_rate": 4.467368265466759e-06, "loss": 0.0003265274572186172, "num_tokens": 50939895.0, "reward": 2.3357911109924316, "reward_std": 0.5305833220481873, "rewards/code_complexity_reward/mean": 0.8587890863418579, "rewards/code_complexity_reward/std": 0.11261392384767532, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.03260336071252823, "step": 256, "step_time": 74.09317213669419 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 476.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 171.759765625, "completions/mean_terminated_length": 171.759765625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23929632757790387, "epoch": 0.29304446978335236, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.042334724217653275, "kl": 0.05447319883387536, "learning_rate": 4.461210651435814e-06, "loss": 0.0002723628713283688, "num_tokens": 51095368.0, "reward": 2.283203125, "reward_std": 0.5317797660827637, "rewards/code_complexity_reward/mean": 0.8482421636581421, "rewards/code_complexity_reward/std": 0.12234193086624146, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 257, "step_time": 54.09633438941091 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 168.615234375, "completions/mean_terminated_length": 166.5913543701172, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23972131125628948, "epoch": 0.29418472063854045, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.04349268227815628, "kl": 0.06026957312133163, "learning_rate": 4.4550219439193435e-06, "loss": 0.0003011459775734693, "num_tokens": 51249935.0, "reward": 2.2275390625, "reward_std": 0.519260585308075, "rewards/code_complexity_reward/mean": 0.8506836295127869, "rewards/code_complexity_reward/std": 0.143735870718956, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 258, "step_time": 67.08912645839155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 164.93359375, "completions/mean_terminated_length": 164.25439453125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24333992158062756, "epoch": 0.2953249714937286, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.04455144330859184, "kl": 0.0594581319601275, "learning_rate": 4.448802241034541e-06, "loss": 0.0002971042413264513, "num_tokens": 51400509.0, "reward": 2.3053712844848633, "reward_std": 0.5244538187980652, "rewards/code_complexity_reward/mean": 0.861132800579071, "rewards/code_complexity_reward/std": 0.12564153969287872, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 259, "step_time": 71.83811240736395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 450.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 163.0234375, "completions/mean_terminated_length": 163.0234375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.24397224048152566, "epoch": 0.29646522234891676, "frac_reward_zero_std": 0.234375, "grad_norm": 0.049175016582012177, "kl": 0.06247003626776859, "learning_rate": 4.4425516413900085e-06, "loss": 0.0003123024362139404, "num_tokens": 51552421.0, "reward": 2.331005811691284, "reward_std": 0.5305323600769043, "rewards/code_complexity_reward/mean": 0.857714831829071, "rewards/code_complexity_reward/std": 0.11751691997051239, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 260, "step_time": 53.78149614483118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 167.486328125, "completions/mean_terminated_length": 166.8121337890625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.238893672125414, "epoch": 0.2976054732041049, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.06017940491437912, "kl": 0.056805647851433605, "learning_rate": 4.4362702440841945e-06, "loss": 0.00028387928614392877, "num_tokens": 51706242.0, "reward": 2.3133301734924316, "reward_std": 0.5328924655914307, "rewards/code_complexity_reward/mean": 0.8485351800918579, "rewards/code_complexity_reward/std": 0.12264534085988998, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 261, "step_time": 49.14740792475641 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 173.42578125, "completions/mean_terminated_length": 172.09805297851562, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23790658498182893, "epoch": 0.29874572405929306, "frac_reward_zero_std": 0.1875, "grad_norm": 0.05158659815788269, "kl": 0.05598692153580487, "learning_rate": 4.429958148703818e-06, "loss": 0.0002797406050376594, "num_tokens": 51862924.0, "reward": 2.2918944358825684, "reward_std": 0.5277711153030396, "rewards/code_complexity_reward/mean": 0.84912109375, "rewards/code_complexity_reward/std": 0.12376278638839722, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 262, "step_time": 56.655423920601606 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 437.0, "completions/mean_length": 168.67578125, "completions/mean_terminated_length": 167.3294219970703, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23818963347002864, "epoch": 0.2998859749144812, "frac_reward_zero_std": 0.171875, "grad_norm": 0.04502458497881889, "kl": 0.061601930879987776, "learning_rate": 4.423615455322293e-06, "loss": 0.0003079274611081928, "num_tokens": 52017474.0, "reward": 2.36767578125, "reward_std": 0.5493926405906677, "rewards/code_complexity_reward/mean": 0.8570312261581421, "rewards/code_complexity_reward/std": 0.11766387522220612, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 263, "step_time": 77.58852999936789 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 161.2421875, "completions/mean_terminated_length": 160.55577087402344, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23255661339499056, "epoch": 0.3010262257696693, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.046109747141599655, "kl": 0.06697620055638254, "learning_rate": 4.417242264498143e-06, "loss": 0.0003349308390170336, "num_tokens": 52168566.0, "reward": 2.349414110183716, "reward_std": 0.548995852470398, "rewards/code_complexity_reward/mean": 0.8514648079872131, "rewards/code_complexity_reward/std": 0.1284896433353424, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 264, "step_time": 76.33839866053313 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 504.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 171.064453125, "completions/mean_terminated_length": 171.064453125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24305301369167864, "epoch": 0.30216647662485746, "frac_reward_zero_std": 0.234375, "grad_norm": 0.050438299775123596, "kl": 0.08678782763308845, "learning_rate": 4.410838677273403e-06, "loss": 0.0004337189020588994, "num_tokens": 52327135.0, "reward": 2.337158203125, "reward_std": 0.532856285572052, "rewards/code_complexity_reward/mean": 0.8467773199081421, "rewards/code_complexity_reward/std": 0.11871936917304993, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 265, "step_time": 66.16636402904987 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 510.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 166.640625, "completions/mean_terminated_length": 166.640625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2497225790284574, "epoch": 0.3033067274800456, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.044916920363903046, "kl": 0.060959411843214184, "learning_rate": 4.404404795172022e-06, "loss": 0.0003047242062166333, "num_tokens": 52481435.0, "reward": 2.330761671066284, "reward_std": 0.5379129648208618, "rewards/code_complexity_reward/mean": 0.8515625, "rewards/code_complexity_reward/std": 0.12693704664707184, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 266, "step_time": 67.47308786492795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 167.75390625, "completions/mean_terminated_length": 167.75390625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23908021650277078, "epoch": 0.30444697833523376, "frac_reward_zero_std": 0.234375, "grad_norm": 0.044794388115406036, "kl": 0.05782474859734066, "learning_rate": 4.397940720198246e-06, "loss": 0.00028908287640661, "num_tokens": 52636989.0, "reward": 2.2777342796325684, "reward_std": 0.5227269530296326, "rewards/code_complexity_reward/mean": 0.849609375, "rewards/code_complexity_reward/std": 0.11944032460451126, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 267, "step_time": 51.94986370485276 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 490.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 170.51953125, "completions/mean_terminated_length": 170.51953125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.24138875189237297, "epoch": 0.3055872291904219, "frac_reward_zero_std": 0.171875, "grad_norm": 0.05201564356684685, "kl": 0.06159069575369358, "learning_rate": 4.39144655483501e-06, "loss": 0.0003080296446569264, "num_tokens": 52792831.0, "reward": 2.2342772483825684, "reward_std": 0.5011941194534302, "rewards/code_complexity_reward/mean": 0.849609375, "rewards/code_complexity_reward/std": 0.12878265976905823, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.019099153578281403, "step": 268, "step_time": 74.45195434894413 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 437.0, "completions/max_terminated_length": 437.0, "completions/mean_length": 167.236328125, "completions/mean_terminated_length": 167.236328125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24232583330012858, "epoch": 0.30672748004561, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.0492396354675293, "kl": 0.05560210163821466, "learning_rate": 4.38492240204231e-06, "loss": 0.00027799938106909394, "num_tokens": 52946828.0, "reward": 2.2557616233825684, "reward_std": 0.5188718438148499, "rewards/code_complexity_reward/mean": 0.85595703125, "rewards/code_complexity_reward/std": 0.1306639015674591, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 269, "step_time": 52.239137971773744 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 163.205078125, "completions/mean_terminated_length": 162.5225067138672, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2406784612685442, "epoch": 0.30786773090079816, "frac_reward_zero_std": 0.234375, "grad_norm": 0.047445740550756454, "kl": 0.060348339728079736, "learning_rate": 4.378368365255564e-06, "loss": 0.0003017211565747857, "num_tokens": 53099257.0, "reward": 2.358837842941284, "reward_std": 0.528772234916687, "rewards/code_complexity_reward/mean": 0.857714831829071, "rewards/code_complexity_reward/std": 0.11847056448459625, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 270, "step_time": 56.660911072045565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 388.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 167.806640625, "completions/mean_terminated_length": 167.806640625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23857447039335966, "epoch": 0.3090079817559863, "frac_reward_zero_std": 0.203125, "grad_norm": 0.04333798959851265, "kl": 0.06314529752125964, "learning_rate": 4.371784548383985e-06, "loss": 0.0003156646271236241, "num_tokens": 53254318.0, "reward": 2.31494140625, "reward_std": 0.5052332878112793, "rewards/code_complexity_reward/mean": 0.865039050579071, "rewards/code_complexity_reward/std": 0.09548719972372055, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 271, "step_time": 57.39339284878224 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 437.0, "completions/mean_length": 162.35546875, "completions/mean_terminated_length": 160.9843292236328, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2390076054725796, "epoch": 0.31014823261117447, "frac_reward_zero_std": 0.234375, "grad_norm": 0.04666849598288536, "kl": 0.06358792871469632, "learning_rate": 4.36517105580892e-06, "loss": 0.0003176745376549661, "num_tokens": 53405576.0, "reward": 2.30029296875, "reward_std": 0.5094915628433228, "rewards/code_complexity_reward/mean": 0.8609374761581421, "rewards/code_complexity_reward/std": 0.10657218843698502, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 272, "step_time": 75.95782518014312 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 160.255859375, "completions/mean_terminated_length": 160.255859375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2476355794351548, "epoch": 0.3112884834663626, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.05186496302485466, "kl": 0.0647299776901491, "learning_rate": 4.358527992382206e-06, "loss": 0.00032363785430788994, "num_tokens": 53556755.0, "reward": 2.297412395477295, "reward_std": 0.5296302437782288, "rewards/code_complexity_reward/mean": 0.8602539300918579, "rewards/code_complexity_reward/std": 0.13088339567184448, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 273, "step_time": 49.664752571843565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 176.267578125, "completions/mean_terminated_length": 173.62400817871094, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23483550432138145, "epoch": 0.3124287343215507, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.04966617375612259, "kl": 0.05924312607385218, "learning_rate": 4.351855463424498e-06, "loss": 0.0002961533609777689, "num_tokens": 53718400.0, "reward": 2.2722654342651367, "reward_std": 0.5446837544441223, "rewards/code_complexity_reward/mean": 0.8392578363418579, "rewards/code_complexity_reward/std": 0.15135343372821808, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 274, "step_time": 49.67578369099647 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 404.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 164.228515625, "completions/mean_terminated_length": 164.228515625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2337050309870392, "epoch": 0.31356898517673887, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.061594534665346146, "kl": 0.061304085480514914, "learning_rate": 4.345153574723611e-06, "loss": 0.00030652876012027264, "num_tokens": 53869681.0, "reward": 2.2663087844848633, "reward_std": 0.5366236567497253, "rewards/code_complexity_reward/mean": 0.8445312976837158, "rewards/code_complexity_reward/std": 0.13107778131961823, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.020638125017285347, "step": 275, "step_time": 57.853969952091575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 471.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 165.357421875, "completions/mean_terminated_length": 165.357421875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24034602055326104, "epoch": 0.314709236031927, "frac_reward_zero_std": 0.21875, "grad_norm": 0.0501362681388855, "kl": 0.06425081402994692, "learning_rate": 4.338422432532829e-06, "loss": 0.0003211579751223326, "num_tokens": 54024420.0, "reward": 2.2987794876098633, "reward_std": 0.5403573513031006, "rewards/code_complexity_reward/mean": 0.8508789539337158, "rewards/code_complexity_reward/std": 0.13045984506607056, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 276, "step_time": 60.849402212537825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 469.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 164.345703125, "completions/mean_terminated_length": 164.345703125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.24118800600990653, "epoch": 0.31584948688711517, "frac_reward_zero_std": 0.234375, "grad_norm": 0.0453239269554615, "kl": 0.060969569836743176, "learning_rate": 4.331662143569235e-06, "loss": 0.00030475962557829916, "num_tokens": 54176221.0, "reward": 2.264404535293579, "reward_std": 0.4920242130756378, "rewards/code_complexity_reward/mean": 0.8531249761581421, "rewards/code_complexity_reward/std": 0.1019381582736969, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 277, "step_time": 84.79197936784476 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 169.408203125, "completions/mean_terminated_length": 168.06471252441406, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23571208585053682, "epoch": 0.3169897377423033, "frac_reward_zero_std": 0.1875, "grad_norm": 0.04322164133191109, "kl": 0.05877769639482722, "learning_rate": 4.324872815012005e-06, "loss": 0.00029391393763944507, "num_tokens": 54331962.0, "reward": 2.294189453125, "reward_std": 0.5527366399765015, "rewards/code_complexity_reward/mean": 0.8370116949081421, "rewards/code_complexity_reward/std": 0.1427718847990036, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.03612105920910835, "step": 278, "step_time": 56.96896636299789 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 172.25390625, "completions/mean_terminated_length": 170.9215850830078, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.24122904613614082, "epoch": 0.3181299885974915, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.04490979388356209, "kl": 0.06286385178100318, "learning_rate": 4.318054554500719e-06, "loss": 0.00031419360311701894, "num_tokens": 54492604.0, "reward": 2.227587938308716, "reward_std": 0.5338900089263916, "rewards/code_complexity_reward/mean": 0.8363281488418579, "rewards/code_complexity_reward/std": 0.15189151465892792, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 279, "step_time": 52.90343701466918 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 164.556640625, "completions/mean_terminated_length": 163.19412231445312, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2401852854527533, "epoch": 0.31927023945267957, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.04883811995387077, "kl": 0.0626464429369662, "learning_rate": 4.3112074701336505e-06, "loss": 0.0003131492994725704, "num_tokens": 54645181.0, "reward": 2.3102540969848633, "reward_std": 0.5531663298606873, "rewards/code_complexity_reward/mean": 0.8474609851837158, "rewards/code_complexity_reward/std": 0.13990281522274017, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 280, "step_time": 77.60690736677498 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 166.54296875, "completions/mean_terminated_length": 165.86692810058594, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23679558746516705, "epoch": 0.3204104903078677, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.04463973268866539, "kl": 0.058719266613479704, "learning_rate": 4.304331670466052e-06, "loss": 0.0002936566015705466, "num_tokens": 54799923.0, "reward": 2.2998046875, "reward_std": 0.5236972570419312, "rewards/code_complexity_reward/mean": 0.8466796875, "rewards/code_complexity_reward/std": 0.12507809698581696, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 281, "step_time": 67.97880144324154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 160.845703125, "completions/mean_terminated_length": 160.15850830078125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24360378412529826, "epoch": 0.3215507411630559, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.043944817036390305, "kl": 0.06586185545893386, "learning_rate": 4.297427264508436e-06, "loss": 0.00032935268245637417, "num_tokens": 54950136.0, "reward": 2.299609422683716, "reward_std": 0.5545421242713928, "rewards/code_complexity_reward/mean": 0.8456054925918579, "rewards/code_complexity_reward/std": 0.14683625102043152, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 282, "step_time": 58.452052501030266 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 159.44140625, "completions/mean_terminated_length": 159.44140625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23781880689784884, "epoch": 0.322690992018244, "frac_reward_zero_std": 0.1875, "grad_norm": 0.05615396425127983, "kl": 0.06970972870476544, "learning_rate": 4.290494361724844e-06, "loss": 0.0003483319887891412, "num_tokens": 55101786.0, "reward": 2.3297853469848633, "reward_std": 0.5402458310127258, "rewards/code_complexity_reward/mean": 0.857226550579071, "rewards/code_complexity_reward/std": 0.1282370388507843, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 283, "step_time": 48.078729735687375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 439.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 163.404296875, "completions/mean_terminated_length": 163.404296875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2404913934879005, "epoch": 0.3238312428734322, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.04298240691423416, "kl": 0.06400370423216373, "learning_rate": 4.283533072031116e-06, "loss": 0.000319777027470991, "num_tokens": 55253861.0, "reward": 2.2547850608825684, "reward_std": 0.4844035506248474, "rewards/code_complexity_reward/mean": 0.86474609375, "rewards/code_complexity_reward/std": 0.09627266973257065, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 284, "step_time": 51.84738012030721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 467.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 164.77734375, "completions/mean_terminated_length": 164.77734375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23799291881732643, "epoch": 0.3249714937286203, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.0524308905005455, "kl": 0.06821578613016754, "learning_rate": 4.276543505793142e-06, "loss": 0.0003410053614061326, "num_tokens": 55405811.0, "reward": 2.2526369094848633, "reward_std": 0.48564305901527405, "rewards/code_complexity_reward/mean": 0.865527331829071, "rewards/code_complexity_reward/std": 0.10364262759685516, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 285, "step_time": 55.7209284696728 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 164.1171875, "completions/mean_terminated_length": 163.4364013671875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23212787811644375, "epoch": 0.3261117445838084, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.04737062007188797, "kl": 0.06443106307415292, "learning_rate": 4.269525773825115e-06, "loss": 0.0003220220096409321, "num_tokens": 55557971.0, "reward": 2.3521485328674316, "reward_std": 0.559832751750946, "rewards/code_complexity_reward/mean": 0.8504883050918579, "rewards/code_complexity_reward/std": 0.12560909986495972, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 286, "step_time": 56.3862270116806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 164.1328125, "completions/mean_terminated_length": 162.7686309814453, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24533583433367312, "epoch": 0.3272519954389966, "frac_reward_zero_std": 0.21875, "grad_norm": 0.04488753527402878, "kl": 0.06587268190924078, "learning_rate": 4.262479987387776e-06, "loss": 0.00032924924744293094, "num_tokens": 55711243.0, "reward": 2.278125047683716, "reward_std": 0.5470768213272095, "rewards/code_complexity_reward/mean": 0.8460937142372131, "rewards/code_complexity_reward/std": 0.15155544877052307, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.033960822969675064, "step": 287, "step_time": 72.01058118510991 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 419.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 158.3125, "completions/mean_terminated_length": 158.3125, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.23630426940508187, "epoch": 0.32839224629418473, "frac_reward_zero_std": 0.1875, "grad_norm": 0.048958078026771545, "kl": 0.07160816091345623, "learning_rate": 4.255406258186644e-06, "loss": 0.0003578856121748686, "num_tokens": 55861443.0, "reward": 2.2850584983825684, "reward_std": 0.5434059500694275, "rewards/code_complexity_reward/mean": 0.85546875, "rewards/code_complexity_reward/std": 0.1446683704853058, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 288, "step_time": 61.393328784033656 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 510.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 167.470703125, "completions/mean_terminated_length": 167.470703125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2504578970838338, "epoch": 0.3295324971493729, "frac_reward_zero_std": 0.1875, "grad_norm": 0.04500032588839531, "kl": 0.06524736000574194, "learning_rate": 4.248304698370253e-06, "loss": 0.00032612550421617925, "num_tokens": 56016872.0, "reward": 2.2500977516174316, "reward_std": 0.5108485221862793, "rewards/code_complexity_reward/mean": 0.844921886920929, "rewards/code_complexity_reward/std": 0.12807314097881317, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.020638125017285347, "step": 289, "step_time": 64.24829788692296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 446.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 163.818359375, "completions/mean_terminated_length": 163.818359375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2420300105586648, "epoch": 0.330672748004561, "frac_reward_zero_std": 0.1875, "grad_norm": 0.04204472154378891, "kl": 0.06387318883207627, "learning_rate": 4.241175420528369e-06, "loss": 0.0003193722805008292, "num_tokens": 56172883.0, "reward": 2.301074266433716, "reward_std": 0.5065556764602661, "rewards/code_complexity_reward/mean": 0.8607421517372131, "rewards/code_complexity_reward/std": 0.09872888028621674, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 290, "step_time": 62.36417565122247 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 144.83203125, "completions/mean_terminated_length": 144.83203125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2376760069746524, "epoch": 0.33181299885974913, "frac_reward_zero_std": 0.3125, "grad_norm": 0.051583416759967804, "kl": 0.07325081044109538, "learning_rate": 4.234018537690204e-06, "loss": 0.00036621541948989034, "num_tokens": 56314589.0, "reward": 2.3622071743011475, "reward_std": 0.521881639957428, "rewards/code_complexity_reward/mean": 0.8753905892372131, "rewards/code_complexity_reward/std": 0.10826800018548965, "rewards/code_execution_reward/mean": 0.390625, "rewards/code_execution_reward/std": 0.48836761713027954, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.019099153578281403, "step": 291, "step_time": 50.421766691841185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 158.640625, "completions/mean_terminated_length": 157.94911193847656, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2411780678667128, "epoch": 0.3329532497149373, "frac_reward_zero_std": 0.203125, "grad_norm": 0.05666510388255119, "kl": 0.08831558749079704, "learning_rate": 4.226834163322629e-06, "loss": 0.0004414700670167804, "num_tokens": 56465601.0, "reward": 2.3082032203674316, "reward_std": 0.5339075922966003, "rewards/code_complexity_reward/mean": 0.8600585460662842, "rewards/code_complexity_reward/std": 0.12954607605934143, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.020638125017285347, "step": 292, "step_time": 95.51948346290737 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 156.08203125, "completions/mean_terminated_length": 153.98428344726562, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.240862132050097, "epoch": 0.33409350057012543, "frac_reward_zero_std": 0.25, "grad_norm": 0.05528130754828453, "kl": 0.07160151982679963, "learning_rate": 4.21962241132837e-06, "loss": 0.0003579448093660176, "num_tokens": 56614399.0, "reward": 2.277636766433716, "reward_std": 0.5387961268424988, "rewards/code_complexity_reward/mean": 0.8523437976837158, "rewards/code_complexity_reward/std": 0.14105547964572906, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 293, "step_time": 56.179328690283 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 156.560546875, "completions/mean_terminated_length": 156.560546875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23485528375022113, "epoch": 0.3352337514253136, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.05181717127561569, "kl": 0.06987551396014169, "learning_rate": 4.212383396044204e-06, "loss": 0.00034925551153719425, "num_tokens": 56760534.0, "reward": 2.293994188308716, "reward_std": 0.5154266953468323, "rewards/code_complexity_reward/mean": 0.8604491949081421, "rewards/code_complexity_reward/std": 0.12264690548181534, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 294, "step_time": 49.203859590925276 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 467.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 156.1953125, "completions/mean_terminated_length": 156.1953125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23524594795890152, "epoch": 0.3363740022805017, "frac_reward_zero_std": 0.265625, "grad_norm": 0.0508393719792366, "kl": 0.07250940863741562, "learning_rate": 4.205117232239148e-06, "loss": 0.00036239653127267957, "num_tokens": 56909690.0, "reward": 2.3439455032348633, "reward_std": 0.5272166132926941, "rewards/code_complexity_reward/mean": 0.864062488079071, "rewards/code_complexity_reward/std": 0.1118580773472786, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 295, "step_time": 83.39061332400888 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 492.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 166.283203125, "completions/mean_terminated_length": 166.283203125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2380561411846429, "epoch": 0.33751425313568983, "frac_reward_zero_std": 0.234375, "grad_norm": 0.04681697487831116, "kl": 0.06737470225198194, "learning_rate": 4.197824035112637e-06, "loss": 0.0003367048338986933, "num_tokens": 57064951.0, "reward": 2.2461915016174316, "reward_std": 0.5067380666732788, "rewards/code_complexity_reward/mean": 0.845898449420929, "rewards/code_complexity_reward/std": 0.12458451837301254, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 296, "step_time": 65.69497712049633 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 484.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 161.365234375, "completions/mean_terminated_length": 161.365234375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23555381991900504, "epoch": 0.338654503990878, "frac_reward_zero_std": 0.203125, "grad_norm": 0.05290525406599045, "kl": 0.073184008768294, "learning_rate": 4.190503920292698e-06, "loss": 0.0003658133209683001, "num_tokens": 57217382.0, "reward": 2.2933595180511475, "reward_std": 0.5313875079154968, "rewards/code_complexity_reward/mean": 0.8515625, "rewards/code_complexity_reward/std": 0.13310788571834564, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 297, "step_time": 63.83588883280754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 156.208984375, "completions/mean_terminated_length": 155.51272583007812, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.23888868372887373, "epoch": 0.33979475484606614, "frac_reward_zero_std": 0.25, "grad_norm": 0.05239173024892807, "kl": 0.07324799220077693, "learning_rate": 4.183157003834118e-06, "loss": 0.00036617449950426817, "num_tokens": 57366225.0, "reward": 2.335205078125, "reward_std": 0.5476574301719666, "rewards/code_complexity_reward/mean": 0.8614257574081421, "rewards/code_complexity_reward/std": 0.13429279625415802, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 298, "step_time": 64.66044539399445 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 162.662109375, "completions/mean_terminated_length": 161.9784698486328, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23790403408929706, "epoch": 0.3409350057012543, "frac_reward_zero_std": 0.203125, "grad_norm": 0.051691990345716476, "kl": 0.06787464342778549, "learning_rate": 4.175783402216604e-06, "loss": 0.00033934664679691195, "num_tokens": 57518776.0, "reward": 2.2887208461761475, "reward_std": 0.5197409391403198, "rewards/code_complexity_reward/mean": 0.8563476800918579, "rewards/code_complexity_reward/std": 0.12148679047822952, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 299, "step_time": 67.75956806447357 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 149.671875, "completions/mean_terminated_length": 149.671875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24012503307312727, "epoch": 0.34207525655644244, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.04847465455532074, "kl": 0.07688300259178504, "learning_rate": 4.168383232342934e-06, "loss": 0.00038437440525740385, "num_tokens": 57663884.0, "reward": 2.2979981899261475, "reward_std": 0.5284583568572998, "rewards/code_complexity_reward/mean": 0.8603515028953552, "rewards/code_complexity_reward/std": 0.12484931200742722, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 300, "step_time": 50.69314260780811 }, { "epoch": 0.34207525655644244, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 236.5, "eval_completions/max_terminated_length": 236.5, "eval_completions/mean_length": 155.18, "eval_completions/mean_terminated_length": 155.18, "eval_completions/min_length": 94.88, "eval_completions/min_terminated_length": 94.88, "eval_entropy": 0.2422045087814331, "eval_frac_reward_zero_std": 0.27, "eval_kl": 0.06951393343508244, "eval_loss": 0.0003490610106382519, "eval_num_tokens": 57663884.0, "eval_reward": 2.225187554359436, "eval_reward_std": 0.373583753965795, "eval_rewards/code_complexity_reward/mean": 0.85424999833107, "eval_rewards/code_complexity_reward/std": 0.09202601574361324, "eval_rewards/code_execution_reward/mean": 0.28, "eval_rewards/code_execution_reward/std": 0.3078084021806717, "eval_rewards/code_syntax_reward/mean": 0.49125, "eval_rewards/code_syntax_reward/std": 0.022306769788265228, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.4996875, "eval_rewards/xmlcount_reward_func/std": 0.000883883461356163, "eval_runtime": 584.4448, "eval_samples_per_second": 0.171, "eval_steps_per_second": 0.022, "step": 300 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 478.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 155.451171875, "completions/mean_terminated_length": 155.451171875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2415571438614279, "epoch": 0.34321550741163054, "frac_reward_zero_std": 0.171875, "grad_norm": 0.051255226135253906, "kl": 0.07011319871526212, "learning_rate": 4.160956611537106e-06, "loss": 0.00035032074083574116, "num_tokens": 57813467.0, "reward": 2.3020997047424316, "reward_std": 0.5246695280075073, "rewards/code_complexity_reward/mean": 0.8600585460662842, "rewards/code_complexity_reward/std": 0.11854217201471329, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 301, "step_time": 65.03757978510112 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 468.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 150.810546875, "completions/mean_terminated_length": 150.810546875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24295532959513366, "epoch": 0.3443557582668187, "frac_reward_zero_std": 0.25, "grad_norm": 0.04382210224866867, "kl": 0.07401645911158994, "learning_rate": 4.153503657542479e-06, "loss": 0.0003700536326505244, "num_tokens": 57958282.0, "reward": 2.2657716274261475, "reward_std": 0.5097931027412415, "rewards/code_complexity_reward/mean": 0.8603515625, "rewards/code_complexity_reward/std": 0.12082670629024506, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 302, "step_time": 54.42134018614888 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 150.43359375, "completions/mean_terminated_length": 149.7260284423828, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24408766813576221, "epoch": 0.34549600912200684, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.04705377668142319, "kl": 0.07407142478041351, "learning_rate": 4.146024488519901e-06, "loss": 0.00037020992022007704, "num_tokens": 58103744.0, "reward": 2.3743653297424316, "reward_std": 0.5398397445678711, "rewards/code_complexity_reward/mean": 0.8673828840255737, "rewards/code_complexity_reward/std": 0.1228756308555603, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 303, "step_time": 75.81469448376447 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 506.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 148.767578125, "completions/mean_terminated_length": 148.767578125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24649694189429283, "epoch": 0.346636259977195, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.05103423818945885, "kl": 0.0774504539440386, "learning_rate": 4.138519223045842e-06, "loss": 0.0003872902598232031, "num_tokens": 58248437.0, "reward": 2.289599657058716, "reward_std": 0.5087858438491821, "rewards/code_complexity_reward/mean": 0.8721679449081421, "rewards/code_complexity_reward/std": 0.11404236406087875, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.021347908303141594, "step": 304, "step_time": 46.870723659172654 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 156.91015625, "completions/mean_terminated_length": 156.91015625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24019284429959953, "epoch": 0.34777651083238315, "frac_reward_zero_std": 0.265625, "grad_norm": 0.04921876639127731, "kl": 0.0749114173813723, "learning_rate": 4.130987980110508e-06, "loss": 0.0003744091372936964, "num_tokens": 58398459.0, "reward": 2.2837891578674316, "reward_std": 0.5007952451705933, "rewards/code_complexity_reward/mean": 0.8666015863418579, "rewards/code_complexity_reward/std": 0.10509417951107025, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 305, "step_time": 52.9015642311424 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 152.142578125, "completions/mean_terminated_length": 151.4383544921875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24918708228506148, "epoch": 0.34891676168757124, "frac_reward_zero_std": 0.265625, "grad_norm": 0.04607396945357323, "kl": 0.07641922717448324, "learning_rate": 4.123430879115963e-06, "loss": 0.00038207252509891987, "num_tokens": 58545264.0, "reward": 2.2527832984924316, "reward_std": 0.5186406373977661, "rewards/code_complexity_reward/mean": 0.8636718988418579, "rewards/code_complexity_reward/std": 0.13634376227855682, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 306, "step_time": 75.77395214512944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 429.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 152.40234375, "completions/mean_terminated_length": 152.40234375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23856558534316719, "epoch": 0.3500570125427594, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.051078468561172485, "kl": 0.07702591503039002, "learning_rate": 4.115848039874225e-06, "loss": 0.0003850272041745484, "num_tokens": 58690258.0, "reward": 2.3229005336761475, "reward_std": 0.5376975536346436, "rewards/code_complexity_reward/mean": 0.8634765148162842, "rewards/code_complexity_reward/std": 0.12895038723945618, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 307, "step_time": 69.41620851866901 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 152.900390625, "completions/mean_terminated_length": 151.49217224121094, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23891098168678582, "epoch": 0.35119726339794755, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.04444656893610954, "kl": 0.07875015499303117, "learning_rate": 4.108239582605374e-06, "loss": 0.00039380055386573076, "num_tokens": 58836771.0, "reward": 2.2798829078674316, "reward_std": 0.530003011226654, "rewards/code_complexity_reward/mean": 0.8649413585662842, "rewards/code_complexity_reward/std": 0.13445691764354706, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 308, "step_time": 59.08185982890427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 145.99609375, "completions/mean_terminated_length": 145.99609375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24154944554902613, "epoch": 0.3523375142531357, "frac_reward_zero_std": 0.21875, "grad_norm": 0.05733642354607582, "kl": 0.08130229113157839, "learning_rate": 4.100605627935647e-06, "loss": 0.00040645257104188204, "num_tokens": 58979305.0, "reward": 2.3550782203674316, "reward_std": 0.5183271169662476, "rewards/code_complexity_reward/mean": 0.8761718273162842, "rewards/code_complexity_reward/std": 0.10326657444238663, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 309, "step_time": 49.22428749129176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 151.216796875, "completions/mean_terminated_length": 150.51075744628906, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24535666569136083, "epoch": 0.35347776510832385, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.046350233256816864, "kl": 0.0775298533262685, "learning_rate": 4.0929462968955176e-06, "loss": 0.0003875322872772813, "num_tokens": 59126476.0, "reward": 2.2962892055511475, "reward_std": 0.516596794128418, "rewards/code_complexity_reward/mean": 0.86767578125, "rewards/code_complexity_reward/std": 0.11193488538265228, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 310, "step_time": 56.93920504581183 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 154.521484375, "completions/mean_terminated_length": 153.82191467285156, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23323591449297965, "epoch": 0.35461801596351195, "frac_reward_zero_std": 0.21875, "grad_norm": 0.04477803781628609, "kl": 0.07713026774581522, "learning_rate": 4.085261710917786e-06, "loss": 0.00038552930345758796, "num_tokens": 59273711.0, "reward": 2.2652344703674316, "reward_std": 0.5092153549194336, "rewards/code_complexity_reward/mean": 0.8590819835662842, "rewards/code_complexity_reward/std": 0.12950506806373596, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 311, "step_time": 48.801338424906135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 152.505859375, "completions/mean_terminated_length": 151.80235290527344, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24645623215474188, "epoch": 0.3557582668187001, "frac_reward_zero_std": 0.25, "grad_norm": 0.04791852831840515, "kl": 0.08979721571085975, "learning_rate": 4.0775519918356486e-06, "loss": 0.0004488584236241877, "num_tokens": 59422470.0, "reward": 2.250927686691284, "reward_std": 0.5121951103210449, "rewards/code_complexity_reward/mean": 0.868457019329071, "rewards/code_complexity_reward/std": 0.12635241448879242, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 312, "step_time": 68.46254638768733 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 471.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 150.578125, "completions/mean_terminated_length": 150.578125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2431730031967163, "epoch": 0.35689851767388825, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.04532944783568382, "kl": 0.08157721173483878, "learning_rate": 4.069817261880769e-06, "loss": 0.000407743122195825, "num_tokens": 59568498.0, "reward": 2.296191453933716, "reward_std": 0.5310239791870117, "rewards/code_complexity_reward/mean": 0.8568359613418579, "rewards/code_complexity_reward/std": 0.13249875605106354, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 313, "step_time": 63.437474311329424 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 383.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 147.09375, "completions/mean_terminated_length": 147.09375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2432594585698098, "epoch": 0.3580387685290764, "frac_reward_zero_std": 0.296875, "grad_norm": 0.04699109122157097, "kl": 0.08193584706168622, "learning_rate": 4.062057643681335e-06, "loss": 0.00040980122867040336, "num_tokens": 59714998.0, "reward": 2.3163087368011475, "reward_std": 0.5255415439605713, "rewards/code_complexity_reward/mean": 0.865234375, "rewards/code_complexity_reward/std": 0.11374407261610031, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 314, "step_time": 72.32640192471445 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 431.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 153.10546875, "completions/mean_terminated_length": 153.10546875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23999151727184653, "epoch": 0.35917901938426455, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.04466055706143379, "kl": 0.08106641785707325, "learning_rate": 4.054273260260125e-06, "loss": 0.0004051071300636977, "num_tokens": 59861924.0, "reward": 2.3053712844848633, "reward_std": 0.5189892053604126, "rewards/code_complexity_reward/mean": 0.864550769329071, "rewards/code_complexity_reward/std": 0.11355400830507278, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 315, "step_time": 50.28099675383419 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 453.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 159.6796875, "completions/mean_terminated_length": 159.6796875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23642858606763184, "epoch": 0.3603192702394527, "frac_reward_zero_std": 0.234375, "grad_norm": 0.04649047181010246, "kl": 0.08092391508398578, "learning_rate": 4.046464235032546e-06, "loss": 0.00040438881842419505, "num_tokens": 60013320.0, "reward": 2.3015623092651367, "reward_std": 0.5129129886627197, "rewards/code_complexity_reward/mean": 0.8597656488418579, "rewards/code_complexity_reward/std": 0.10714376717805862, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 316, "step_time": 71.27701679896563 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 146.486328125, "completions/mean_terminated_length": 145.7710418701172, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24513429240323603, "epoch": 0.3614595210946408, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.05101662129163742, "kl": 0.08965911401901394, "learning_rate": 4.0386306918046815e-06, "loss": 0.0004482795484364033, "num_tokens": 60155813.0, "reward": 2.3197755813598633, "reward_std": 0.535926342010498, "rewards/code_complexity_reward/mean": 0.8713866472244263, "rewards/code_complexity_reward/std": 0.13114216923713684, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 317, "step_time": 66.20207439921796 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 155.052734375, "completions/mean_terminated_length": 154.3542022705078, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2348857291508466, "epoch": 0.36259977194982895, "frac_reward_zero_std": 0.265625, "grad_norm": 0.049660343676805496, "kl": 0.08075080125126988, "learning_rate": 4.0307727547713316e-06, "loss": 0.00040371561772190034, "num_tokens": 60304104.0, "reward": 2.23681640625, "reward_std": 0.519487738609314, "rewards/code_complexity_reward/mean": 0.8570312261581421, "rewards/code_complexity_reward/std": 0.14327512681484222, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 318, "step_time": 57.479535453021526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 428.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 146.41015625, "completions/mean_terminated_length": 146.41015625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24184280028566718, "epoch": 0.3637400228050171, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.0532829575240612, "kl": 0.08352462574839592, "learning_rate": 4.0228905485140415e-06, "loss": 0.00041749683441594243, "num_tokens": 60447510.0, "reward": 2.3244142532348633, "reward_std": 0.5169695019721985, "rewards/code_complexity_reward/mean": 0.871874988079071, "rewards/code_complexity_reward/std": 0.11687382310628891, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 319, "step_time": 50.38525109086186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 416.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 150.220703125, "completions/mean_terminated_length": 150.220703125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2458191078621894, "epoch": 0.36488027366020526, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.050211239606142044, "kl": 0.08548981876811013, "learning_rate": 4.014984197999125e-06, "loss": 0.0004274926905054599, "num_tokens": 60594119.0, "reward": 2.323535203933716, "reward_std": 0.521002471446991, "rewards/code_complexity_reward/mean": 0.8695312738418579, "rewards/code_complexity_reward/std": 0.11403201520442963, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 320, "step_time": 67.76964551303536 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 353.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 140.84375, "completions/mean_terminated_length": 140.84375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24532390665262938, "epoch": 0.3660205245153934, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.06749877333641052, "kl": 0.08680760534480214, "learning_rate": 4.007053828575684e-06, "loss": 0.00043389739585109055, "num_tokens": 60734375.0, "reward": 2.3768067359924316, "reward_std": 0.5214296579360962, "rewards/code_complexity_reward/mean": 0.8809570074081421, "rewards/code_complexity_reward/std": 0.10209959000349045, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 321, "step_time": 47.6248261006549 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 373.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 138.794921875, "completions/mean_terminated_length": 138.794921875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24146926845423877, "epoch": 0.3671607753705815, "frac_reward_zero_std": 0.3359375, "grad_norm": 0.05594800412654877, "kl": 0.0891994014964439, "learning_rate": 3.999099565973623e-06, "loss": 0.00044585426803678274, "num_tokens": 60874390.0, "reward": 2.316699266433716, "reward_std": 0.494499534368515, "rewards/code_complexity_reward/mean": 0.8817383050918579, "rewards/code_complexity_reward/std": 0.09159120172262192, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 322, "step_time": 58.00103221926838 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 147.373046875, "completions/mean_terminated_length": 146.65948486328125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24344864767044783, "epoch": 0.36830102622576966, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.05176732689142227, "kl": 0.08505458303261548, "learning_rate": 3.991121536301653e-06, "loss": 0.00042529444908723235, "num_tokens": 61019041.0, "reward": 2.2273926734924316, "reward_std": 0.5029345750808716, "rewards/code_complexity_reward/mean": 0.866894543170929, "rewards/code_complexity_reward/std": 0.13599930703639984, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 323, "step_time": 68.36126183811575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 411.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 153.53515625, "completions/mean_terminated_length": 153.53515625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23913892032578588, "epoch": 0.3694412770809578, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.04576624184846878, "kl": 0.08384055265923962, "learning_rate": 3.983119866045297e-06, "loss": 0.0004192324122413993, "num_tokens": 61165951.0, "reward": 2.2879881858825684, "reward_std": 0.496294766664505, "rewards/code_complexity_reward/mean": 0.86279296875, "rewards/code_complexity_reward/std": 0.10154101997613907, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 324, "step_time": 41.1731313848868 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 142.916015625, "completions/mean_terminated_length": 142.916015625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24225484929047525, "epoch": 0.37058152793614596, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.04908119514584541, "kl": 0.08763065398670733, "learning_rate": 3.975094682064875e-06, "loss": 0.0004380669561214745, "num_tokens": 61306612.0, "reward": 2.2179689407348633, "reward_std": 0.46424970030784607, "rewards/code_complexity_reward/mean": 0.8796874284744263, "rewards/code_complexity_reward/std": 0.10134853422641754, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 325, "step_time": 58.54854205995798 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 153.01953125, "completions/mean_terminated_length": 153.01953125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2353818181436509, "epoch": 0.3717217787913341, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.051413487643003464, "kl": 0.08499727898743004, "learning_rate": 3.967046111593505e-06, "loss": 0.00042497290996834636, "num_tokens": 61453346.0, "reward": 2.2492189407348633, "reward_std": 0.49652600288391113, "rewards/code_complexity_reward/mean": 0.86572265625, "rewards/code_complexity_reward/std": 0.12264160066843033, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 326, "step_time": 53.437907808460295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 139.12890625, "completions/mean_terminated_length": 137.6666717529297, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24468160420656204, "epoch": 0.3728620296465222, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.049793459475040436, "kl": 0.09435899392701685, "learning_rate": 3.958974282235079e-06, "loss": 0.0004717320261988789, "num_tokens": 61593336.0, "reward": 2.303515672683716, "reward_std": 0.5299769639968872, "rewards/code_complexity_reward/mean": 0.8746093511581421, "rewards/code_complexity_reward/std": 0.12855452299118042, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 327, "step_time": 61.15378224104643 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 145.787109375, "completions/mean_terminated_length": 145.07044982910156, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23775542876683176, "epoch": 0.37400228050171036, "frac_reward_zero_std": 0.265625, "grad_norm": 0.0542609803378582, "kl": 0.09409465454518795, "learning_rate": 3.9508793219622375e-06, "loss": 0.00047046173131093383, "num_tokens": 61736647.0, "reward": 2.281298875808716, "reward_std": 0.5391775965690613, "rewards/code_complexity_reward/mean": 0.8587890863418579, "rewards/code_complexity_reward/std": 0.15112395584583282, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 328, "step_time": 76.15695321373641 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 145.427734375, "completions/mean_terminated_length": 145.427734375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24300231458619237, "epoch": 0.3751425313568985, "frac_reward_zero_std": 0.265625, "grad_norm": 0.04756368324160576, "kl": 0.0889827812789008, "learning_rate": 3.942761359114345e-06, "loss": 0.0004448198596946895, "num_tokens": 61880654.0, "reward": 2.2643065452575684, "reward_std": 0.5279693007469177, "rewards/code_complexity_reward/mean": 0.8600585460662842, "rewards/code_complexity_reward/std": 0.13975584506988525, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.028648728504776955, "step": 329, "step_time": 52.73668203223497 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 476.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 146.705078125, "completions/mean_terminated_length": 146.705078125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24185658479109406, "epoch": 0.37628278221208666, "frac_reward_zero_std": 0.296875, "grad_norm": 0.04890678822994232, "kl": 0.09222708223387599, "learning_rate": 3.934620522395458e-06, "loss": 0.0004611426847986877, "num_tokens": 62024903.0, "reward": 2.283007860183716, "reward_std": 0.5184127688407898, "rewards/code_complexity_reward/mean": 0.8617187738418579, "rewards/code_complexity_reward/std": 0.12927378714084625, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 330, "step_time": 74.77203152887523 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 140.052734375, "completions/mean_terminated_length": 139.32485961914062, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2471125484444201, "epoch": 0.3774230330672748, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.054104000329971313, "kl": 0.09655787807423621, "learning_rate": 3.926456940872274e-06, "loss": 0.00048272404819726944, "num_tokens": 62166102.0, "reward": 2.3252930641174316, "reward_std": 0.5222073793411255, "rewards/code_complexity_reward/mean": 0.8825194835662842, "rewards/code_complexity_reward/std": 0.12212284654378891, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.03121940791606903, "step": 331, "step_time": 68.1091973548755 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 139.68359375, "completions/mean_terminated_length": 139.68359375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23701427062042058, "epoch": 0.37856328392246297, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.050812188535928726, "kl": 0.09307460050331429, "learning_rate": 3.918270743972097e-06, "loss": 0.0004653455107472837, "num_tokens": 62307684.0, "reward": 2.2865235805511475, "reward_std": 0.5176687240600586, "rewards/code_complexity_reward/mean": 0.873730480670929, "rewards/code_complexity_reward/std": 0.12816768884658813, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 332, "step_time": 39.76583269238472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 144.552734375, "completions/mean_terminated_length": 143.8336639404297, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24012543400749564, "epoch": 0.37970353477765106, "frac_reward_zero_std": 0.3359375, "grad_norm": 0.050038114190101624, "kl": 0.09393420646665618, "learning_rate": 3.910062061480778e-06, "loss": 0.0004696321557275951, "num_tokens": 62448635.0, "reward": 2.265380859375, "reward_std": 0.5041525363922119, "rewards/code_complexity_reward/mean": 0.869921863079071, "rewards/code_complexity_reward/std": 0.12692198157310486, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 333, "step_time": 49.65620758291334 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 471.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 139.880859375, "completions/mean_terminated_length": 139.880859375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2430057127494365, "epoch": 0.3808437856328392, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.05008656159043312, "kl": 0.09022766497218981, "learning_rate": 3.901831023540662e-06, "loss": 0.00045096786925569177, "num_tokens": 62590186.0, "reward": 2.278271436691284, "reward_std": 0.5198320150375366, "rewards/code_complexity_reward/mean": 0.868457019329071, "rewards/code_complexity_reward/std": 0.1305796205997467, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 334, "step_time": 52.30252857785672 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 144.943359375, "completions/mean_terminated_length": 144.943359375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23138721124269068, "epoch": 0.38198403648802737, "frac_reward_zero_std": 0.34375, "grad_norm": 0.04714575409889221, "kl": 0.08908967167371884, "learning_rate": 3.89357776064852e-06, "loss": 0.00044551468454301357, "num_tokens": 62733805.0, "reward": 2.3143067359924316, "reward_std": 0.5165241360664368, "rewards/code_complexity_reward/mean": 0.8624999523162842, "rewards/code_complexity_reward/std": 0.11442017555236816, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 335, "step_time": 50.015685085207224 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 144.408203125, "completions/mean_terminated_length": 142.9666748046875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24298282293602824, "epoch": 0.3831242873432155, "frac_reward_zero_std": 0.3359375, "grad_norm": 0.04595186188817024, "kl": 0.09306597977411002, "learning_rate": 3.885302403653483e-06, "loss": 0.00046534318244084716, "num_tokens": 62878994.0, "reward": 2.2510743141174316, "reward_std": 0.5200023651123047, "rewards/code_complexity_reward/mean": 0.864941418170929, "rewards/code_complexity_reward/std": 0.14251169562339783, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 336, "step_time": 56.531663740985096 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 394.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 138.671875, "completions/mean_terminated_length": 138.671875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24970223591662943, "epoch": 0.38426453819840367, "frac_reward_zero_std": 0.3359375, "grad_norm": 0.05119774863123894, "kl": 0.11362573364749551, "learning_rate": 3.8770050837549675e-06, "loss": 0.0005678415764123201, "num_tokens": 63019774.0, "reward": 2.2884767055511475, "reward_std": 0.5033197402954102, "rewards/code_complexity_reward/mean": 0.8779296875, "rewards/code_complexity_reward/std": 0.11200542002916336, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 337, "step_time": 47.92598834540695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 418.0, "completions/max_terminated_length": 418.0, "completions/mean_length": 147.447265625, "completions/mean_terminated_length": 147.447265625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2326176268979907, "epoch": 0.38540478905359177, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.046668026596307755, "kl": 0.0935648342128843, "learning_rate": 3.868685932500596e-06, "loss": 0.0004676897369790822, "num_tokens": 63163427.0, "reward": 2.3336424827575684, "reward_std": 0.5343185663223267, "rewards/code_complexity_reward/mean": 0.85986328125, "rewards/code_complexity_reward/std": 0.12169457226991653, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 338, "step_time": 64.47841781098396 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 144.4609375, "completions/mean_terminated_length": 144.4609375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23746109171770513, "epoch": 0.3865450399087799, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.052389957010746, "kl": 0.09617461229208857, "learning_rate": 3.860345081784107e-06, "loss": 0.00048067833995446563, "num_tokens": 63306799.0, "reward": 2.296191453933716, "reward_std": 0.48790672421455383, "rewards/code_complexity_reward/mean": 0.8775390386581421, "rewards/code_complexity_reward/std": 0.09244132786989212, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 339, "step_time": 76.48058678861707 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 379.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 135.953125, "completions/mean_terminated_length": 135.953125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2419378391932696, "epoch": 0.38768529076396807, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.04745148494839668, "kl": 0.09851671638898551, "learning_rate": 3.851982663843272e-06, "loss": 0.0004925375105813146, "num_tokens": 63443587.0, "reward": 2.34765625, "reward_std": 0.5129004716873169, "rewards/code_complexity_reward/mean": 0.8907226324081421, "rewards/code_complexity_reward/std": 0.09720589965581894, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 340, "step_time": 56.05253504682332 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 423.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 138.291015625, "completions/mean_terminated_length": 138.291015625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23935696529224515, "epoch": 0.3888255416191562, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.0482211597263813, "kl": 0.10342675034189597, "learning_rate": 3.84359881125779e-06, "loss": 0.0005170535296201706, "num_tokens": 63584748.0, "reward": 2.2745118141174316, "reward_std": 0.5095634460449219, "rewards/code_complexity_reward/mean": 0.8747069835662842, "rewards/code_complexity_reward/std": 0.13486932218074799, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 341, "step_time": 43.24562596157193 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 144.669921875, "completions/mean_terminated_length": 143.95108032226562, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2414632081054151, "epoch": 0.3899657924743444, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.05270516499876976, "kl": 0.11128554318565875, "learning_rate": 3.835193656947192e-06, "loss": 0.0005562930018641055, "num_tokens": 63726359.0, "reward": 2.2938966751098633, "reward_std": 0.5059944987297058, "rewards/code_complexity_reward/mean": 0.871874988079071, "rewards/code_complexity_reward/std": 0.10360430926084518, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 342, "step_time": 63.98435707297176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 501.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 137.771484375, "completions/mean_terminated_length": 137.771484375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23066698526963592, "epoch": 0.39110604332953247, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.05429472029209137, "kl": 0.1005627986160107, "learning_rate": 3.826767334168731e-06, "loss": 0.0005027024890296161, "num_tokens": 63864642.0, "reward": 2.3688478469848633, "reward_std": 0.532829999923706, "rewards/code_complexity_reward/mean": 0.875781238079071, "rewards/code_complexity_reward/std": 0.12306973338127136, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 343, "step_time": 48.53677408117801 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 503.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 142.4375, "completions/mean_terminated_length": 142.4375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24248190922662616, "epoch": 0.3922462941847206, "frac_reward_zero_std": 0.3125, "grad_norm": 0.04674766585230827, "kl": 0.09893363283481449, "learning_rate": 3.8183199765152704e-06, "loss": 0.0004946303088217974, "num_tokens": 64007294.0, "reward": 2.2970705032348633, "reward_std": 0.511560320854187, "rewards/code_complexity_reward/mean": 0.871874988079071, "rewards/code_complexity_reward/std": 0.12239456176757812, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 344, "step_time": 48.444135185331106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 141.47265625, "completions/mean_terminated_length": 141.47265625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2326144182588905, "epoch": 0.3933865450399088, "frac_reward_zero_std": 0.265625, "grad_norm": 0.05848526954650879, "kl": 0.09788200299954042, "learning_rate": 3.809851717913164e-06, "loss": 0.0004892013967037201, "num_tokens": 64148184.0, "reward": 2.2926759719848633, "reward_std": 0.5007627010345459, "rewards/code_complexity_reward/mean": 0.8752930164337158, "rewards/code_complexity_reward/std": 0.11221794784069061, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 345, "step_time": 47.858713454566896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 134.0546875, "completions/mean_terminated_length": 134.0546875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24803337920457125, "epoch": 0.3945267958950969, "frac_reward_zero_std": 0.3125, "grad_norm": 0.050423700362443924, "kl": 0.10744945437181741, "learning_rate": 3.8013626926201343e-06, "loss": 0.0005369381979107857, "num_tokens": 64285504.0, "reward": 2.2906737327575684, "reward_std": 0.49010977149009705, "rewards/code_complexity_reward/mean": 0.88525390625, "rewards/code_complexity_reward/std": 0.1059960350394249, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 346, "step_time": 45.76086327806115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 412.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 135.818359375, "completions/mean_terminated_length": 135.818359375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24359867093153298, "epoch": 0.3956670467502851, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.05141959711909294, "kl": 0.10631204082164913, "learning_rate": 3.792853035223144e-06, "loss": 0.0005314150475896895, "num_tokens": 64423071.0, "reward": 2.2697267532348633, "reward_std": 0.5121054649353027, "rewards/code_complexity_reward/mean": 0.878222644329071, "rewards/code_complexity_reward/std": 0.13512104749679565, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 347, "step_time": 50.29658919200301 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 134.076171875, "completions/mean_terminated_length": 133.3365936279297, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2346201278269291, "epoch": 0.39680729760547323, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.09201809018850327, "kl": 0.11162420263281092, "learning_rate": 3.7843228806362635e-06, "loss": 0.0005579321878030896, "num_tokens": 64559170.0, "reward": 2.2823243141174316, "reward_std": 0.5205442309379578, "rewards/code_complexity_reward/mean": 0.8751952648162842, "rewards/code_complexity_reward/std": 0.13059385120868683, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 348, "step_time": 56.506881047040224 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 426.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 137.751953125, "completions/mean_terminated_length": 137.751953125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24168867454864085, "epoch": 0.3979475484606613, "frac_reward_zero_std": 0.28125, "grad_norm": 0.06180218979716301, "kl": 0.10838352487189695, "learning_rate": 3.775772364098529e-06, "loss": 0.0005416886997409165, "num_tokens": 64697063.0, "reward": 2.3006837368011475, "reward_std": 0.5072416067123413, "rewards/code_complexity_reward/mean": 0.88818359375, "rewards/code_complexity_reward/std": 0.10922256857156754, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 349, "step_time": 42.99926423467696 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 138.09375, "completions/mean_terminated_length": 138.09375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23081625998020172, "epoch": 0.3990877993158495, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.04760453850030899, "kl": 0.11510873964289203, "learning_rate": 3.7672016211717977e-06, "loss": 0.0005753110162913799, "num_tokens": 64836427.0, "reward": 2.2852540016174316, "reward_std": 0.5045219659805298, "rewards/code_complexity_reward/mean": 0.876660168170929, "rewards/code_complexity_reward/std": 0.11748829483985901, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 350, "step_time": 81.34354112669826 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 130.685546875, "completions/mean_terminated_length": 130.685546875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24803048721514642, "epoch": 0.40022805017103763, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.05068599432706833, "kl": 0.1255942468997091, "learning_rate": 3.758610787738604e-06, "loss": 0.0006278419168666005, "num_tokens": 64972734.0, "reward": 2.2940430641174316, "reward_std": 0.48980674147605896, "rewards/code_complexity_reward/mean": 0.889355480670929, "rewards/code_complexity_reward/std": 0.11282258480787277, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 351, "step_time": 52.010246213525534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 477.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 135.427734375, "completions/mean_terminated_length": 135.427734375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24031746690161526, "epoch": 0.4013683010262258, "frac_reward_zero_std": 0.34375, "grad_norm": 0.048284370452165604, "kl": 0.11404553084867075, "learning_rate": 3.7500000000000005e-06, "loss": 0.0005701166810467839, "num_tokens": 65109921.0, "reward": 2.318359375, "reward_std": 0.48183730244636536, "rewards/code_complexity_reward/mean": 0.8931640386581421, "rewards/code_complexity_reward/std": 0.08458571135997772, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 352, "step_time": 57.52352385595441 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 129.2421875, "completions/mean_terminated_length": 128.49314880371094, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23521115072071552, "epoch": 0.40250855188141393, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.05429527908563614, "kl": 0.11120319465408102, "learning_rate": 3.7413693944734e-06, "loss": 0.0005558189586736262, "num_tokens": 65243193.0, "reward": 2.3497557640075684, "reward_std": 0.5458939671516418, "rewards/code_complexity_reward/mean": 0.884765625, "rewards/code_complexity_reward/std": 0.13581795990467072, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 353, "step_time": 56.71119786426425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 387.0, "completions/max_terminated_length": 387.0, "completions/mean_length": 127.15234375, "completions/mean_terminated_length": 127.15234375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24310318427160382, "epoch": 0.40364880273660203, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.06270001828670502, "kl": 0.11255746555980295, "learning_rate": 3.7327191079904096e-06, "loss": 0.0005625977064482868, "num_tokens": 65376359.0, "reward": 2.3143067359924316, "reward_std": 0.4981292486190796, "rewards/code_complexity_reward/mean": 0.8898437023162842, "rewards/code_complexity_reward/std": 0.10215090215206146, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 354, "step_time": 56.44047520495951 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 140.546875, "completions/mean_terminated_length": 140.546875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24816568847745657, "epoch": 0.4047890535917902, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.06335047632455826, "kl": 0.11160050355829298, "learning_rate": 3.7240492776946663e-06, "loss": 0.0005577892297878861, "num_tokens": 65515443.0, "reward": 2.288330078125, "reward_std": 0.5130791664123535, "rewards/code_complexity_reward/mean": 0.8746094107627869, "rewards/code_complexity_reward/std": 0.1290862113237381, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 355, "step_time": 50.728254615329206 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 134.92578125, "completions/mean_terminated_length": 134.92578125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.25100550847128034, "epoch": 0.40592930444697833, "frac_reward_zero_std": 0.28125, "grad_norm": 0.05522535368800163, "kl": 0.14337460312526673, "learning_rate": 3.7153600410396558e-06, "loss": 0.0007172225741669536, "num_tokens": 65652293.0, "reward": 2.336474895477295, "reward_std": 0.5271716117858887, "rewards/code_complexity_reward/mean": 0.8822265863418579, "rewards/code_complexity_reward/std": 0.11881065368652344, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 356, "step_time": 51.61766483448446 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 136.5546875, "completions/mean_terminated_length": 136.5546875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2380962825845927, "epoch": 0.4070695553021665, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.050345953553915024, "kl": 0.11759294907096773, "learning_rate": 3.7066515357865384e-06, "loss": 0.0005880466196686029, "num_tokens": 65794137.0, "reward": 2.2784667015075684, "reward_std": 0.48901981115341187, "rewards/code_complexity_reward/mean": 0.8828125, "rewards/code_complexity_reward/std": 0.1092471033334732, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 357, "step_time": 84.42273805942386 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 138.8984375, "completions/mean_terminated_length": 138.8984375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24320385209284723, "epoch": 0.40820980615735464, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06667108088731766, "kl": 0.10775606567040086, "learning_rate": 3.6979239000019622e-06, "loss": 0.0005387474084272981, "num_tokens": 65933725.0, "reward": 2.281982421875, "reward_std": 0.49580812454223633, "rewards/code_complexity_reward/mean": 0.8902343511581421, "rewards/code_complexity_reward/std": 0.11126536875963211, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 358, "step_time": 58.01320493686944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 133.798828125, "completions/mean_terminated_length": 133.05870056152344, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24391084886156023, "epoch": 0.40935005701254273, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.05882536619901657, "kl": 0.12031214701710269, "learning_rate": 3.689177272055877e-06, "loss": 0.0006015563849359751, "num_tokens": 66073362.0, "reward": 2.236328125, "reward_std": 0.4875340759754181, "rewards/code_complexity_reward/mean": 0.8804687857627869, "rewards/code_complexity_reward/std": 0.12617111206054688, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02465205453336239, "step": 359, "step_time": 86.66271214187145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 134.126953125, "completions/mean_terminated_length": 133.38748168945312, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23379125143401325, "epoch": 0.4104903078677309, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.05663944408297539, "kl": 0.12371783051639795, "learning_rate": 3.6804117906193367e-06, "loss": 0.000618450460024178, "num_tokens": 66210987.0, "reward": 2.3274903297424316, "reward_std": 0.5228416919708252, "rewards/code_complexity_reward/mean": 0.8854491710662842, "rewards/code_complexity_reward/std": 0.13021205365657806, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 360, "step_time": 65.95660157129169 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 442.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 141.470703125, "completions/mean_terminated_length": 141.470703125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2348247233312577, "epoch": 0.41163055872291904, "frac_reward_zero_std": 0.359375, "grad_norm": 0.047044266015291214, "kl": 0.11165488639380783, "learning_rate": 3.671627594662303e-06, "loss": 0.0005581655423156917, "num_tokens": 66352192.0, "reward": 2.29736328125, "reward_std": 0.4796447455883026, "rewards/code_complexity_reward/mean": 0.8868163824081421, "rewards/code_complexity_reward/std": 0.08807031065225601, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 361, "step_time": 50.73786941356957 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 397.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 127.5, "completions/mean_terminated_length": 127.5, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24324861890636384, "epoch": 0.4127708095781072, "frac_reward_zero_std": 0.375, "grad_norm": 0.04826129600405693, "kl": 0.12431881856173277, "learning_rate": 3.6628248234514434e-06, "loss": 0.0006215584580786526, "num_tokens": 66486804.0, "reward": 2.2669434547424316, "reward_std": 0.49199292063713074, "rewards/code_complexity_reward/mean": 0.8917968273162842, "rewards/code_complexity_reward/std": 0.1193225085735321, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 362, "step_time": 49.110326004214585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 130.185546875, "completions/mean_terminated_length": 130.185546875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24235447589308023, "epoch": 0.41391106043329534, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.05071071535348892, "kl": 0.12383586284704506, "learning_rate": 3.6540036165479203e-06, "loss": 0.0006190636195242405, "num_tokens": 66623259.0, "reward": 2.2733888626098633, "reward_std": 0.4952648878097534, "rewards/code_complexity_reward/mean": 0.887011706829071, "rewards/code_complexity_reward/std": 0.10881974548101425, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 363, "step_time": 62.73782292380929 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 382.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 128.44140625, "completions/mean_terminated_length": 128.44140625, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.23988607223145664, "epoch": 0.4150513112884835, "frac_reward_zero_std": 0.34375, "grad_norm": 0.05775224044919014, "kl": 0.1250284230336547, "learning_rate": 3.6451641138051806e-06, "loss": 0.0006250040605664253, "num_tokens": 66758917.0, "reward": 2.29296875, "reward_std": 0.5133271813392639, "rewards/code_complexity_reward/mean": 0.8838866949081421, "rewards/code_complexity_reward/std": 0.12070053815841675, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 364, "step_time": 68.94683268573135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 135.69921875, "completions/mean_terminated_length": 134.9628143310547, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2429511984810233, "epoch": 0.4161915621436716, "frac_reward_zero_std": 0.328125, "grad_norm": 0.060240548104047775, "kl": 0.1241556212771684, "learning_rate": 3.6363064553667378e-06, "loss": 0.0006207016995176673, "num_tokens": 66896299.0, "reward": 2.2720704078674316, "reward_std": 0.5347638130187988, "rewards/code_complexity_reward/mean": 0.8727538585662842, "rewards/code_complexity_reward/std": 0.1534496694803238, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 365, "step_time": 49.70657617878169 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 140.173828125, "completions/mean_terminated_length": 140.173828125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23506920272484422, "epoch": 0.41733181299885974, "frac_reward_zero_std": 0.34375, "grad_norm": 0.056363265961408615, "kl": 0.12771084543783218, "learning_rate": 3.627430781663948e-06, "loss": 0.0006386853056028485, "num_tokens": 67037428.0, "reward": 2.2575197219848633, "reward_std": 0.4945961833000183, "rewards/code_complexity_reward/mean": 0.871386706829071, "rewards/code_complexity_reward/std": 0.12266061455011368, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 366, "step_time": 49.11522895190865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 399.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 136.19921875, "completions/mean_terminated_length": 136.19921875, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.23457503784447908, "epoch": 0.4184720638540479, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.06304360181093216, "kl": 0.14178057003300637, "learning_rate": 3.618537233413789e-06, "loss": 0.0007088965503498912, "num_tokens": 67175946.0, "reward": 2.323047161102295, "reward_std": 0.5290911793708801, "rewards/code_complexity_reward/mean": 0.8817383050918579, "rewards/code_complexity_reward/std": 0.12431221455335617, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 367, "step_time": 48.88306345511228 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 130.517578125, "completions/mean_terminated_length": 130.517578125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24349382799118757, "epoch": 0.41961231470923605, "frac_reward_zero_std": 0.265625, "grad_norm": 0.06610674411058426, "kl": 0.11598533391952515, "learning_rate": 3.6096259516166226e-06, "loss": 0.0005799740320071578, "num_tokens": 67311995.0, "reward": 2.2335939407348633, "reward_std": 0.4923528730869293, "rewards/code_complexity_reward/mean": 0.879687488079071, "rewards/code_complexity_reward/std": 0.12743709981441498, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 368, "step_time": 48.611808717250824 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 407.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 136.5078125, "completions/mean_terminated_length": 136.5078125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23948925570584834, "epoch": 0.4207525655644242, "frac_reward_zero_std": 0.34375, "grad_norm": 0.05595802515745163, "kl": 0.11139996955171227, "learning_rate": 3.600697077553964e-06, "loss": 0.000557053426746279, "num_tokens": 67450767.0, "reward": 2.2623047828674316, "reward_std": 0.4838132858276367, "rewards/code_complexity_reward/mean": 0.8849608898162842, "rewards/code_complexity_reward/std": 0.11124802380800247, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 369, "step_time": 50.075851381756365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 129.185546875, "completions/mean_terminated_length": 129.185546875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2372443953063339, "epoch": 0.4218928164196123, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.05580110102891922, "kl": 0.13161034556105733, "learning_rate": 3.5917507527862394e-06, "loss": 0.0006579150212928653, "num_tokens": 67585002.0, "reward": 2.280517578125, "reward_std": 0.5217459201812744, "rewards/code_complexity_reward/mean": 0.8794921636581421, "rewards/code_complexity_reward/std": 0.14179909229278564, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 370, "step_time": 36.17253524437547 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 445.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 134.244140625, "completions/mean_terminated_length": 134.244140625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2475564752239734, "epoch": 0.42303306727480045, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.061336833983659744, "kl": 0.11969125689938664, "learning_rate": 3.5827871191505425e-06, "loss": 0.0005983071168884635, "num_tokens": 67721895.0, "reward": 2.2611327171325684, "reward_std": 0.5052688121795654, "rewards/code_complexity_reward/mean": 0.8818359375, "rewards/code_complexity_reward/std": 0.13635651767253876, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 371, "step_time": 52.88881162367761 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 333.0, "completions/mean_length": 133.8671875, "completions/mean_terminated_length": 133.127197265625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23644091677851975, "epoch": 0.4241733181299886, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.05687389895319939, "kl": 0.11702511424664408, "learning_rate": 3.573806318758388e-06, "loss": 0.0005850282032042742, "num_tokens": 67859447.0, "reward": 2.2213380336761475, "reward_std": 0.5076133012771606, "rewards/code_complexity_reward/mean": 0.8720703125, "rewards/code_complexity_reward/std": 0.14638905227184296, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 372, "step_time": 57.661560621112585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 128.294921875, "completions/mean_terminated_length": 127.54402923583984, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24694095156155527, "epoch": 0.42531356898517675, "frac_reward_zero_std": 0.296875, "grad_norm": 0.05542644113302231, "kl": 0.1185934814857319, "learning_rate": 3.5648084939934523e-06, "loss": 0.0005927849560976028, "num_tokens": 67993802.0, "reward": 2.309033155441284, "reward_std": 0.5283570885658264, "rewards/code_complexity_reward/mean": 0.888964831829071, "rewards/code_complexity_reward/std": 0.13137948513031006, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 373, "step_time": 64.46525597758591 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 481.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 131.01171875, "completions/mean_terminated_length": 131.01171875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23153228615410626, "epoch": 0.4264538198403649, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.08389084786176682, "kl": 0.12858850718475878, "learning_rate": 3.5557937875093242e-06, "loss": 0.0006429023342207074, "num_tokens": 68127436.0, "reward": 2.3134765625, "reward_std": 0.5287655591964722, "rewards/code_complexity_reward/mean": 0.8824218511581421, "rewards/code_complexity_reward/std": 0.12803494930267334, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 374, "step_time": 55.86721396725625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 132.4375, "completions/mean_terminated_length": 131.69471740722656, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23928201850503683, "epoch": 0.427594070695553, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.04809296876192093, "kl": 0.12316750013269484, "learning_rate": 3.5467623422272353e-06, "loss": 0.0006157839670777321, "num_tokens": 68263796.0, "reward": 2.311084270477295, "reward_std": 0.5353963375091553, "rewards/code_complexity_reward/mean": 0.8802734613418579, "rewards/code_complexity_reward/std": 0.13498395681381226, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 375, "step_time": 55.96657151449472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 336.0, "completions/max_terminated_length": 336.0, "completions/mean_length": 123.79296875, "completions/mean_terminated_length": 123.79296875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24143511964939535, "epoch": 0.42873432155074115, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.04701266437768936, "kl": 0.12810889270622283, "learning_rate": 3.537714301333801e-06, "loss": 0.0006405559834092855, "num_tokens": 68395854.0, "reward": 2.3187499046325684, "reward_std": 0.49659308791160583, "rewards/code_complexity_reward/mean": 0.89892578125, "rewards/code_complexity_reward/std": 0.10742410272359848, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 376, "step_time": 43.94949145335704 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 456.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 130.984375, "completions/mean_terminated_length": 130.984375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24237047019414604, "epoch": 0.4298745724059293, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.049849674105644226, "kl": 0.1249371095909737, "learning_rate": 3.528649808278747e-06, "loss": 0.0006245831027626991, "num_tokens": 68532214.0, "reward": 2.2520508766174316, "reward_std": 0.48743578791618347, "rewards/code_complexity_reward/mean": 0.8903319835662842, "rewards/code_complexity_reward/std": 0.12504123151302338, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 377, "step_time": 80.27314197737724 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 491.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 124.3984375, "completions/mean_terminated_length": 124.3984375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2345402860082686, "epoch": 0.43101482326111745, "frac_reward_zero_std": 0.3359375, "grad_norm": 0.059948552399873734, "kl": 0.1291619922267273, "learning_rate": 3.519569006772633e-06, "loss": 0.0006456288974732161, "num_tokens": 68665106.0, "reward": 2.3424315452575684, "reward_std": 0.5070081949234009, "rewards/code_complexity_reward/mean": 0.89599609375, "rewards/code_complexity_reward/std": 0.10875826328992844, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.019900046288967133, "step": 378, "step_time": 64.46456888783723 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 121.4765625, "completions/mean_terminated_length": 121.4765625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23919007671065629, "epoch": 0.4321550741163056, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.05731254443526268, "kl": 0.1362942554987967, "learning_rate": 3.5104720407845794e-06, "loss": 0.0006814985536038876, "num_tokens": 68794838.0, "reward": 2.3048830032348633, "reward_std": 0.4926818609237671, "rewards/code_complexity_reward/mean": 0.897753894329071, "rewards/code_complexity_reward/std": 0.10294029861688614, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 379, "step_time": 45.33194672502577 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 122.208984375, "completions/mean_terminated_length": 121.44618225097656, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23641603835858405, "epoch": 0.43329532497149376, "frac_reward_zero_std": 0.421875, "grad_norm": 0.051437538117170334, "kl": 0.1318818723084405, "learning_rate": 3.5013590545399818e-06, "loss": 0.0006592122954316437, "num_tokens": 68926105.0, "reward": 2.2978515625, "reward_std": 0.5345662236213684, "rewards/code_complexity_reward/mean": 0.8829101920127869, "rewards/code_complexity_reward/std": 0.14823134243488312, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 380, "step_time": 55.40224204864353 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 121.544921875, "completions/mean_terminated_length": 121.544921875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2427958264015615, "epoch": 0.43443557582668185, "frac_reward_zero_std": 0.34375, "grad_norm": 0.06016857549548149, "kl": 0.1295687984675169, "learning_rate": 3.492230192518221e-06, "loss": 0.0006478886352851987, "num_tokens": 69056304.0, "reward": 2.27587890625, "reward_std": 0.4932214021682739, "rewards/code_complexity_reward/mean": 0.8970702886581421, "rewards/code_complexity_reward/std": 0.11712977290153503, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 381, "step_time": 87.95851822383702 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 121.046875, "completions/mean_terminated_length": 121.046875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.25101572507992387, "epoch": 0.43557582668187, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.05399470031261444, "kl": 0.1366410170448944, "learning_rate": 3.483085599450381e-06, "loss": 0.0006831755745224655, "num_tokens": 69186696.0, "reward": 2.258251905441284, "reward_std": 0.49246925115585327, "rewards/code_complexity_reward/mean": 0.895800769329071, "rewards/code_complexity_reward/std": 0.1267034262418747, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 382, "step_time": 56.805841537192464 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 118.99609375, "completions/mean_terminated_length": 118.99609375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23669573874212801, "epoch": 0.43671607753705816, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04780752584338188, "kl": 0.14313053037039936, "learning_rate": 3.473925420316946e-06, "loss": 0.0007156103383749723, "num_tokens": 69315642.0, "reward": 2.32177734375, "reward_std": 0.513420045375824, "rewards/code_complexity_reward/mean": 0.8995116949081421, "rewards/code_complexity_reward/std": 0.11714457720518112, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 383, "step_time": 45.41680517885834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 444.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 118.974609375, "completions/mean_terminated_length": 118.974609375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23468273947946727, "epoch": 0.4378563283922463, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.046462953090667725, "kl": 0.15378708811476827, "learning_rate": 3.464749800345507e-06, "loss": 0.0007691212231293321, "num_tokens": 69446637.0, "reward": 2.2728514671325684, "reward_std": 0.46837905049324036, "rewards/code_complexity_reward/mean": 0.90625, "rewards/code_complexity_reward/std": 0.0933188796043396, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 384, "step_time": 52.97746030241251 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 511.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 133.53515625, "completions/mean_terminated_length": 133.53515625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23848383664153516, "epoch": 0.43899657924743446, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.05010687932372093, "kl": 0.1295872904593125, "learning_rate": 3.4555588850084575e-06, "loss": 0.0006477666320279241, "num_tokens": 69584343.0, "reward": 2.2803711891174316, "reward_std": 0.5056437849998474, "rewards/code_complexity_reward/mean": 0.8805663585662842, "rewards/code_complexity_reward/std": 0.1278616487979889, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 385, "step_time": 61.10678591579199 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 438.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 123.619140625, "completions/mean_terminated_length": 123.619140625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23748830519616604, "epoch": 0.44013683010262256, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.057984210550785065, "kl": 0.14121253800112754, "learning_rate": 3.4463528200206868e-06, "loss": 0.0007061326177790761, "num_tokens": 69716784.0, "reward": 2.3252930641174316, "reward_std": 0.5506800413131714, "rewards/code_complexity_reward/mean": 0.885449230670929, "rewards/code_complexity_reward/std": 0.14557461440563202, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 386, "step_time": 47.22611601650715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 129.91015625, "completions/mean_terminated_length": 129.1624298095703, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24284753063693643, "epoch": 0.4412770809578107, "frac_reward_zero_std": 0.375, "grad_norm": 0.05499423295259476, "kl": 0.14115254813805223, "learning_rate": 3.4371317513372692e-06, "loss": 0.0007056049071252346, "num_tokens": 69853022.0, "reward": 2.245410442352295, "reward_std": 0.479946106672287, "rewards/code_complexity_reward/mean": 0.8919921517372131, "rewards/code_complexity_reward/std": 0.12308944761753082, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 387, "step_time": 65.49390120990574 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 376.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 121.734375, "completions/mean_terminated_length": 121.734375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23127337242476642, "epoch": 0.44241733181299886, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.05196168273687363, "kl": 0.140639515244402, "learning_rate": 3.427895825151153e-06, "loss": 0.0007031418499536812, "num_tokens": 69982390.0, "reward": 2.286426067352295, "reward_std": 0.5175740718841553, "rewards/code_complexity_reward/mean": 0.8875976800918579, "rewards/code_complexity_reward/std": 0.13644739985466003, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 388, "step_time": 58.419655366800725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 119.830078125, "completions/mean_terminated_length": 119.830078125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23510882025584579, "epoch": 0.443557582668187, "frac_reward_zero_std": 0.3359375, "grad_norm": 0.05806722864508629, "kl": 0.13576922717038542, "learning_rate": 3.4186451878908393e-06, "loss": 0.0006787673337385058, "num_tokens": 70113283.0, "reward": 2.3277344703674316, "reward_std": 0.5045801997184753, "rewards/code_complexity_reward/mean": 0.8937499523162842, "rewards/code_complexity_reward/std": 0.10903418064117432, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 389, "step_time": 37.07674381043762 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 391.0, "completions/mean_length": 120.361328125, "completions/mean_terminated_length": 119.59490966796875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23193160095252097, "epoch": 0.44469783352337516, "frac_reward_zero_std": 0.375, "grad_norm": 0.05687374994158745, "kl": 0.15590731415431947, "learning_rate": 3.4093799862180627e-06, "loss": 0.0007796958088874817, "num_tokens": 70243924.0, "reward": 2.2829103469848633, "reward_std": 0.5328755974769592, "rewards/code_complexity_reward/mean": 0.887988269329071, "rewards/code_complexity_reward/std": 0.14945188164710999, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 390, "step_time": 75.06318667158484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 419.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 124.8046875, "completions/mean_terminated_length": 124.8046875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2359098179731518, "epoch": 0.44583808437856326, "frac_reward_zero_std": 0.375, "grad_norm": 0.05229226127266884, "kl": 0.139411632088013, "learning_rate": 3.4001003670254656e-06, "loss": 0.0006969214882701635, "num_tokens": 70378948.0, "reward": 2.2578125, "reward_std": 0.49403491616249084, "rewards/code_complexity_reward/mean": 0.889453113079071, "rewards/code_complexity_reward/std": 0.1329188197851181, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 391, "step_time": 49.232722433283925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 121.251953125, "completions/mean_terminated_length": 121.251953125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23668664158321917, "epoch": 0.4469783352337514, "frac_reward_zero_std": 0.34375, "grad_norm": 0.06518787145614624, "kl": 0.14603422454092652, "learning_rate": 3.390806477434269e-06, "loss": 0.0007301373407244682, "num_tokens": 70508853.0, "reward": 2.2533693313598633, "reward_std": 0.4954831600189209, "rewards/code_complexity_reward/mean": 0.890429675579071, "rewards/code_complexity_reward/std": 0.130916565656662, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 392, "step_time": 44.35004546865821 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 118.87109375, "completions/mean_terminated_length": 118.87109375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2373962712008506, "epoch": 0.44811858608893956, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.05481765791773796, "kl": 0.15264043188653886, "learning_rate": 3.381498464791939e-06, "loss": 0.0007630919571965933, "num_tokens": 70639235.0, "reward": 2.30029296875, "reward_std": 0.47805556654930115, "rewards/code_complexity_reward/mean": 0.9073241949081421, "rewards/code_complexity_reward/std": 0.09199925512075424, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 393, "step_time": 49.60849203541875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 321.0, "completions/max_terminated_length": 321.0, "completions/mean_length": 117.01171875, "completions/mean_terminated_length": 117.01171875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24781238590367138, "epoch": 0.4492588369441277, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.051011376082897186, "kl": 0.1461547848302871, "learning_rate": 3.372176476669853e-06, "loss": 0.0007307039340957999, "num_tokens": 70767949.0, "reward": 2.2955079078674316, "reward_std": 0.483625590801239, "rewards/code_complexity_reward/mean": 0.9064452648162842, "rewards/code_complexity_reward/std": 0.09075385332107544, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 394, "step_time": 46.64195475354791 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 411.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 119.4609375, "completions/mean_terminated_length": 119.4609375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23041015304625034, "epoch": 0.45039908779931587, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.05310584232211113, "kl": 0.15537988347932696, "learning_rate": 3.362840660860958e-06, "loss": 0.0007770503289066255, "num_tokens": 70898117.0, "reward": 2.3145995140075684, "reward_std": 0.49473056197166443, "rewards/code_complexity_reward/mean": 0.89697265625, "rewards/code_complexity_reward/std": 0.1031576544046402, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.0275954008102417, "step": 395, "step_time": 41.14403156749904 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 115.2890625, "completions/mean_terminated_length": 115.2890625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24169186130166054, "epoch": 0.45153933865450396, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.054822154343128204, "kl": 0.14878933469299227, "learning_rate": 3.353491165377429e-06, "loss": 0.0007438104948960245, "num_tokens": 71024697.0, "reward": 2.2953615188598633, "reward_std": 0.4909396767616272, "rewards/code_complexity_reward/mean": 0.9036132097244263, "rewards/code_complexity_reward/std": 0.10237698256969452, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 396, "step_time": 54.361553927883506 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 119.306640625, "completions/mean_terminated_length": 119.306640625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2342367540113628, "epoch": 0.4526795895096921, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.05231962352991104, "kl": 0.15001653006765991, "learning_rate": 3.3441281384483215e-06, "loss": 0.0007500287028960884, "num_tokens": 71154338.0, "reward": 2.2881836891174316, "reward_std": 0.5058774352073669, "rewards/code_complexity_reward/mean": 0.8932616710662842, "rewards/code_complexity_reward/std": 0.12090050429105759, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 397, "step_time": 42.76451295148581 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 122.943359375, "completions/mean_terminated_length": 122.943359375, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.23505561728961766, "epoch": 0.45381984036488027, "frac_reward_zero_std": 0.34375, "grad_norm": 0.059835128486156464, "kl": 0.15285688953008503, "learning_rate": 3.3347517285172225e-06, "loss": 0.0007643225253559649, "num_tokens": 71286989.0, "reward": 2.285449266433716, "reward_std": 0.5090071558952332, "rewards/code_complexity_reward/mean": 0.8895508050918579, "rewards/code_complexity_reward/std": 0.12672756612300873, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 398, "step_time": 61.578950674273074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 400.0, "completions/max_terminated_length": 400.0, "completions/mean_length": 120.861328125, "completions/mean_terminated_length": 120.861328125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23990347515791655, "epoch": 0.4549600912200684, "frac_reward_zero_std": 0.453125, "grad_norm": 0.048552658408880234, "kl": 0.1459363616304472, "learning_rate": 3.325362084239894e-06, "loss": 0.0007295535178855062, "num_tokens": 71417202.0, "reward": 2.2510743141174316, "reward_std": 0.46060439944267273, "rewards/code_complexity_reward/mean": 0.9059569835662842, "rewards/code_complexity_reward/std": 0.09711424261331558, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 399, "step_time": 57.99772274773568 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 123.59765625, "completions/mean_terminated_length": 122.83757019042969, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2368939050938934, "epoch": 0.45610034207525657, "frac_reward_zero_std": 0.40625, "grad_norm": 0.05477568879723549, "kl": 0.14853612054139376, "learning_rate": 3.31595935448192e-06, "loss": 0.0007427138043567538, "num_tokens": 71549880.0, "reward": 2.306396484375, "reward_std": 0.5045772194862366, "rewards/code_complexity_reward/mean": 0.9009765386581421, "rewards/code_complexity_reward/std": 0.11492796987295151, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 400, "step_time": 61.6933317463845 }, { "epoch": 0.45610034207525657, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 190.8, "eval_completions/max_terminated_length": 190.8, "eval_completions/mean_length": 118.8325, "eval_completions/mean_terminated_length": 118.8325, "eval_completions/min_length": 72.82, "eval_completions/min_terminated_length": 72.82, "eval_entropy": 0.23794658929109574, "eval_frac_reward_zero_std": 0.45, "eval_kl": 0.17117660254240036, "eval_loss": 0.0008558540139347315, "eval_num_tokens": 71549880.0, "eval_reward": 2.255000092983246, "eval_reward_std": 0.3508962706103921, "eval_rewards/code_complexity_reward/mean": 0.9043749845027924, "eval_rewards/code_complexity_reward/std": 0.05942907802760601, "eval_rewards/code_execution_reward/mean": 0.255, "eval_rewards/code_execution_reward/std": 0.3249868035316467, "eval_rewards/code_syntax_reward/mean": 0.49625, "eval_rewards/code_syntax_reward/std": 0.010606601536273956, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.499375, "eval_rewards/xmlcount_reward_func/std": 0.001767766922712326, "eval_runtime": 441.4472, "eval_samples_per_second": 0.227, "eval_steps_per_second": 0.029, "step": 400 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 480.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 122.619140625, "completions/mean_terminated_length": 122.619140625, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.23240923834964633, "epoch": 0.4572405929304447, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.04984649270772934, "kl": 0.14413501613307744, "learning_rate": 3.3065436883163453e-06, "loss": 0.0007207003072835505, "num_tokens": 71682929.0, "reward": 2.2345216274261475, "reward_std": 0.512353241443634, "rewards/code_complexity_reward/mean": 0.8857421875, "rewards/code_complexity_reward/std": 0.15845954418182373, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.019900046288967133, "step": 401, "step_time": 54.53022719640285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 115.734375, "completions/mean_terminated_length": 115.734375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2367185177281499, "epoch": 0.4583808437856328, "frac_reward_zero_std": 0.4375, "grad_norm": 0.054644860327243805, "kl": 0.15427015582099557, "learning_rate": 3.2971152350213106e-06, "loss": 0.0007712830556556582, "num_tokens": 71811929.0, "reward": 2.299560546875, "reward_std": 0.4975734353065491, "rewards/code_complexity_reward/mean": 0.9068359136581421, "rewards/code_complexity_reward/std": 0.10908868163824081, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 402, "step_time": 45.118199698626995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 121.2734375, "completions/mean_terminated_length": 121.2734375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23648336273618042, "epoch": 0.45952109464082097, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.05276469141244888, "kl": 0.1623565899208188, "learning_rate": 3.2876741440776853e-06, "loss": 0.0008117217803373933, "num_tokens": 71943853.0, "reward": 2.2759766578674316, "reward_std": 0.5153742432594299, "rewards/code_complexity_reward/mean": 0.8888671398162842, "rewards/code_complexity_reward/std": 0.13650330901145935, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.022032126784324646, "step": 403, "step_time": 55.01487617008388 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 461.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 118.81640625, "completions/mean_terminated_length": 118.81640625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23132115020416677, "epoch": 0.4606613454960091, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.04893947392702103, "kl": 0.15707223943900317, "learning_rate": 3.2782205651667013e-06, "loss": 0.000785376294516027, "num_tokens": 72072607.0, "reward": 2.266894578933716, "reward_std": 0.4929073750972748, "rewards/code_complexity_reward/mean": 0.8954101204872131, "rewards/code_complexity_reward/std": 0.11797138303518295, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 404, "step_time": 53.34035706426948 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 113.880859375, "completions/mean_terminated_length": 113.880859375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23704025405459106, "epoch": 0.4618015963511973, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.05118609964847565, "kl": 0.17301914794370532, "learning_rate": 3.2687546481675776e-06, "loss": 0.000865153968334198, "num_tokens": 72200122.0, "reward": 2.2791991233825684, "reward_std": 0.48164138197898865, "rewards/code_complexity_reward/mean": 0.90478515625, "rewards/code_complexity_reward/std": 0.10827571898698807, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 405, "step_time": 36.937382616102695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 461.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 124.03515625, "completions/mean_terminated_length": 124.03515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23047756077721715, "epoch": 0.4629418472063854, "frac_reward_zero_std": 0.359375, "grad_norm": 0.05154126510024071, "kl": 0.14750345924403518, "learning_rate": 3.259276543155142e-06, "loss": 0.0007373878615908325, "num_tokens": 72332456.0, "reward": 2.2466800212860107, "reward_std": 0.4930919408798218, "rewards/code_complexity_reward/mean": 0.891796886920929, "rewards/code_complexity_reward/std": 0.1283307522535324, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 406, "step_time": 62.654511521570385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 119.39453125, "completions/mean_terminated_length": 119.39453125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22797776735387743, "epoch": 0.4640820980615735, "frac_reward_zero_std": 0.375, "grad_norm": 0.053510893136262894, "kl": 0.1626154554542154, "learning_rate": 3.2497864003974554e-06, "loss": 0.0008130439091473818, "num_tokens": 72462826.0, "reward": 2.283447265625, "reward_std": 0.5420441031455994, "rewards/code_complexity_reward/mean": 0.8848632574081421, "rewards/code_complexity_reward/std": 0.15539979934692383, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 407, "step_time": 43.34789809025824 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 316.0, "completions/max_terminated_length": 316.0, "completions/mean_length": 112.435546875, "completions/mean_terminated_length": 112.435546875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24146250379271805, "epoch": 0.4652223489167617, "frac_reward_zero_std": 0.421875, "grad_norm": 0.055091701447963715, "kl": 0.1643863545032218, "learning_rate": 3.2402843703534283e-06, "loss": 0.0008217976428568363, "num_tokens": 72587997.0, "reward": 2.2462892532348633, "reward_std": 0.4481271207332611, "rewards/code_complexity_reward/mean": 0.9109374284744263, "rewards/code_complexity_reward/std": 0.08807297050952911, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 408, "step_time": 43.22042842581868 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 440.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 119.513671875, "completions/mean_terminated_length": 119.513671875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23630913253873587, "epoch": 0.4663625997719498, "frac_reward_zero_std": 0.359375, "grad_norm": 0.05518146604299545, "kl": 0.1627484739292413, "learning_rate": 3.2307706036704328e-06, "loss": 0.0008138243574649096, "num_tokens": 72717800.0, "reward": 2.28759765625, "reward_std": 0.47424620389938354, "rewards/code_complexity_reward/mean": 0.9014648199081421, "rewards/code_complexity_reward/std": 0.10209546983242035, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 409, "step_time": 44.57774845696986 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 114.96484375, "completions/mean_terminated_length": 114.96484375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23422666871920228, "epoch": 0.467502850627138, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.05357418581843376, "kl": 0.16693991888314486, "learning_rate": 3.221245251181919e-06, "loss": 0.0008345325477421284, "num_tokens": 72845906.0, "reward": 2.245898485183716, "reward_std": 0.4932745397090912, "rewards/code_complexity_reward/mean": 0.9007812738418579, "rewards/code_complexity_reward/std": 0.1370266228914261, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.03121940791606903, "step": 410, "step_time": 50.34485499560833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 121.4921875, "completions/mean_terminated_length": 121.4921875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23058838956058025, "epoch": 0.46864310148232613, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04833970591425896, "kl": 0.15725798439234495, "learning_rate": 3.2117084639050204e-06, "loss": 0.0007862384663894773, "num_tokens": 72979182.0, "reward": 2.3604493141174316, "reward_std": 0.5041542053222656, "rewards/code_complexity_reward/mean": 0.9000976085662842, "rewards/code_complexity_reward/std": 0.10329686850309372, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 411, "step_time": 58.11837682966143 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 119.044921875, "completions/mean_terminated_length": 118.27593231201172, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23619383201003075, "epoch": 0.4697833523375142, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06341002881526947, "kl": 0.1614475721726194, "learning_rate": 3.2021603930381582e-06, "loss": 0.0008071271004155278, "num_tokens": 73108557.0, "reward": 2.2973146438598633, "reward_std": 0.48670071363449097, "rewards/code_complexity_reward/mean": 0.899707019329071, "rewards/code_complexity_reward/std": 0.10075502842664719, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 412, "step_time": 57.438233986496925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 272.0, "completions/mean_length": 112.712890625, "completions/mean_terminated_length": 111.93150329589844, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24062100891023874, "epoch": 0.4709236031927024, "frac_reward_zero_std": 0.46875, "grad_norm": 0.07552368193864822, "kl": 0.24624964187387377, "learning_rate": 3.1926011899586483e-06, "loss": 0.0012327749282121658, "num_tokens": 73235962.0, "reward": 2.2969727516174316, "reward_std": 0.519773542881012, "rewards/code_complexity_reward/mean": 0.902050793170929, "rewards/code_complexity_reward/std": 0.13048416376113892, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 413, "step_time": 57.63817667681724 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 400.0, "completions/max_terminated_length": 400.0, "completions/mean_length": 114.375, "completions/mean_terminated_length": 114.375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22655797936022282, "epoch": 0.47206385404789053, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.0535585917532444, "kl": 0.17566453118342906, "learning_rate": 3.1830310062202996e-06, "loss": 0.0008781616925261915, "num_tokens": 73362670.0, "reward": 2.343554973602295, "reward_std": 0.5064142942428589, "rewards/code_complexity_reward/mean": 0.9095703363418579, "rewards/code_complexity_reward/std": 0.10311303287744522, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 414, "step_time": 50.148019461892545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 118.041015625, "completions/mean_terminated_length": 117.27005767822266, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23672349355183542, "epoch": 0.4732041049030787, "frac_reward_zero_std": 0.5, "grad_norm": 0.056923530995845795, "kl": 0.1762691430049017, "learning_rate": 3.1734499935510093e-06, "loss": 0.0008814205066300929, "num_tokens": 73490055.0, "reward": 2.2789061069488525, "reward_std": 0.473827600479126, "rewards/code_complexity_reward/mean": 0.9071288704872131, "rewards/code_complexity_reward/std": 0.09448054432868958, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 415, "step_time": 49.131408584304154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 110.52734375, "completions/mean_terminated_length": 110.52734375, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.23430724651552737, "epoch": 0.47434435575826683, "frac_reward_zero_std": 0.515625, "grad_norm": 0.051604971289634705, "kl": 0.16569223755504936, "learning_rate": 3.1638583038503596e-06, "loss": 0.0008284861105494201, "num_tokens": 73614893.0, "reward": 2.3153321743011475, "reward_std": 0.5275986790657043, "rewards/code_complexity_reward/mean": 0.89501953125, "rewards/code_complexity_reward/std": 0.1277519017457962, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 416, "step_time": 63.02522189449519 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 362.0, "completions/max_terminated_length": 362.0, "completions/mean_length": 112.046875, "completions/mean_terminated_length": 112.046875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23504252266138792, "epoch": 0.475484606613455, "frac_reward_zero_std": 0.46875, "grad_norm": 0.050566814839839935, "kl": 0.1658735469682142, "learning_rate": 3.15425608918721e-06, "loss": 0.0008295408915728331, "num_tokens": 73741049.0, "reward": 2.3495607376098633, "reward_std": 0.5250813364982605, "rewards/code_complexity_reward/mean": 0.9080078601837158, "rewards/code_complexity_reward/std": 0.12396078556776047, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 417, "step_time": 49.9953505275771 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 307.0, "completions/max_terminated_length": 307.0, "completions/mean_length": 115.326171875, "completions/mean_terminated_length": 115.326171875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23787945695221424, "epoch": 0.4766248574686431, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04725063964724541, "kl": 0.1686120309168473, "learning_rate": 3.144643501797282e-06, "loss": 0.0008429404115304351, "num_tokens": 73868880.0, "reward": 2.3011231422424316, "reward_std": 0.4992704689502716, "rewards/code_complexity_reward/mean": 0.9059569835662842, "rewards/code_complexity_reward/std": 0.10740061849355698, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 418, "step_time": 60.97927251365036 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 300.0, "completions/max_terminated_length": 300.0, "completions/mean_length": 115.923828125, "completions/mean_terminated_length": 115.923828125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23645846918225288, "epoch": 0.47776510832383123, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04904581978917122, "kl": 0.17918801202904433, "learning_rate": 3.1350206940807523e-06, "loss": 0.0008956976234912872, "num_tokens": 73998073.0, "reward": 2.2343263626098633, "reward_std": 0.48001110553741455, "rewards/code_complexity_reward/mean": 0.8982422351837158, "rewards/code_complexity_reward/std": 0.12933987379074097, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 419, "step_time": 34.642344935797155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 379.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 112.40625, "completions/mean_terminated_length": 112.40625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2389662810601294, "epoch": 0.4789053591790194, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.052907466888427734, "kl": 0.17964438337367028, "learning_rate": 3.125387818599831e-06, "loss": 0.0008981736027635634, "num_tokens": 74123221.0, "reward": 2.3136720657348633, "reward_std": 0.49664098024368286, "rewards/code_complexity_reward/mean": 0.90673828125, "rewards/code_complexity_reward/std": 0.11151205003261566, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 420, "step_time": 45.52068513259292 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 379.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 109.31640625, "completions/mean_terminated_length": 109.31640625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23218866204842925, "epoch": 0.48004561003420754, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05251403898000717, "kl": 0.164981305366382, "learning_rate": 3.1157450280763464e-06, "loss": 0.0008249063976109028, "num_tokens": 74247831.0, "reward": 2.2679686546325684, "reward_std": 0.4914129376411438, "rewards/code_complexity_reward/mean": 0.9052734375, "rewards/code_complexity_reward/std": 0.12196019291877747, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 421, "step_time": 45.917470660060644 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 112.09765625, "completions/mean_terminated_length": 111.31507110595703, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23334999312646687, "epoch": 0.4811858608893957, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.0505448542535305, "kl": 0.16546817065682262, "learning_rate": 3.10609247538932e-06, "loss": 0.000827274692710489, "num_tokens": 74374561.0, "reward": 2.328906536102295, "reward_std": 0.5482733249664307, "rewards/code_complexity_reward/mean": 0.8934569954872131, "rewards/code_complexity_reward/std": 0.14857825636863708, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 422, "step_time": 51.55601093173027 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 400.0, "completions/max_terminated_length": 400.0, "completions/mean_length": 109.861328125, "completions/mean_terminated_length": 109.861328125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23976706387475133, "epoch": 0.4823261117445838, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05241825431585312, "kl": 0.17604907089844346, "learning_rate": 3.096430313572547e-06, "loss": 0.0008802064694464207, "num_tokens": 74498506.0, "reward": 2.282519817352295, "reward_std": 0.5050063729286194, "rewards/code_complexity_reward/mean": 0.9051758050918579, "rewards/code_complexity_reward/std": 0.12724527716636658, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 423, "step_time": 50.88289766572416 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 417.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 115.705078125, "completions/mean_terminated_length": 115.705078125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2329453385900706, "epoch": 0.48346636259977194, "frac_reward_zero_std": 0.375, "grad_norm": 0.05683957412838936, "kl": 0.16857241420075297, "learning_rate": 3.0867586958121653e-06, "loss": 0.000842796522192657, "num_tokens": 74628875.0, "reward": 2.2835938930511475, "reward_std": 0.4948135018348694, "rewards/code_complexity_reward/mean": 0.89990234375, "rewards/code_complexity_reward/std": 0.12052149325609207, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 424, "step_time": 49.623203922994435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 438.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 117.435546875, "completions/mean_terminated_length": 117.435546875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24005130399018526, "epoch": 0.4846066134549601, "frac_reward_zero_std": 0.359375, "grad_norm": 0.06594999879598618, "kl": 0.1746646233368665, "learning_rate": 3.0770777754442333e-06, "loss": 0.0008733106078580022, "num_tokens": 74756590.0, "reward": 2.233691453933716, "reward_std": 0.48627206683158875, "rewards/code_complexity_reward/mean": 0.8983398675918579, "rewards/code_complexity_reward/std": 0.12776176631450653, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 425, "step_time": 44.88058257009834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 113.01171875, "completions/mean_terminated_length": 113.01171875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22876376239582896, "epoch": 0.48574686431014824, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05152444913983345, "kl": 0.16748477495275438, "learning_rate": 3.0673877059522906e-06, "loss": 0.0008373686578124762, "num_tokens": 74883092.0, "reward": 2.3667969703674316, "reward_std": 0.5134788751602173, "rewards/code_complexity_reward/mean": 0.9093749523162842, "rewards/code_complexity_reward/std": 0.10058537125587463, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 426, "step_time": 58.61482121516019 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 111.33203125, "completions/mean_terminated_length": 111.33203125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23116892180405557, "epoch": 0.4868871151653364, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.057770904153585434, "kl": 0.20288880716543645, "learning_rate": 3.057688640964934e-06, "loss": 0.001014235196635127, "num_tokens": 75005762.0, "reward": 2.282031297683716, "reward_std": 0.5086333155632019, "rewards/code_complexity_reward/mean": 0.9056640267372131, "rewards/code_complexity_reward/std": 0.133876234292984, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 427, "step_time": 70.46336170006543 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 118.955078125, "completions/mean_terminated_length": 118.18590545654297, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.22757803229615092, "epoch": 0.4880273660205245, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04811795800924301, "kl": 0.16572184406686574, "learning_rate": 3.047980734253372e-06, "loss": 0.0008286358788609505, "num_tokens": 75136203.0, "reward": 2.3095216751098633, "reward_std": 0.5327332615852356, "rewards/code_complexity_reward/mean": 0.895312488079071, "rewards/code_complexity_reward/std": 0.13489684462547302, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 428, "step_time": 55.22847595345229 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 333.0, "completions/max_terminated_length": 333.0, "completions/mean_length": 112.9140625, "completions/mean_terminated_length": 112.9140625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2342336920555681, "epoch": 0.48916761687571264, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05413978174328804, "kl": 0.17504776630084962, "learning_rate": 3.038264139728997e-06, "loss": 0.0008751060231588781, "num_tokens": 75262299.0, "reward": 2.257617235183716, "reward_std": 0.4997069835662842, "rewards/code_complexity_reward/mean": 0.8958984017372131, "rewards/code_complexity_reward/std": 0.1325010508298874, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 429, "step_time": 46.22160107642412 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 372.0, "completions/max_terminated_length": 372.0, "completions/mean_length": 106.00390625, "completions/mean_terminated_length": 106.00390625, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.24609452206641436, "epoch": 0.4903078677309008, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.05445019155740738, "kl": 0.17756430129520595, "learning_rate": 3.0285390114409353e-06, "loss": 0.0008877534419298172, "num_tokens": 75387113.0, "reward": 2.2746095657348633, "reward_std": 0.5075938701629639, "rewards/code_complexity_reward/mean": 0.9041014909744263, "rewards/code_complexity_reward/std": 0.1335308998823166, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 430, "step_time": 46.33261924888939 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 112.40625, "completions/mean_terminated_length": 112.40625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23240871005691588, "epoch": 0.49144811858608894, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.05042530596256256, "kl": 0.18588783242739737, "learning_rate": 3.018805503573612e-06, "loss": 0.0009298303630203009, "num_tokens": 75512765.0, "reward": 2.27001953125, "reward_std": 0.4904814064502716, "rewards/code_complexity_reward/mean": 0.9073241949081421, "rewards/code_complexity_reward/std": 0.11857795715332031, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 431, "step_time": 43.11223577614874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 107.1015625, "completions/mean_terminated_length": 106.30919647216797, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2368435396347195, "epoch": 0.4925883694412771, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05113636702299118, "kl": 0.18200151436030865, "learning_rate": 3.0090637704443033e-06, "loss": 0.0009102706098929048, "num_tokens": 75635205.0, "reward": 2.3194828033447266, "reward_std": 0.5133183002471924, "rewards/code_complexity_reward/mean": 0.91015625, "rewards/code_complexity_reward/std": 0.12064071744680405, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 432, "step_time": 68.64942654501647 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 104.052734375, "completions/mean_terminated_length": 104.052734375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24592346884310246, "epoch": 0.49372862029646525, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.04578899219632149, "kl": 0.19611343974247575, "learning_rate": 2.9993139665006904e-06, "loss": 0.0009805852314457297, "num_tokens": 75755640.0, "reward": 2.2889161109924316, "reward_std": 0.4955284595489502, "rewards/code_complexity_reward/mean": 0.910351574420929, "rewards/code_complexity_reward/std": 0.11130023747682571, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 433, "step_time": 36.65891058743 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 107.21484375, "completions/mean_terminated_length": 107.21484375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.23805980291217566, "epoch": 0.49486887115165334, "frac_reward_zero_std": 0.5, "grad_norm": 0.052785009145736694, "kl": 0.18794721306767315, "learning_rate": 2.989556246318412e-06, "loss": 0.0009396467939950526, "num_tokens": 75877050.0, "reward": 2.320117235183716, "reward_std": 0.529367983341217, "rewards/code_complexity_reward/mean": 0.9037109613418579, "rewards/code_complexity_reward/std": 0.13959649205207825, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 434, "step_time": 48.030887100845575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 112.609375, "completions/mean_terminated_length": 111.82778930664062, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23472063126973808, "epoch": 0.4960091220068415, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.059337712824344635, "kl": 0.17115302791353315, "learning_rate": 2.9797907645986124e-06, "loss": 0.0008558117551729083, "num_tokens": 76002430.0, "reward": 2.3250489234924316, "reward_std": 0.4958080053329468, "rewards/code_complexity_reward/mean": 0.9108397960662842, "rewards/code_complexity_reward/std": 0.09577512741088867, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 435, "step_time": 49.675271577201784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 362.0, "completions/max_terminated_length": 362.0, "completions/mean_length": 111.75, "completions/mean_terminated_length": 111.75, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.23212554142810404, "epoch": 0.49714937286202965, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.04796263948082924, "kl": 0.1721554072573781, "learning_rate": 2.9700176761654875e-06, "loss": 0.0008605992770753801, "num_tokens": 76129278.0, "reward": 2.2867188453674316, "reward_std": 0.4797864854335785, "rewards/code_complexity_reward/mean": 0.909375011920929, "rewards/code_complexity_reward/std": 0.09666655212640762, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 436, "step_time": 48.72444056998938 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 400.0, "completions/max_terminated_length": 400.0, "completions/mean_length": 109.365234375, "completions/mean_terminated_length": 109.365234375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23799709184095263, "epoch": 0.4982896237172178, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.04959632828831673, "kl": 0.19064749660901725, "learning_rate": 2.960237135963834e-06, "loss": 0.0009531002142466605, "num_tokens": 76255205.0, "reward": 2.258349657058716, "reward_std": 0.4649978280067444, "rewards/code_complexity_reward/mean": 0.9144530892372131, "rewards/code_complexity_reward/std": 0.10139001160860062, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 437, "step_time": 60.394335077144206 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 352.0, "completions/max_terminated_length": 352.0, "completions/mean_length": 104.69140625, "completions/mean_terminated_length": 104.69140625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2278820078354329, "epoch": 0.49942987457240595, "frac_reward_zero_std": 0.53125, "grad_norm": 0.054004937410354614, "kl": 0.18087539612315595, "learning_rate": 2.9504492990565885e-06, "loss": 0.0009041979210451245, "num_tokens": 76376803.0, "reward": 2.3523926734924316, "reward_std": 0.4982163906097412, "rewards/code_complexity_reward/mean": 0.917675793170929, "rewards/code_complexity_reward/std": 0.08631999790668488, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 438, "step_time": 54.89290520362556 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 113.94140625, "completions/mean_terminated_length": 113.94140625, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.24298643646761775, "epoch": 0.500570125427594, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.05519157275557518, "kl": 0.1715865439036861, "learning_rate": 2.9406543206223735e-06, "loss": 0.0008578792912885547, "num_tokens": 76503065.0, "reward": 2.2601563930511475, "reward_std": 0.496725469827652, "rewards/code_complexity_reward/mean": 0.904296875, "rewards/code_complexity_reward/std": 0.1299220472574234, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 439, "step_time": 46.89895376842469 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 110.01171875, "completions/mean_terminated_length": 110.01171875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24233383871614933, "epoch": 0.5017103762827823, "frac_reward_zero_std": 0.546875, "grad_norm": 0.046482380479574203, "kl": 0.17824005032889545, "learning_rate": 2.930852355953034e-06, "loss": 0.0008911177865229547, "num_tokens": 76626371.0, "reward": 2.2728517055511475, "reward_std": 0.4727250337600708, "rewards/code_complexity_reward/mean": 0.9130859375, "rewards/code_complexity_reward/std": 0.10109309107065201, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 440, "step_time": 46.74690879229456 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 320.0, "completions/max_terminated_length": 320.0, "completions/mean_length": 104.013671875, "completions/mean_terminated_length": 104.013671875, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.23722872277721763, "epoch": 0.5028506271379704, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.06542715430259705, "kl": 0.18040235224179924, "learning_rate": 2.9210435604511756e-06, "loss": 0.0009018882410600781, "num_tokens": 76747494.0, "reward": 2.31103515625, "reward_std": 0.5299081206321716, "rewards/code_complexity_reward/mean": 0.9053710699081421, "rewards/code_complexity_reward/std": 0.13994497060775757, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 441, "step_time": 36.21777518000454 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 404.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 108.828125, "completions/mean_terminated_length": 108.828125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24474742566235363, "epoch": 0.5039908779931584, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.06162301450967789, "kl": 0.1944248175714165, "learning_rate": 2.9112280896277017e-06, "loss": 0.0009722596732899547, "num_tokens": 76872770.0, "reward": 2.2455568313598633, "reward_std": 0.4905526638031006, "rewards/code_complexity_reward/mean": 0.900683581829071, "rewards/code_complexity_reward/std": 0.1362573504447937, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 442, "step_time": 53.981163466349244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 111.666015625, "completions/mean_terminated_length": 111.666015625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24319406016729772, "epoch": 0.5051311288483467, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04553794860839844, "kl": 0.17779007519129664, "learning_rate": 2.90140609909935e-06, "loss": 0.0008889483287930489, "num_tokens": 76999467.0, "reward": 2.2851076126098633, "reward_std": 0.5062903761863708, "rewards/code_complexity_reward/mean": 0.9001953601837158, "rewards/code_complexity_reward/std": 0.12590168416500092, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 443, "step_time": 35.791216298006475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 102.9375, "completions/mean_terminated_length": 102.9375, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.23722713952884078, "epoch": 0.5062713797035348, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06138548627495766, "kl": 0.21120882243849337, "learning_rate": 2.8915777445862185e-06, "loss": 0.00105600047390908, "num_tokens": 77120379.0, "reward": 2.279101848602295, "reward_std": 0.5161887407302856, "rewards/code_complexity_reward/mean": 0.9027343988418579, "rewards/code_complexity_reward/std": 0.14325112104415894, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 444, "step_time": 44.23997884243727 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 114.0625, "completions/mean_terminated_length": 113.28376007080078, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2372206838335842, "epoch": 0.507411630558723, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.05490034446120262, "kl": 0.17666995828039944, "learning_rate": 2.8817431819093065e-06, "loss": 0.0008831472368910909, "num_tokens": 77246383.0, "reward": 2.2513673305511475, "reward_std": 0.4945550858974457, "rewards/code_complexity_reward/mean": 0.9044921398162842, "rewards/code_complexity_reward/std": 0.1305915117263794, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 445, "step_time": 57.94060829188675 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 108.736328125, "completions/mean_terminated_length": 108.736328125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24337741290219128, "epoch": 0.508551881413911, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.05870966613292694, "kl": 0.1948393762577325, "learning_rate": 2.8719025669880357e-06, "loss": 0.0009741566027514637, "num_tokens": 77371124.0, "reward": 2.2030272483825684, "reward_std": 0.44374942779541016, "rewards/code_complexity_reward/mean": 0.9111328125, "rewards/code_complexity_reward/std": 0.11767931282520294, "rewards/code_execution_reward/mean": 0.19921875, "rewards/code_execution_reward/std": 0.39980348944664, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 446, "step_time": 45.396877732127905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 107.732421875, "completions/mean_terminated_length": 107.732421875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22862260858528316, "epoch": 0.5096921322690992, "frac_reward_zero_std": 0.578125, "grad_norm": 0.045311663299798965, "kl": 0.18426130444277078, "learning_rate": 2.8620560558377825e-06, "loss": 0.0009213717421516776, "num_tokens": 77495375.0, "reward": 2.33251953125, "reward_std": 0.49610233306884766, "rewards/code_complexity_reward/mean": 0.9170898199081421, "rewards/code_complexity_reward/std": 0.09669654816389084, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 447, "step_time": 43.39699338003993 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 107.2421875, "completions/mean_terminated_length": 107.2421875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24192879325710237, "epoch": 0.5108323831242874, "frac_reward_zero_std": 0.546875, "grad_norm": 0.04553447291254997, "kl": 0.18491268903017044, "learning_rate": 2.8522038045674026e-06, "loss": 0.0009245827095583081, "num_tokens": 77618275.0, "reward": 2.263476848602295, "reward_std": 0.4845196604728699, "rewards/code_complexity_reward/mean": 0.9105468392372131, "rewards/code_complexity_reward/std": 0.12113332003355026, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 448, "step_time": 35.501124404370785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 329.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 105.197265625, "completions/mean_terminated_length": 105.197265625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24366810615174472, "epoch": 0.5119726339794755, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05404802784323692, "kl": 0.18897520960308611, "learning_rate": 2.8423459693767586e-06, "loss": 0.0009448027121834457, "num_tokens": 77740568.0, "reward": 2.259765625, "reward_std": 0.4957770109176636, "rewards/code_complexity_reward/mean": 0.9068359136581421, "rewards/code_complexity_reward/std": 0.13030202686786652, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 449, "step_time": 36.800744803622365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 112.123046875, "completions/mean_terminated_length": 112.123046875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23195182345807552, "epoch": 0.5131128848346637, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05356838181614876, "kl": 0.18279191246256232, "learning_rate": 2.8324827065542405e-06, "loss": 0.0009140484035015106, "num_tokens": 77868659.0, "reward": 2.2997074127197266, "reward_std": 0.5133491158485413, "rewards/code_complexity_reward/mean": 0.89794921875, "rewards/code_complexity_reward/std": 0.1251249611377716, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 450, "step_time": 37.551959845237434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 110.357421875, "completions/mean_terminated_length": 110.357421875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23220455134287477, "epoch": 0.5142531356898518, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.06168607249855995, "kl": 0.1933183983201161, "learning_rate": 2.822614172474289e-06, "loss": 0.0009664876852184534, "num_tokens": 77992646.0, "reward": 2.261963129043579, "reward_std": 0.49553754925727844, "rewards/code_complexity_reward/mean": 0.9014648199081421, "rewards/code_complexity_reward/std": 0.1284896582365036, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 451, "step_time": 40.66366094723344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 417.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 108.69140625, "completions/mean_terminated_length": 108.69140625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24710508878342807, "epoch": 0.5153933865450399, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.04915876314043999, "kl": 0.1869146975222975, "learning_rate": 2.8127405235949173e-06, "loss": 0.0009346554288640618, "num_tokens": 78115244.0, "reward": 2.28515625, "reward_std": 0.4739644229412079, "rewards/code_complexity_reward/mean": 0.9117187261581421, "rewards/code_complexity_reward/std": 0.09746982157230377, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 452, "step_time": 54.116836440749466 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 460.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 108.94140625, "completions/mean_terminated_length": 108.94140625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23937284969724715, "epoch": 0.5165336374002281, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.138168603181839, "kl": 0.1843214735854417, "learning_rate": 2.80286191645523e-06, "loss": 0.0009214976453222334, "num_tokens": 78242554.0, "reward": 2.3004393577575684, "reward_std": 0.5046806931495667, "rewards/code_complexity_reward/mean": 0.90283203125, "rewards/code_complexity_reward/std": 0.1251096874475479, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 453, "step_time": 54.62842313479632 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 393.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 107.361328125, "completions/mean_terminated_length": 107.361328125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2304025285411626, "epoch": 0.5176738882554162, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.047150373458862305, "kl": 0.19458393286913633, "learning_rate": 2.792978507672941e-06, "loss": 0.0009726958815008402, "num_tokens": 78367459.0, "reward": 2.301562786102295, "reward_std": 0.4796622097492218, "rewards/code_complexity_reward/mean": 0.9161132574081421, "rewards/code_complexity_reward/std": 0.09011370688676834, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.029158055782318115, "step": 454, "step_time": 58.838712693192065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 314.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 102.9296875, "completions/mean_terminated_length": 102.9296875, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.23798142233863473, "epoch": 0.5188141391106044, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.05385367199778557, "kl": 0.19722217437811196, "learning_rate": 2.7830904539418884e-06, "loss": 0.0009861319558694959, "num_tokens": 78488879.0, "reward": 2.305469036102295, "reward_std": 0.530405580997467, "rewards/code_complexity_reward/mean": 0.9017578363418579, "rewards/code_complexity_reward/std": 0.14254753291606903, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 455, "step_time": 46.344327511265874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 109.431640625, "completions/mean_terminated_length": 109.431640625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2480546950828284, "epoch": 0.5199543899657925, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.05713410675525665, "kl": 0.2024509753100574, "learning_rate": 2.7731979120295564e-06, "loss": 0.0010122009553015232, "num_tokens": 78611576.0, "reward": 2.2076172828674316, "reward_std": 0.4651353657245636, "rewards/code_complexity_reward/mean": 0.900585949420929, "rewards/code_complexity_reward/std": 0.12916125357151031, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 456, "step_time": 37.08164987899363 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 108.115234375, "completions/mean_terminated_length": 108.115234375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2443903994280845, "epoch": 0.5210946408209807, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.05257890000939369, "kl": 0.19610307621769607, "learning_rate": 2.763301038774583e-06, "loss": 0.0009805620647966862, "num_tokens": 78735783.0, "reward": 2.2540040016174316, "reward_std": 0.51125168800354, "rewards/code_complexity_reward/mean": 0.9025390148162842, "rewards/code_complexity_reward/std": 0.14592775702476501, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.024685947224497795, "step": 457, "step_time": 48.97916947118938 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 331.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 104.044921875, "completions/mean_terminated_length": 104.044921875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23573180474340916, "epoch": 0.5222348916761688, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.0471046045422554, "kl": 0.19235390261746943, "learning_rate": 2.7533999910842766e-06, "loss": 0.000961745681706816, "num_tokens": 78859606.0, "reward": 2.302734375, "reward_std": 0.4634934663772583, "rewards/code_complexity_reward/mean": 0.9253906011581421, "rewards/code_complexity_reward/std": 0.06952342391014099, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 458, "step_time": 37.562405202537775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 105.13671875, "completions/mean_terminated_length": 105.13671875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2342922897078097, "epoch": 0.5233751425313569, "frac_reward_zero_std": 0.484375, "grad_norm": 0.10220733284950256, "kl": 0.27697004238143563, "learning_rate": 2.743494925932129e-06, "loss": 0.001383814262226224, "num_tokens": 78980996.0, "reward": 2.358691453933716, "reward_std": 0.5119751691818237, "rewards/code_complexity_reward/mean": 0.9159179925918579, "rewards/code_complexity_reward/std": 0.09376624971628189, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 459, "step_time": 36.44097064062953 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 368.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 101.330078125, "completions/mean_terminated_length": 101.330078125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24126577726565301, "epoch": 0.5245153933865451, "frac_reward_zero_std": 0.546875, "grad_norm": 0.0614197738468647, "kl": 0.21519569330848753, "learning_rate": 2.7335860003553257e-06, "loss": 0.0010758002754300833, "num_tokens": 79102709.0, "reward": 2.25390625, "reward_std": 0.520475447177887, "rewards/code_complexity_reward/mean": 0.9019531011581421, "rewards/code_complexity_reward/std": 0.15723296999931335, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 460, "step_time": 57.61878322251141 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 330.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 106.107421875, "completions/mean_terminated_length": 106.107421875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23129998706281185, "epoch": 0.5256556442417332, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.04950848966836929, "kl": 0.1976704120170325, "learning_rate": 2.723673371452254e-06, "loss": 0.000988383311778307, "num_tokens": 79223852.0, "reward": 2.2648439407348633, "reward_std": 0.5272834300994873, "rewards/code_complexity_reward/mean": 0.896289050579071, "rewards/code_complexity_reward/std": 0.15532271564006805, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 461, "step_time": 41.72265569772571 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 330.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 103.6171875, "completions/mean_terminated_length": 103.6171875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24190597445704043, "epoch": 0.5267958950969214, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.0474436916410923, "kl": 0.19521035603247583, "learning_rate": 2.713757196380017e-06, "loss": 0.0009759720414876938, "num_tokens": 79345692.0, "reward": 2.239258050918579, "reward_std": 0.5035896897315979, "rewards/code_complexity_reward/mean": 0.9048827886581421, "rewards/code_complexity_reward/std": 0.1479659378528595, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 462, "step_time": 36.02973040565848 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 505.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 103.705078125, "completions/mean_terminated_length": 103.705078125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23796756798401475, "epoch": 0.5279361459521095, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.04722388833761215, "kl": 0.20292710652574897, "learning_rate": 2.70383763235194e-06, "loss": 0.0010146787390112877, "num_tokens": 79466941.0, "reward": 2.2798829078674316, "reward_std": 0.4924534559249878, "rewards/code_complexity_reward/mean": 0.9123046398162842, "rewards/code_complexity_reward/std": 0.11664320528507233, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 463, "step_time": 58.278388718143106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 103.16796875, "completions/mean_terminated_length": 101.56471252441406, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23628454795107245, "epoch": 0.5290763968072976, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05134318396449089, "kl": 0.20514730596914887, "learning_rate": 2.693914836635076e-06, "loss": 0.0010256264358758926, "num_tokens": 79586439.0, "reward": 2.282470703125, "reward_std": 0.5063048005104065, "rewards/code_complexity_reward/mean": 0.9126953482627869, "rewards/code_complexity_reward/std": 0.13068749010562897, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 464, "step_time": 55.95567652769387 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 296.0, "completions/max_terminated_length": 296.0, "completions/mean_length": 103.09375, "completions/mean_terminated_length": 103.09375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24929023371078074, "epoch": 0.5302166476624858, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05481567978858948, "kl": 0.2142236630897969, "learning_rate": 2.6839889665477144e-06, "loss": 0.0010711699724197388, "num_tokens": 79709211.0, "reward": 2.319824457168579, "reward_std": 0.4949081242084503, "rewards/code_complexity_reward/mean": 0.9180663824081421, "rewards/code_complexity_reward/std": 0.09682216495275497, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 465, "step_time": 34.6630142070353 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 104.197265625, "completions/mean_terminated_length": 104.197265625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23971208417788148, "epoch": 0.5313568985176739, "frac_reward_zero_std": 0.53125, "grad_norm": 0.059798385947942734, "kl": 0.2232184454333037, "learning_rate": 2.6740601794568866e-06, "loss": 0.0011162813752889633, "num_tokens": 79830872.0, "reward": 2.3226561546325684, "reward_std": 0.5235681533813477, "rewards/code_complexity_reward/mean": 0.9033203125, "rewards/code_complexity_reward/std": 0.13435614109039307, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 466, "step_time": 42.00783613976091 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 101.087890625, "completions/mean_terminated_length": 101.087890625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23204103973694146, "epoch": 0.5324971493728621, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.04642561823129654, "kl": 0.20075751328840852, "learning_rate": 2.664128632775871e-06, "loss": 0.0010038872715085745, "num_tokens": 79950317.0, "reward": 2.30322265625, "reward_std": 0.4912585914134979, "rewards/code_complexity_reward/mean": 0.9170898199081421, "rewards/code_complexity_reward/std": 0.10182256251573563, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 467, "step_time": 49.30427387915552 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 447.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 103.998046875, "completions/mean_terminated_length": 103.998046875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23964713653549552, "epoch": 0.5336374002280502, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.05800063535571098, "kl": 0.20128846145235002, "learning_rate": 2.6541944839616957e-06, "loss": 0.0010064702946692705, "num_tokens": 80070252.0, "reward": 2.319824457168579, "reward_std": 0.5347444415092468, "rewards/code_complexity_reward/mean": 0.9024413824081421, "rewards/code_complexity_reward/std": 0.14830712974071503, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 468, "step_time": 52.26042387075722 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 305.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 98.65234375, "completions/mean_terminated_length": 98.65234375, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.2314787234645337, "epoch": 0.5347776510832383, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05521227419376373, "kl": 0.19462514226324856, "learning_rate": 2.644257890512646e-06, "loss": 0.0009731660829856992, "num_tokens": 80188422.0, "reward": 2.376757860183716, "reward_std": 0.5134272575378418, "rewards/code_complexity_reward/mean": 0.9212890863418579, "rewards/code_complexity_reward/std": 0.09117572009563446, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 469, "step_time": 35.6777740707621 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 284.0, "completions/max_terminated_length": 284.0, "completions/mean_length": 102.912109375, "completions/mean_terminated_length": 102.912109375, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.24123973515816033, "epoch": 0.5359179019384265, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.0720754861831665, "kl": 0.19782517524436116, "learning_rate": 2.634319009965762e-06, "loss": 0.0009891776135191321, "num_tokens": 80310925.0, "reward": 2.2364745140075684, "reward_std": 0.47339582443237305, "rewards/code_complexity_reward/mean": 0.9130859375, "rewards/code_complexity_reward/std": 0.12845860421657562, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 470, "step_time": 33.6600648406893 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 459.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 107.318359375, "completions/mean_terminated_length": 107.318359375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23650883347727358, "epoch": 0.5370581527936146, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.04814857244491577, "kl": 0.1962862324435264, "learning_rate": 2.6243779998943496e-06, "loss": 0.0009814836084842682, "num_tokens": 80435988.0, "reward": 2.3001952171325684, "reward_std": 0.46917712688446045, "rewards/code_complexity_reward/mean": 0.9208984375, "rewards/code_complexity_reward/std": 0.07790538668632507, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 471, "step_time": 50.753160178661346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 341.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 112.865234375, "completions/mean_terminated_length": 112.865234375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24503588629886508, "epoch": 0.5381984036488028, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.06196412071585655, "kl": 0.18641503620892763, "learning_rate": 2.614435017905469e-06, "loss": 0.0009319972014054656, "num_tokens": 80562827.0, "reward": 2.259570598602295, "reward_std": 0.4832884967327118, "rewards/code_complexity_reward/mean": 0.9095702767372131, "rewards/code_complexity_reward/std": 0.11367465555667877, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 472, "step_time": 46.66227752901614 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 435.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 105.205078125, "completions/mean_terminated_length": 105.205078125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.24265872803516686, "epoch": 0.5393386545039909, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.07971926033496857, "kl": 0.22289951471611857, "learning_rate": 2.6044902216374497e-06, "loss": 0.0011143251322209835, "num_tokens": 80684944.0, "reward": 2.277099847793579, "reward_std": 0.46737414598464966, "rewards/code_complexity_reward/mean": 0.9136718511581421, "rewards/code_complexity_reward/std": 0.10135381668806076, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 473, "step_time": 51.75458444561809 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 105.099609375, "completions/mean_terminated_length": 105.099609375, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.23523924592882395, "epoch": 0.540478905359179, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.050982553511857986, "kl": 0.2005504653789103, "learning_rate": 2.5945437687573816e-06, "loss": 0.0010029502445831895, "num_tokens": 80808095.0, "reward": 2.2518556118011475, "reward_std": 0.5017669200897217, "rewards/code_complexity_reward/mean": 0.906933605670929, "rewards/code_complexity_reward/std": 0.1404346376657486, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 474, "step_time": 50.78532674536109 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 413.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 108.630859375, "completions/mean_terminated_length": 108.630859375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2371052554808557, "epoch": 0.5416191562143672, "frac_reward_zero_std": 0.53125, "grad_norm": 0.06034395098686218, "kl": 0.21278914506547153, "learning_rate": 2.584595816958621e-06, "loss": 0.0010636176448315382, "num_tokens": 80932726.0, "reward": 2.2267580032348633, "reward_std": 0.4289851188659668, "rewards/code_complexity_reward/mean": 0.911914050579071, "rewards/code_complexity_reward/std": 0.08838985115289688, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 475, "step_time": 41.58961428608745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 283.0, "completions/max_terminated_length": 283.0, "completions/mean_length": 100.8203125, "completions/mean_terminated_length": 100.8203125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24426949280314147, "epoch": 0.5427594070695553, "frac_reward_zero_std": 0.703125, "grad_norm": 0.041792791336774826, "kl": 0.2043659589253366, "learning_rate": 2.574646523958288e-06, "loss": 0.00102186796721071, "num_tokens": 81052842.0, "reward": 2.263671875, "reward_std": 0.46399587392807007, "rewards/code_complexity_reward/mean": 0.9185546636581421, "rewards/code_complexity_reward/std": 0.09974262863397598, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 476, "step_time": 33.96289095655084 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 286.0, "completions/max_terminated_length": 286.0, "completions/mean_length": 101.7265625, "completions/mean_terminated_length": 101.7265625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23327317042276263, "epoch": 0.5438996579247435, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.05613543465733528, "kl": 0.2011943224351853, "learning_rate": 2.564696047494765e-06, "loss": 0.001006135600619018, "num_tokens": 81172870.0, "reward": 2.300830364227295, "reward_std": 0.4948863089084625, "rewards/code_complexity_reward/mean": 0.9188476204872131, "rewards/code_complexity_reward/std": 0.11662759631872177, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 477, "step_time": 37.10273604467511 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 102.240234375, "completions/mean_terminated_length": 102.240234375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.241031841840595, "epoch": 0.5450399087799316, "frac_reward_zero_std": 0.515625, "grad_norm": 0.07340163737535477, "kl": 0.24986138381063938, "learning_rate": 2.5547445453252e-06, "loss": 0.001248850952833891, "num_tokens": 81295201.0, "reward": 2.2797365188598633, "reward_std": 0.49473825097084045, "rewards/code_complexity_reward/mean": 0.912402331829071, "rewards/code_complexity_reward/std": 0.1255808025598526, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 478, "step_time": 46.131964180618525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 379.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 99.458984375, "completions/mean_terminated_length": 99.458984375, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.23745072237215936, "epoch": 0.5461801596351197, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.054194312542676926, "kl": 0.19351855502463877, "learning_rate": 2.5447921752230003e-06, "loss": 0.0009676806512288749, "num_tokens": 81414572.0, "reward": 2.3511719703674316, "reward_std": 0.5445501208305359, "rewards/code_complexity_reward/mean": 0.908398449420929, "rewards/code_complexity_reward/std": 0.141862690448761, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 479, "step_time": 38.97613990493119 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 101.287109375, "completions/mean_terminated_length": 101.287109375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22977805277332664, "epoch": 0.5473204104903079, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.05853716656565666, "kl": 0.2056430580560118, "learning_rate": 2.5348390949753343e-06, "loss": 0.0010281086433678865, "num_tokens": 81535859.0, "reward": 2.3287110328674316, "reward_std": 0.47850659489631653, "rewards/code_complexity_reward/mean": 0.922070324420929, "rewards/code_complexity_reward/std": 0.07883204519748688, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 480, "step_time": 37.07260154560208 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 301.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 99.755859375, "completions/mean_terminated_length": 99.755859375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24297057488001883, "epoch": 0.548460661345496, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.053734421730041504, "kl": 0.21095470082946122, "learning_rate": 2.5248854623806297e-06, "loss": 0.001054772175848484, "num_tokens": 81656062.0, "reward": 2.2529296875, "reward_std": 0.4891696274280548, "rewards/code_complexity_reward/mean": 0.9126952886581421, "rewards/code_complexity_reward/std": 0.12857399880886078, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 481, "step_time": 44.837462102063 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 109.646484375, "completions/mean_terminated_length": 109.646484375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2347847456112504, "epoch": 0.5496009122006842, "frac_reward_zero_std": 0.515625, "grad_norm": 0.055256955325603485, "kl": 0.19515642570331693, "learning_rate": 2.514931435246071e-06, "loss": 0.0009757272200658917, "num_tokens": 81779669.0, "reward": 2.285937547683716, "reward_std": 0.508823812007904, "rewards/code_complexity_reward/mean": 0.9066406488418579, "rewards/code_complexity_reward/std": 0.13298781216144562, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 482, "step_time": 60.42930214293301 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 101.91015625, "completions/mean_terminated_length": 101.10762786865234, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24606741918250918, "epoch": 0.5507411630558723, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05436175316572189, "kl": 0.19938672054558992, "learning_rate": 2.504977171385098e-06, "loss": 0.0009970113169401884, "num_tokens": 81900399.0, "reward": 2.3194825649261475, "reward_std": 0.5319927334785461, "rewards/code_complexity_reward/mean": 0.90966796875, "rewards/code_complexity_reward/std": 0.13566277921199799, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 483, "step_time": 50.730299120768905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 302.0, "completions/max_terminated_length": 302.0, "completions/mean_length": 103.169921875, "completions/mean_terminated_length": 103.169921875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24591229925863445, "epoch": 0.5518814139110604, "frac_reward_zero_std": 0.546875, "grad_norm": 0.06667172163724899, "kl": 0.2652133568190038, "learning_rate": 2.4950228286149028e-06, "loss": 0.0013258474646136165, "num_tokens": 82022382.0, "reward": 2.270263671875, "reward_std": 0.48857375979423523, "rewards/code_complexity_reward/mean": 0.9126952886581421, "rewards/code_complexity_reward/std": 0.11664320528507233, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 484, "step_time": 43.857896791771054 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 299.0, "completions/max_terminated_length": 299.0, "completions/mean_length": 99.4453125, "completions/mean_terminated_length": 99.4453125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2328601493500173, "epoch": 0.5530216647662486, "frac_reward_zero_std": 0.546875, "grad_norm": 0.060617588460445404, "kl": 0.21456570993177593, "learning_rate": 2.48506856475393e-06, "loss": 0.0010730898939073086, "num_tokens": 82141218.0, "reward": 2.3400392532348633, "reward_std": 0.49255308508872986, "rewards/code_complexity_reward/mean": 0.92041015625, "rewards/code_complexity_reward/std": 0.0953335240483284, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 485, "step_time": 35.112833293154836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 107.5625, "completions/mean_terminated_length": 106.77103424072266, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.24050318775698543, "epoch": 0.5541619156214367, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.04512679576873779, "kl": 0.20051782974041998, "learning_rate": 2.475114537619371e-06, "loss": 0.0010026412783190608, "num_tokens": 82265370.0, "reward": 2.3414554595947266, "reward_std": 0.5196966528892517, "rewards/code_complexity_reward/mean": 0.91455078125, "rewards/code_complexity_reward/std": 0.11939259618520737, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 486, "step_time": 50.00972652994096 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 284.0, "completions/max_terminated_length": 284.0, "completions/mean_length": 103.095703125, "completions/mean_terminated_length": 103.095703125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24041444156318903, "epoch": 0.5553021664766249, "frac_reward_zero_std": 0.53125, "grad_norm": 0.06369040906429291, "kl": 0.203793419059366, "learning_rate": 2.4651609050246674e-06, "loss": 0.0010187751613557339, "num_tokens": 82388499.0, "reward": 2.2286620140075684, "reward_std": 0.4761432409286499, "rewards/code_complexity_reward/mean": 0.9130859375, "rewards/code_complexity_reward/std": 0.13098546862602234, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 487, "step_time": 36.43831306044012 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 100.12890625, "completions/mean_terminated_length": 100.12890625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23604215867817402, "epoch": 0.556442417331813, "frac_reward_zero_std": 0.578125, "grad_norm": 0.06658715009689331, "kl": 0.20432009245269, "learning_rate": 2.4552078247770005e-06, "loss": 0.0010215772781521082, "num_tokens": 82506625.0, "reward": 2.311523675918579, "reward_std": 0.5058406591415405, "rewards/code_complexity_reward/mean": 0.917285144329071, "rewards/code_complexity_reward/std": 0.11462971568107605, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.019099153578281403, "step": 488, "step_time": 38.241735867224634 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 99.74609375, "completions/mean_terminated_length": 99.74609375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2407698428723961, "epoch": 0.5575826681870011, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05706597492098808, "kl": 0.20220001484267414, "learning_rate": 2.4452554546748008e-06, "loss": 0.0010109103750437498, "num_tokens": 82626527.0, "reward": 2.3606934547424316, "reward_std": 0.5185022354125977, "rewards/code_complexity_reward/mean": 0.9113280773162842, "rewards/code_complexity_reward/std": 0.11242998391389847, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 489, "step_time": 46.98257016763091 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 101.373046875, "completions/mean_terminated_length": 101.373046875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2364902680274099, "epoch": 0.5587229190421893, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05135255306959152, "kl": 0.2132515364792198, "learning_rate": 2.4353039525052354e-06, "loss": 0.0010661170817911625, "num_tokens": 82748058.0, "reward": 2.304004192352295, "reward_std": 0.5027114152908325, "rewards/code_complexity_reward/mean": 0.9129883050918579, "rewards/code_complexity_reward/std": 0.12172221392393112, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 490, "step_time": 40.65447751060128 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 427.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 105.42578125, "completions/mean_terminated_length": 105.42578125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24162239930592477, "epoch": 0.5598631698973774, "frac_reward_zero_std": 0.5, "grad_norm": 0.060415685176849365, "kl": 0.1981296851299703, "learning_rate": 2.425353476041713e-06, "loss": 0.0009906892664730549, "num_tokens": 82871016.0, "reward": 2.2398438453674316, "reward_std": 0.4855690598487854, "rewards/code_complexity_reward/mean": 0.9083983898162842, "rewards/code_complexity_reward/std": 0.1330721229314804, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 491, "step_time": 80.06854197196662 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 490.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 108.99609375, "completions/mean_terminated_length": 108.99609375, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.24538918305188417, "epoch": 0.5610034207525656, "frac_reward_zero_std": 0.53125, "grad_norm": 0.0553731769323349, "kl": 0.212768301833421, "learning_rate": 2.4154041830413803e-06, "loss": 0.0010638711974024773, "num_tokens": 82995894.0, "reward": 2.2177248001098633, "reward_std": 0.49477165937423706, "rewards/code_complexity_reward/mean": 0.897265613079071, "rewards/code_complexity_reward/std": 0.15360108017921448, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 492, "step_time": 56.60814825166017 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 271.0, "completions/max_terminated_length": 271.0, "completions/mean_length": 103.25, "completions/mean_terminated_length": 103.25, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23788962373510003, "epoch": 0.5621436716077537, "frac_reward_zero_std": 0.5, "grad_norm": 0.06074690818786621, "kl": 0.23771553696133196, "learning_rate": 2.4054562312426193e-06, "loss": 0.0011884437408298254, "num_tokens": 83115966.0, "reward": 2.215576171875, "reward_std": 0.4934728741645813, "rewards/code_complexity_reward/mean": 0.9043945074081421, "rewards/code_complexity_reward/std": 0.15142902731895447, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 493, "step_time": 40.15107671357691 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 372.0, "completions/max_terminated_length": 372.0, "completions/mean_length": 104.68359375, "completions/mean_terminated_length": 104.68359375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2445000831503421, "epoch": 0.5632839224629419, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.056761108338832855, "kl": 0.21161409048363566, "learning_rate": 2.395509778362552e-06, "loss": 0.001058161724358797, "num_tokens": 83239692.0, "reward": 2.2462892532348633, "reward_std": 0.49907538294792175, "rewards/code_complexity_reward/mean": 0.903124988079071, "rewards/code_complexity_reward/std": 0.1465853899717331, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 494, "step_time": 52.31133410334587 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 106.1875, "completions/mean_terminated_length": 106.1875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2509633169975132, "epoch": 0.56442417331813, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.09712827205657959, "kl": 0.2561535146087408, "learning_rate": 2.3855649820945313e-06, "loss": 0.0012809366453438997, "num_tokens": 83362036.0, "reward": 2.241943359375, "reward_std": 0.46415001153945923, "rewards/code_complexity_reward/mean": 0.9136718511581421, "rewards/code_complexity_reward/std": 0.10844296216964722, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 495, "step_time": 39.29651462379843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 104.470703125, "completions/mean_terminated_length": 104.470703125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23054820066317916, "epoch": 0.5655644241733181, "frac_reward_zero_std": 0.578125, "grad_norm": 0.053985901176929474, "kl": 0.20186702464707196, "learning_rate": 2.375622000105651e-06, "loss": 0.0010094116441905499, "num_tokens": 83485465.0, "reward": 2.2847657203674316, "reward_std": 0.5039689540863037, "rewards/code_complexity_reward/mean": 0.9093749523162842, "rewards/code_complexity_reward/std": 0.12713924050331116, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 496, "step_time": 45.31286651175469 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 331.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 101.818359375, "completions/mean_terminated_length": 101.818359375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24331930559128523, "epoch": 0.5667046750285063, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.06092429533600807, "kl": 0.2139163964893669, "learning_rate": 2.3656809900342383e-06, "loss": 0.0010696192039176822, "num_tokens": 83606076.0, "reward": 2.2916994094848633, "reward_std": 0.5083495378494263, "rewards/code_complexity_reward/mean": 0.913378894329071, "rewards/code_complexity_reward/std": 0.1287897378206253, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 497, "step_time": 62.34046792984009 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 362.0, "completions/max_terminated_length": 362.0, "completions/mean_length": 101.15625, "completions/mean_terminated_length": 101.15625, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.23735854332335293, "epoch": 0.5678449258836944, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.06598976254463196, "kl": 0.2028232249431312, "learning_rate": 2.355742109487355e-06, "loss": 0.0010141560342162848, "num_tokens": 83725264.0, "reward": 2.356982707977295, "reward_std": 0.5035794973373413, "rewards/code_complexity_reward/mean": 0.9195312261581421, "rewards/code_complexity_reward/std": 0.09802039712667465, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 498, "step_time": 47.23018692061305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 439.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 103.71875, "completions/mean_terminated_length": 103.71875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24714827351272106, "epoch": 0.5689851767388826, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.05814561992883682, "kl": 0.21171522163785994, "learning_rate": 2.3458055160383055e-06, "loss": 0.0010584035189822316, "num_tokens": 83846576.0, "reward": 2.3082032203674316, "reward_std": 0.519060492515564, "rewards/code_complexity_reward/mean": 0.9064452648162842, "rewards/code_complexity_reward/std": 0.13299742341041565, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 499, "step_time": 50.28518621902913 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 102.439453125, "completions/mean_terminated_length": 102.439453125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23895062366500497, "epoch": 0.5701254275940707, "frac_reward_zero_std": 0.578125, "grad_norm": 0.048550210893154144, "kl": 0.20230866805650294, "learning_rate": 2.33587136722413e-06, "loss": 0.0010114770848304033, "num_tokens": 83967857.0, "reward": 2.3571290969848633, "reward_std": 0.529624342918396, "rewards/code_complexity_reward/mean": 0.9114258289337158, "rewards/code_complexity_reward/std": 0.12489259243011475, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 500, "step_time": 43.90329029597342 }, { "epoch": 0.5701254275940707, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0025, "eval_completions/max_length": 152.76, "eval_completions/max_terminated_length": 152.02, "eval_completions/mean_length": 104.2025, "eval_completions/mean_terminated_length": 103.42214294433593, "eval_completions/min_length": 73.02, "eval_completions/min_terminated_length": 73.02, "eval_entropy": 0.24162528306245803, "eval_frac_reward_zero_std": 0.57, "eval_kl": 0.2088210688531399, "eval_loss": 0.0010469518601894379, "eval_num_tokens": 83967857.0, "eval_reward": 2.236937608718872, "eval_reward_std": 0.41162257984280587, "eval_rewards/code_complexity_reward/mean": 0.8984999799728394, "eval_rewards/code_complexity_reward/std": 0.0886570330709219, "eval_rewards/code_execution_reward/mean": 0.255, "eval_rewards/code_execution_reward/std": 0.3103350007534027, "eval_rewards/code_syntax_reward/mean": 0.48375, "eval_rewards/code_syntax_reward/std": 0.03863603860139847, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.4996875, "eval_rewards/xmlcount_reward_func/std": 0.000883883461356163, "eval_runtime": 355.7303, "eval_samples_per_second": 0.281, "eval_steps_per_second": 0.037, "step": 500 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 323.0, "completions/max_terminated_length": 323.0, "completions/mean_length": 103.28125, "completions/mean_terminated_length": 103.28125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.25176406209357083, "epoch": 0.5712656784492588, "frac_reward_zero_std": 0.53125, "grad_norm": 0.22157417237758636, "kl": 0.2059282581321895, "learning_rate": 2.325939820543114e-06, "loss": 0.0010295556858181953, "num_tokens": 84090865.0, "reward": 2.2987794876098633, "reward_std": 0.48141056299209595, "rewards/code_complexity_reward/mean": 0.918261706829071, "rewards/code_complexity_reward/std": 0.09354668855667114, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 501, "step_time": 42.716503717936575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 105.306640625, "completions/mean_terminated_length": 105.306640625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2395950008649379, "epoch": 0.572405929304447, "frac_reward_zero_std": 0.546875, "grad_norm": 0.052701689302921295, "kl": 0.19922777288593352, "learning_rate": 2.3160110334522864e-06, "loss": 0.0009959356393665075, "num_tokens": 84212378.0, "reward": 2.307178020477295, "reward_std": 0.48985153436660767, "rewards/code_complexity_reward/mean": 0.9173828363418579, "rewards/code_complexity_reward/std": 0.10440674424171448, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 502, "step_time": 65.1213151793927 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 106.087890625, "completions/mean_terminated_length": 105.29354095458984, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.2431990981567651, "epoch": 0.5735461801596351, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.050450582057237625, "kl": 0.20580612705089152, "learning_rate": 2.3060851633649246e-06, "loss": 0.0010289433412253857, "num_tokens": 84335771.0, "reward": 2.2472167015075684, "reward_std": 0.5070523619651794, "rewards/code_complexity_reward/mean": 0.904296875, "rewards/code_complexity_reward/std": 0.14159800112247467, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 503, "step_time": 50.51374810375273 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 107.486328125, "completions/mean_terminated_length": 106.69471740722656, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23488203063607216, "epoch": 0.5746864310148233, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.06297370046377182, "kl": 0.19894957565702498, "learning_rate": 2.296162367648061e-06, "loss": 0.000994524103589356, "num_tokens": 84460952.0, "reward": 2.295947313308716, "reward_std": 0.5389246344566345, "rewards/code_complexity_reward/mean": 0.8999999761581421, "rewards/code_complexity_reward/std": 0.1564340889453888, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 504, "step_time": 48.915934729389846 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 103.83984375, "completions/mean_terminated_length": 103.83984375, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.23540962766855955, "epoch": 0.5758266818700114, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.055267538875341415, "kl": 0.1951015272643417, "learning_rate": 2.2862428036199834e-06, "loss": 0.0009756851941347122, "num_tokens": 84582970.0, "reward": 2.3182618618011475, "reward_std": 0.48936188220977783, "rewards/code_complexity_reward/mean": 0.91455078125, "rewards/code_complexity_reward/std": 0.09741368144750595, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02701912261545658, "step": 505, "step_time": 36.270060228183866 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 288.0, "completions/max_terminated_length": 288.0, "completions/mean_length": 101.01171875, "completions/mean_terminated_length": 101.01171875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24133104272186756, "epoch": 0.5769669327251995, "frac_reward_zero_std": 0.671875, "grad_norm": 0.04981307312846184, "kl": 0.20129562867805362, "learning_rate": 2.2763266285477476e-06, "loss": 0.0010064742527902126, "num_tokens": 84702728.0, "reward": 2.259521484375, "reward_std": 0.510830283164978, "rewards/code_complexity_reward/mean": 0.9102538824081421, "rewards/code_complexity_reward/std": 0.14760757982730865, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 506, "step_time": 41.382069155573845 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 396.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 101.681640625, "completions/mean_terminated_length": 101.681640625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23622044385410845, "epoch": 0.5781071835803877, "frac_reward_zero_std": 0.5, "grad_norm": 0.0585935078561306, "kl": 0.2177732759155333, "learning_rate": 2.2664139996446756e-06, "loss": 0.0010885689407587051, "num_tokens": 84823645.0, "reward": 2.345508098602295, "reward_std": 0.5486501455307007, "rewards/code_complexity_reward/mean": 0.9095703363418579, "rewards/code_complexity_reward/std": 0.14600421488285065, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 507, "step_time": 48.22398897446692 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 425.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 104.75390625, "completions/mean_terminated_length": 104.75390625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23855017009191215, "epoch": 0.5792474344355758, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.061113569885492325, "kl": 0.22568887891247869, "learning_rate": 2.256505074067872e-06, "loss": 0.0011280201142653823, "num_tokens": 84947591.0, "reward": 2.2713866233825684, "reward_std": 0.4838048815727234, "rewards/code_complexity_reward/mean": 0.91064453125, "rewards/code_complexity_reward/std": 0.11712042987346649, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 508, "step_time": 51.923088774085045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 99.125, "completions/mean_terminated_length": 99.125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24239067593589425, "epoch": 0.580387685290764, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.05773961544036865, "kl": 0.19575178623199463, "learning_rate": 2.246600008915724e-06, "loss": 0.0009787690360099077, "num_tokens": 85065319.0, "reward": 2.313281536102295, "reward_std": 0.4817301630973816, "rewards/code_complexity_reward/mean": 0.9203125238418579, "rewards/code_complexity_reward/std": 0.09091565012931824, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 509, "step_time": 54.73654272593558 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 433.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 108.177734375, "completions/mean_terminated_length": 108.177734375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24631270626559854, "epoch": 0.5815279361459521, "frac_reward_zero_std": 0.484375, "grad_norm": 0.053444307297468185, "kl": 0.1981549117481336, "learning_rate": 2.236698961225417e-06, "loss": 0.0009907165076583624, "num_tokens": 85190282.0, "reward": 2.2804689407348633, "reward_std": 0.49522119760513306, "rewards/code_complexity_reward/mean": 0.910937488079071, "rewards/code_complexity_reward/std": 0.11877349019050598, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 510, "step_time": 44.75441955961287 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 265.0, "completions/max_terminated_length": 265.0, "completions/mean_length": 104.4609375, "completions/mean_terminated_length": 104.4609375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.25139246764592826, "epoch": 0.5826681870011402, "frac_reward_zero_std": 0.609375, "grad_norm": 0.061525195837020874, "kl": 0.23957126052118838, "learning_rate": 2.226802087970444e-06, "loss": 0.001198360463604331, "num_tokens": 85313138.0, "reward": 2.2567384243011475, "reward_std": 0.47614166140556335, "rewards/code_complexity_reward/mean": 0.91357421875, "rewards/code_complexity_reward/std": 0.11580703407526016, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 511, "step_time": 34.59504173323512 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 104.318359375, "completions/mean_terminated_length": 104.318359375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24531145649962127, "epoch": 0.5838084378563284, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05317963659763336, "kl": 0.20278848055750132, "learning_rate": 2.2169095460581116e-06, "loss": 0.0010137776844203472, "num_tokens": 85436057.0, "reward": 2.2763671875, "reward_std": 0.503916323184967, "rewards/code_complexity_reward/mean": 0.9078124761581421, "rewards/code_complexity_reward/std": 0.1340601146221161, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 512, "step_time": 53.56352206505835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 353.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 105.771484375, "completions/mean_terminated_length": 105.771484375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24120025476440787, "epoch": 0.5849486887115165, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.06493277102708817, "kl": 0.19460039143450558, "learning_rate": 2.2070214923270604e-06, "loss": 0.0009731255704537034, "num_tokens": 85558276.0, "reward": 2.2975096702575684, "reward_std": 0.5101081132888794, "rewards/code_complexity_reward/mean": 0.9083983898162842, "rewards/code_complexity_reward/std": 0.12150303274393082, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 513, "step_time": 41.10551432147622 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 102.859375, "completions/mean_terminated_length": 102.859375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23854562733322382, "epoch": 0.5860889395667047, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.062432169914245605, "kl": 0.1909816551487893, "learning_rate": 2.197138083544771e-06, "loss": 0.0009547551744617522, "num_tokens": 85677788.0, "reward": 2.3131837844848633, "reward_std": 0.48548102378845215, "rewards/code_complexity_reward/mean": 0.918261706829071, "rewards/code_complexity_reward/std": 0.09576912969350815, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 514, "step_time": 37.576423082500696 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 362.0, "completions/max_terminated_length": 362.0, "completions/mean_length": 106.466796875, "completions/mean_terminated_length": 106.466796875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2415073188021779, "epoch": 0.5872291904218928, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.059999074786901474, "kl": 0.19927982450462878, "learning_rate": 2.1872594764050835e-06, "loss": 0.0009962893091142178, "num_tokens": 85800683.0, "reward": 2.3082032203674316, "reward_std": 0.4889669418334961, "rewards/code_complexity_reward/mean": 0.9157226085662842, "rewards/code_complexity_reward/std": 0.09718073159456253, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 515, "step_time": 39.39215423632413 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 102.1328125, "completions/mean_terminated_length": 101.33072662353516, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.24458526144735515, "epoch": 0.5883694412770809, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.05318162217736244, "kl": 0.19557107891887426, "learning_rate": 2.177385827525712e-06, "loss": 0.000977599760517478, "num_tokens": 85922543.0, "reward": 2.264404296875, "reward_std": 0.505446195602417, "rewards/code_complexity_reward/mean": 0.9131835699081421, "rewards/code_complexity_reward/std": 0.13666069507598877, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 516, "step_time": 57.58776899520308 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 283.0, "completions/max_terminated_length": 283.0, "completions/mean_length": 104.509765625, "completions/mean_terminated_length": 104.509765625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2370923855341971, "epoch": 0.5895096921322691, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.04880011826753616, "kl": 0.196721795713529, "learning_rate": 2.16751729344576e-06, "loss": 0.0009837265824899077, "num_tokens": 86043024.0, "reward": 2.2914552688598633, "reward_std": 0.48667818307876587, "rewards/code_complexity_reward/mean": 0.918749988079071, "rewards/code_complexity_reward/std": 0.10749778151512146, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 517, "step_time": 35.489426884800196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 303.0, "completions/max_terminated_length": 303.0, "completions/mean_length": 101.693359375, "completions/mean_terminated_length": 101.693359375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2421863512136042, "epoch": 0.5906499429874572, "frac_reward_zero_std": 0.5625, "grad_norm": 0.0625297874212265, "kl": 0.1953068875009194, "learning_rate": 2.1576540306232418e-06, "loss": 0.000976512033957988, "num_tokens": 86162223.0, "reward": 2.2793946266174316, "reward_std": 0.47222277522087097, "rewards/code_complexity_reward/mean": 0.921093761920929, "rewards/code_complexity_reward/std": 0.10186315327882767, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 518, "step_time": 34.43287628144026 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 104.5078125, "completions/mean_terminated_length": 104.5078125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2452405106741935, "epoch": 0.5917901938426454, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.056569211184978485, "kl": 0.19231592351570725, "learning_rate": 2.147796195432597e-06, "loss": 0.0009615470189601183, "num_tokens": 86284047.0, "reward": 2.2980470657348633, "reward_std": 0.4722599685192108, "rewards/code_complexity_reward/mean": 0.916308581829071, "rewards/code_complexity_reward/std": 0.08986098319292068, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 519, "step_time": 47.34474788233638 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 251.0, "completions/max_terminated_length": 251.0, "completions/mean_length": 103.46484375, "completions/mean_terminated_length": 103.46484375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24371023871935904, "epoch": 0.5929304446978335, "frac_reward_zero_std": 0.59375, "grad_norm": 0.04715541750192642, "kl": 0.2130933713633567, "learning_rate": 2.1379439441622183e-06, "loss": 0.0010655894875526428, "num_tokens": 86405969.0, "reward": 2.2938475608825684, "reward_std": 0.46698877215385437, "rewards/code_complexity_reward/mean": 0.92529296875, "rewards/code_complexity_reward/std": 0.0799945667386055, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 520, "step_time": 40.81197663396597 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 310.0, "completions/mean_length": 106.15234375, "completions/mean_terminated_length": 105.35812377929688, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24632415431551635, "epoch": 0.5940706955530216, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.0493898019194603, "kl": 0.20035147643648088, "learning_rate": 2.1280974330119647e-06, "loss": 0.0010017866734415293, "num_tokens": 86530267.0, "reward": 2.2589354515075684, "reward_std": 0.5066432952880859, "rewards/code_complexity_reward/mean": 0.90869140625, "rewards/code_complexity_reward/std": 0.14061523973941803, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 521, "step_time": 64.38177642133087 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 275.0, "completions/max_terminated_length": 275.0, "completions/mean_length": 99.7109375, "completions/mean_terminated_length": 99.7109375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2387722998391837, "epoch": 0.5952109464082098, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.06330642104148865, "kl": 0.249042805749923, "learning_rate": 2.1182568180906947e-06, "loss": 0.001244443585164845, "num_tokens": 86648579.0, "reward": 2.270751953125, "reward_std": 0.49296945333480835, "rewards/code_complexity_reward/mean": 0.9141601324081421, "rewards/code_complexity_reward/std": 0.128287211060524, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 522, "step_time": 42.729477532207966 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 329.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 104.552734375, "completions/mean_terminated_length": 104.552734375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23398909019306302, "epoch": 0.5963511972633979, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.04763493686914444, "kl": 0.20420349156484008, "learning_rate": 2.108422255413782e-06, "loss": 0.00102110649459064, "num_tokens": 86770154.0, "reward": 2.3075196743011475, "reward_std": 0.492971271276474, "rewards/code_complexity_reward/mean": 0.91845703125, "rewards/code_complexity_reward/std": 0.1077076643705368, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 523, "step_time": 40.93885363638401 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 108.263671875, "completions/mean_terminated_length": 106.6803970336914, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24013248761184514, "epoch": 0.5974914481185861, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.053540635854005814, "kl": 0.19479122036136687, "learning_rate": 2.0985939009006506e-06, "loss": 0.000973805203102529, "num_tokens": 86893805.0, "reward": 2.249755859375, "reward_std": 0.5010006427764893, "rewards/code_complexity_reward/mean": 0.9024413824081421, "rewards/code_complexity_reward/std": 0.13670319318771362, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 524, "step_time": 59.53307328186929 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 110.015625, "completions/mean_terminated_length": 109.22896575927734, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24130435683764517, "epoch": 0.5986316989737742, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.05872626230120659, "kl": 0.19505202979780734, "learning_rate": 2.0887719103722987e-06, "loss": 0.0009752219775691628, "num_tokens": 87018277.0, "reward": 2.224169969558716, "reward_std": 0.454742968082428, "rewards/code_complexity_reward/mean": 0.9131835699081421, "rewards/code_complexity_reward/std": 0.1089310348033905, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 525, "step_time": 58.780053650960326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 451.0, "completions/max_terminated_length": 451.0, "completions/mean_length": 105.111328125, "completions/mean_terminated_length": 105.111328125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2482869722880423, "epoch": 0.5997719498289624, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.04903281107544899, "kl": 0.23010863456875086, "learning_rate": 2.0789564395488252e-06, "loss": 0.0011502224951982498, "num_tokens": 87142314.0, "reward": 2.189453125, "reward_std": 0.4655091166496277, "rewards/code_complexity_reward/mean": 0.9068359136581421, "rewards/code_complexity_reward/std": 0.14494802057743073, "rewards/code_execution_reward/mean": 0.193359375, "rewards/code_execution_reward/std": 0.39531853795051575, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 526, "step_time": 44.31518071144819 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 107.796875, "completions/mean_terminated_length": 107.796875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2425726568326354, "epoch": 0.6009122006841505, "frac_reward_zero_std": 0.609375, "grad_norm": 0.04589458554983139, "kl": 0.1941969150211662, "learning_rate": 2.0691476440469674e-06, "loss": 0.000971033878158778, "num_tokens": 87265998.0, "reward": 2.2588868141174316, "reward_std": 0.49668583273887634, "rewards/code_complexity_reward/mean": 0.9035155773162842, "rewards/code_complexity_reward/std": 0.1358719766139984, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 527, "step_time": 59.39416675828397 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 103.21484375, "completions/mean_terminated_length": 103.21484375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24766458291560411, "epoch": 0.6020524515393386, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.05989043042063713, "kl": 0.19006302650086582, "learning_rate": 2.059345679377627e-06, "loss": 0.0009500739979557693, "num_tokens": 87389268.0, "reward": 2.312451124191284, "reward_std": 0.5034980773925781, "rewards/code_complexity_reward/mean": 0.9091796875, "rewards/code_complexity_reward/std": 0.11957896500825882, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 528, "step_time": 51.64611462689936 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 99.212890625, "completions/mean_terminated_length": 99.212890625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2404238418675959, "epoch": 0.6031927023945268, "frac_reward_zero_std": 0.625, "grad_norm": 0.048395074903964996, "kl": 0.2022664377000183, "learning_rate": 2.0495507009434127e-06, "loss": 0.0010113599710166454, "num_tokens": 87510077.0, "reward": 2.2987794876098633, "reward_std": 0.49647462368011475, "rewards/code_complexity_reward/mean": 0.915820300579071, "rewards/code_complexity_reward/std": 0.10737806558609009, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 529, "step_time": 37.475478743202984 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 376.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 100.998046875, "completions/mean_terminated_length": 100.998046875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23856634343974292, "epoch": 0.6043329532497149, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.048619914799928665, "kl": 0.19722355855628848, "learning_rate": 2.0397628640361674e-06, "loss": 0.000986055238172412, "num_tokens": 87630620.0, "reward": 2.3004884719848633, "reward_std": 0.4719683825969696, "rewards/code_complexity_reward/mean": 0.9250975847244263, "rewards/code_complexity_reward/std": 0.0910365879535675, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 530, "step_time": 48.659230592660606 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 104.595703125, "completions/mean_terminated_length": 104.595703125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23595616361126304, "epoch": 0.6054732041049031, "frac_reward_zero_std": 0.59375, "grad_norm": 0.0436982624232769, "kl": 0.1994689181447029, "learning_rate": 2.0299823238345125e-06, "loss": 0.00099729944486171, "num_tokens": 87754897.0, "reward": 2.3039064407348633, "reward_std": 0.5095030069351196, "rewards/code_complexity_reward/mean": 0.911914050579071, "rewards/code_complexity_reward/std": 0.13135842978954315, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 531, "step_time": 46.17271607648581 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 311.0, "completions/max_terminated_length": 311.0, "completions/mean_length": 101.10546875, "completions/mean_terminated_length": 101.10546875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24402942066080868, "epoch": 0.6066134549600912, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05294995754957199, "kl": 0.19923370936885476, "learning_rate": 2.0202092354013885e-06, "loss": 0.000996128306724131, "num_tokens": 87876615.0, "reward": 2.2853517532348633, "reward_std": 0.4758557081222534, "rewards/code_complexity_reward/mean": 0.919726550579071, "rewards/code_complexity_reward/std": 0.09280117601156235, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 532, "step_time": 38.64476901385933 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 100.544921875, "completions/mean_terminated_length": 100.544921875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2391652895603329, "epoch": 0.6077537058152793, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.05232726037502289, "kl": 0.20078171766363084, "learning_rate": 2.0104437536815884e-06, "loss": 0.0010039358166977763, "num_tokens": 87996350.0, "reward": 2.3006348609924316, "reward_std": 0.506287157535553, "rewards/code_complexity_reward/mean": 0.914257824420929, "rewards/code_complexity_reward/std": 0.120549276471138, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 533, "step_time": 43.39897820819169 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 476.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 100.349609375, "completions/mean_terminated_length": 100.349609375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23649663222022355, "epoch": 0.6088939566704675, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.045301686972379684, "kl": 0.20625805133022368, "learning_rate": 2.0006860334993105e-06, "loss": 0.0010315603576600552, "num_tokens": 88117205.0, "reward": 2.308837890625, "reward_std": 0.4879179000854492, "rewards/code_complexity_reward/mean": 0.9151366949081421, "rewards/code_complexity_reward/std": 0.10575596988201141, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 534, "step_time": 74.52359048370272 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 324.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 100.857421875, "completions/mean_terminated_length": 100.857421875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.250890449853614, "epoch": 0.6100342075256556, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05123386159539223, "kl": 0.20193109405227005, "learning_rate": 1.990936229555697e-06, "loss": 0.0010096703190356493, "num_tokens": 88236556.0, "reward": 2.3013672828674316, "reward_std": 0.4637632966041565, "rewards/code_complexity_reward/mean": 0.926953136920929, "rewards/code_complexity_reward/std": 0.07892435789108276, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 535, "step_time": 35.61362033337355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 434.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 105.986328125, "completions/mean_terminated_length": 105.986328125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24200678430497646, "epoch": 0.6111744583808438, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.054810088127851486, "kl": 0.20093966950662434, "learning_rate": 1.981194496426389e-06, "loss": 0.001004635589197278, "num_tokens": 88358961.0, "reward": 2.278857707977295, "reward_std": 0.49728575348854065, "rewards/code_complexity_reward/mean": 0.9134765863418579, "rewards/code_complexity_reward/std": 0.1252942532300949, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 536, "step_time": 44.15322001092136 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 97.6953125, "completions/mean_terminated_length": 96.88453674316406, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.2301249229349196, "epoch": 0.6123147092360319, "frac_reward_zero_std": 0.7109375, "grad_norm": 0.0501900389790535, "kl": 0.20987920719198883, "learning_rate": 1.971460988559065e-06, "loss": 0.0010493106674402952, "num_tokens": 88476881.0, "reward": 2.3617677688598633, "reward_std": 0.514181911945343, "rewards/code_complexity_reward/mean": 0.9221678972244263, "rewards/code_complexity_reward/std": 0.10553092509508133, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 537, "step_time": 55.881766555830836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 105.005859375, "completions/mean_terminated_length": 103.4098129272461, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24106550053693354, "epoch": 0.61345496009122, "frac_reward_zero_std": 0.578125, "grad_norm": 0.06450594961643219, "kl": 0.20770512090530246, "learning_rate": 1.9617358602710034e-06, "loss": 0.0010383290937170386, "num_tokens": 88601184.0, "reward": 2.283935546875, "reward_std": 0.478995680809021, "rewards/code_complexity_reward/mean": 0.9151366949081421, "rewards/code_complexity_reward/std": 0.11152027547359467, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 538, "step_time": 60.980393691919744 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 324.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 102.984375, "completions/mean_terminated_length": 102.984375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24129743268713355, "epoch": 0.6145952109464082, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.04722561314702034, "kl": 0.23924460751004517, "learning_rate": 1.9520192657466286e-06, "loss": 0.0011961410054937005, "num_tokens": 88722824.0, "reward": 2.2979493141174316, "reward_std": 0.512391209602356, "rewards/code_complexity_reward/mean": 0.9118163585662842, "rewards/code_complexity_reward/std": 0.13041414320468903, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 539, "step_time": 36.61361653637141 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 102.5, "completions/mean_terminated_length": 102.5, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23208261164836586, "epoch": 0.6157354618015963, "frac_reward_zero_std": 0.59375, "grad_norm": 0.04756706580519676, "kl": 0.21199946175329387, "learning_rate": 1.9423113590350666e-06, "loss": 0.0010597615037113428, "num_tokens": 88845928.0, "reward": 2.306689739227295, "reward_std": 0.4832269251346588, "rewards/code_complexity_reward/mean": 0.9247070550918579, "rewards/code_complexity_reward/std": 0.09605685621500015, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 540, "step_time": 44.437748128548265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 499.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 106.5390625, "completions/mean_terminated_length": 106.5390625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24358749692328274, "epoch": 0.6168757126567845, "frac_reward_zero_std": 0.59375, "grad_norm": 0.046723730862140656, "kl": 0.19749662862159312, "learning_rate": 1.9326122940477102e-06, "loss": 0.0009874487295746803, "num_tokens": 88968868.0, "reward": 2.2806642055511475, "reward_std": 0.48407837748527527, "rewards/code_complexity_reward/mean": 0.9189453125, "rewards/code_complexity_reward/std": 0.1064571812748909, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 541, "step_time": 57.89719758927822 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 366.0, "completions/max_terminated_length": 366.0, "completions/mean_length": 102.3515625, "completions/mean_terminated_length": 102.3515625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24522325722500682, "epoch": 0.6180159635119726, "frac_reward_zero_std": 0.59375, "grad_norm": 0.04719548299908638, "kl": 0.20165393548086286, "learning_rate": 1.9229222245557675e-06, "loss": 0.0010082360822707415, "num_tokens": 89089160.0, "reward": 2.2904298305511475, "reward_std": 0.49784883856773376, "rewards/code_complexity_reward/mean": 0.916015625, "rewards/code_complexity_reward/std": 0.11918404698371887, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 542, "step_time": 55.56555744912475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 279.0, "completions/max_terminated_length": 279.0, "completions/mean_length": 101.958984375, "completions/mean_terminated_length": 101.958984375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24959208024665713, "epoch": 0.6191562143671607, "frac_reward_zero_std": 0.640625, "grad_norm": 0.04551911726593971, "kl": 0.1946690445765853, "learning_rate": 1.9132413041878356e-06, "loss": 0.0009733869228512049, "num_tokens": 89209739.0, "reward": 2.3009767532348633, "reward_std": 0.4841510057449341, "rewards/code_complexity_reward/mean": 0.920703113079071, "rewards/code_complexity_reward/std": 0.09882426261901855, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 543, "step_time": 43.28312694374472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 99.12890625, "completions/mean_terminated_length": 99.12890625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24284568359144032, "epoch": 0.6202964652223489, "frac_reward_zero_std": 0.609375, "grad_norm": 0.04780049994587898, "kl": 0.19460820429958403, "learning_rate": 1.903569686427454e-06, "loss": 0.0009730259189382195, "num_tokens": 89327965.0, "reward": 2.3427734375, "reward_std": 0.49621516466140747, "rewards/code_complexity_reward/mean": 0.9224609136581421, "rewards/code_complexity_reward/std": 0.09196377545595169, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 544, "step_time": 55.771266049705446 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 457.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 105.95703125, "completions/mean_terminated_length": 105.95703125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24850042234174907, "epoch": 0.621436716077537, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05421530082821846, "kl": 0.2055092144291848, "learning_rate": 1.8939075246106809e-06, "loss": 0.0010275988606736064, "num_tokens": 89451447.0, "reward": 2.2303223609924316, "reward_std": 0.4925668239593506, "rewards/code_complexity_reward/mean": 0.904003918170929, "rewards/code_complexity_reward/std": 0.1386917680501938, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 545, "step_time": 43.905102198943496 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 338.0, "completions/max_terminated_length": 338.0, "completions/mean_length": 102.767578125, "completions/mean_terminated_length": 102.767578125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24411220801994205, "epoch": 0.6225769669327252, "frac_reward_zero_std": 0.671875, "grad_norm": 0.05578114464879036, "kl": 0.19643748027738184, "learning_rate": 1.8842549719236544e-06, "loss": 0.0009823227301239967, "num_tokens": 89573596.0, "reward": 2.314453125, "reward_std": 0.5012958645820618, "rewards/code_complexity_reward/mean": 0.9175781011581421, "rewards/code_complexity_reward/std": 0.11175347119569778, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 546, "step_time": 44.42942895088345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 294.0, "completions/max_terminated_length": 294.0, "completions/mean_length": 100.697265625, "completions/mean_terminated_length": 100.697265625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24508722103200853, "epoch": 0.6237172177879133, "frac_reward_zero_std": 0.578125, "grad_norm": 0.0516214556992054, "kl": 0.20697856973856688, "learning_rate": 1.874612181400169e-06, "loss": 0.0010349326767027378, "num_tokens": 89697949.0, "reward": 2.2164063453674316, "reward_std": 0.476794570684433, "rewards/code_complexity_reward/mean": 0.910351574420929, "rewards/code_complexity_reward/std": 0.1387682557106018, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 547, "step_time": 45.32562912721187 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 105.216796875, "completions/mean_terminated_length": 105.216796875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.2383988662622869, "epoch": 0.6248574686431014, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.05315442755818367, "kl": 0.211147598689422, "learning_rate": 1.8649793059192484e-06, "loss": 0.0010557965142652392, "num_tokens": 89819988.0, "reward": 2.2300782203674316, "reward_std": 0.4731784462928772, "rewards/code_complexity_reward/mean": 0.912304699420929, "rewards/code_complexity_reward/std": 0.13481508195400238, "rewards/code_execution_reward/mean": 0.2265625, "rewards/code_execution_reward/std": 0.4190165400505066, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 548, "step_time": 38.3589221900329 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 352.0, "completions/max_terminated_length": 352.0, "completions/mean_length": 99.021484375, "completions/mean_terminated_length": 99.021484375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24048184440471232, "epoch": 0.6259977194982896, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.052911728620529175, "kl": 0.27143465355038643, "learning_rate": 1.8553564982027183e-06, "loss": 0.001356575870886445, "num_tokens": 89936943.0, "reward": 2.385693311691284, "reward_std": 0.5093035101890564, "rewards/code_complexity_reward/mean": 0.9231445789337158, "rewards/code_complexity_reward/std": 0.09330040961503983, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 549, "step_time": 47.36999005731195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 99.154296875, "completions/mean_terminated_length": 99.154296875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23471854464150965, "epoch": 0.6271379703534777, "frac_reward_zero_std": 0.5625, "grad_norm": 0.04738285765051842, "kl": 0.20711929630488157, "learning_rate": 1.8457439108127914e-06, "loss": 0.0010355403646826744, "num_tokens": 90057414.0, "reward": 2.2953126430511475, "reward_std": 0.4823169410228729, "rewards/code_complexity_reward/mean": 0.9208984375, "rewards/code_complexity_reward/std": 0.10137399286031723, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 550, "step_time": 51.94247026089579 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 269.0, "completions/max_terminated_length": 269.0, "completions/mean_length": 102.38671875, "completions/mean_terminated_length": 102.38671875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23348952317610383, "epoch": 0.6282782212086659, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.043159369379282, "kl": 0.19520728522911668, "learning_rate": 1.8361416961496412e-06, "loss": 0.0009761832770891488, "num_tokens": 90179256.0, "reward": 2.3424317836761475, "reward_std": 0.4727688431739807, "rewards/code_complexity_reward/mean": 0.9306640625, "rewards/code_complexity_reward/std": 0.06893904507160187, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 551, "step_time": 35.94101059809327 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 272.0, "completions/max_terminated_length": 272.0, "completions/mean_length": 102.603515625, "completions/mean_terminated_length": 102.603515625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2505355451721698, "epoch": 0.629418472063854, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.048158735036849976, "kl": 0.21667583216913044, "learning_rate": 1.8265500064489915e-06, "loss": 0.0010835581924766302, "num_tokens": 90300633.0, "reward": 2.257129192352295, "reward_std": 0.4654637575149536, "rewards/code_complexity_reward/mean": 0.9208008050918579, "rewards/code_complexity_reward/std": 0.10599423199892044, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 552, "step_time": 43.4596664858982 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 372.0, "completions/max_terminated_length": 372.0, "completions/mean_length": 103.109375, "completions/mean_terminated_length": 103.109375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24040421284735203, "epoch": 0.6305587229190421, "frac_reward_zero_std": 0.59375, "grad_norm": 0.050306227058172226, "kl": 0.2051699433941394, "learning_rate": 1.816968993779701e-06, "loss": 0.0010258511174470186, "num_tokens": 90420233.0, "reward": 2.324512004852295, "reward_std": 0.5121037364006042, "rewards/code_complexity_reward/mean": 0.9159179925918579, "rewards/code_complexity_reward/std": 0.12353064119815826, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 553, "step_time": 47.023496580310166 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 101.974609375, "completions/mean_terminated_length": 101.974609375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24434501538053155, "epoch": 0.6316989737742303, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.05624137446284294, "kl": 0.21912077208980918, "learning_rate": 1.8073988100413515e-06, "loss": 0.0010953400051221251, "num_tokens": 90540568.0, "reward": 2.3446290493011475, "reward_std": 0.4985060691833496, "rewards/code_complexity_reward/mean": 0.92041015625, "rewards/code_complexity_reward/std": 0.09675976634025574, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 554, "step_time": 45.78417070303112 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 294.0, "completions/max_terminated_length": 294.0, "completions/mean_length": 100.9453125, "completions/mean_terminated_length": 100.9453125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.24444645177572966, "epoch": 0.6328392246294184, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05006890743970871, "kl": 0.19608375942334533, "learning_rate": 1.7978396069618426e-06, "loss": 0.0009803883731365204, "num_tokens": 90660184.0, "reward": 2.2750000953674316, "reward_std": 0.504984974861145, "rewards/code_complexity_reward/mean": 0.910351574420929, "rewards/code_complexity_reward/std": 0.13066993653774261, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 555, "step_time": 60.682782906107605 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 331.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 104.84375, "completions/mean_terminated_length": 104.84375, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.23865050333552063, "epoch": 0.6339794754846066, "frac_reward_zero_std": 0.515625, "grad_norm": 0.06646832823753357, "kl": 0.19892814359627664, "learning_rate": 1.7882915360949802e-06, "loss": 0.00099437334574759, "num_tokens": 90782304.0, "reward": 2.2645509243011475, "reward_std": 0.5543619990348816, "rewards/code_complexity_reward/mean": 0.89697265625, "rewards/code_complexity_reward/std": 0.17677147686481476, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 556, "step_time": 44.72090795543045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 102.96875, "completions/mean_terminated_length": 102.96875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24970356980338693, "epoch": 0.6351197263397947, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.051507458090782166, "kl": 0.19881207030266523, "learning_rate": 1.7787547488180816e-06, "loss": 0.0009940669406205416, "num_tokens": 90903136.0, "reward": 2.25927734375, "reward_std": 0.49226588010787964, "rewards/code_complexity_reward/mean": 0.9151366949081421, "rewards/code_complexity_reward/std": 0.13512472808361053, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 557, "step_time": 37.68146066181362 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 326.0, "completions/mean_length": 100.53125, "completions/mean_terminated_length": 99.72602844238281, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24021309753879905, "epoch": 0.636259977194983, "frac_reward_zero_std": 0.5, "grad_norm": 0.05116770416498184, "kl": 0.20107503654435277, "learning_rate": 1.7692293963295679e-06, "loss": 0.0010056007886305451, "num_tokens": 91020924.0, "reward": 2.297314405441284, "reward_std": 0.4837060272693634, "rewards/code_complexity_reward/mean": 0.919238269329071, "rewards/code_complexity_reward/std": 0.09805348515510559, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 558, "step_time": 57.48064820095897 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 418.0, "completions/max_terminated_length": 418.0, "completions/mean_length": 105.48046875, "completions/mean_terminated_length": 105.48046875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.25528376176953316, "epoch": 0.637400228050171, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05378662049770355, "kl": 0.2107367725111544, "learning_rate": 1.7597156296465734e-06, "loss": 0.0010539947543293238, "num_tokens": 91145482.0, "reward": 2.2035157680511475, "reward_std": 0.483593225479126, "rewards/code_complexity_reward/mean": 0.8994140625, "rewards/code_complexity_reward/std": 0.1477808803319931, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 559, "step_time": 51.214327523484826 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 379.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 104.16796875, "completions/mean_terminated_length": 104.16796875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24339681514538825, "epoch": 0.6385404789053591, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05618056282401085, "kl": 0.2109694709070027, "learning_rate": 1.7502135996025454e-06, "loss": 0.0010550071019679308, "num_tokens": 91267004.0, "reward": 2.305957317352295, "reward_std": 0.5064830780029297, "rewards/code_complexity_reward/mean": 0.9139648079872131, "rewards/code_complexity_reward/std": 0.11839311569929123, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 560, "step_time": 43.32556575257331 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 414.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 102.111328125, "completions/mean_terminated_length": 102.111328125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24059816868975759, "epoch": 0.6396807297605474, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.0514136366546154, "kl": 0.2377165996003896, "learning_rate": 1.7407234568448583e-06, "loss": 0.0011888756416738033, "num_tokens": 91388721.0, "reward": 2.2550783157348633, "reward_std": 0.4583468437194824, "rewards/code_complexity_reward/mean": 0.9187499284744263, "rewards/code_complexity_reward/std": 0.09049763530492783, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 561, "step_time": 51.25272238813341 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 101.30078125, "completions/mean_terminated_length": 101.30078125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23328338284045458, "epoch": 0.6408209806157354, "frac_reward_zero_std": 0.6875, "grad_norm": 0.03604469075798988, "kl": 0.20397131703794003, "learning_rate": 1.7312453518324232e-06, "loss": 0.001019877614453435, "num_tokens": 91509879.0, "reward": 2.32568359375, "reward_std": 0.5081522464752197, "rewards/code_complexity_reward/mean": 0.9200195074081421, "rewards/code_complexity_reward/std": 0.11622149497270584, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 562, "step_time": 40.246188422665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 372.0, "completions/max_terminated_length": 372.0, "completions/mean_length": 100.12890625, "completions/mean_terminated_length": 100.12890625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23743281397037208, "epoch": 0.6419612314709237, "frac_reward_zero_std": 0.6640625, "grad_norm": 0.04692812263965607, "kl": 0.20903773279860616, "learning_rate": 1.7217794348332989e-06, "loss": 0.0010453101713210344, "num_tokens": 91629425.0, "reward": 2.29296875, "reward_std": 0.4733704626560211, "rewards/code_complexity_reward/mean": 0.9234374761581421, "rewards/code_complexity_reward/std": 0.09246303141117096, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 563, "step_time": 39.14715112838894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 103.365234375, "completions/mean_terminated_length": 102.56555938720703, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2500198867637664, "epoch": 0.6431014823261118, "frac_reward_zero_std": 0.671875, "grad_norm": 0.044396527111530304, "kl": 0.20245190965943038, "learning_rate": 1.712325855922316e-06, "loss": 0.001012413646094501, "num_tokens": 91751316.0, "reward": 2.284228801727295, "reward_std": 0.4867938458919525, "rewards/code_complexity_reward/mean": 0.9188476800918579, "rewards/code_complexity_reward/std": 0.1093086302280426, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 564, "step_time": 65.9506185669452 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 413.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 106.62890625, "completions/mean_terminated_length": 106.62890625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24550755391828716, "epoch": 0.6442417331812998, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.048110805451869965, "kl": 0.2097178075928241, "learning_rate": 1.7028847649786907e-06, "loss": 0.0010487909894436598, "num_tokens": 91874946.0, "reward": 2.268847703933716, "reward_std": 0.5010861754417419, "rewards/code_complexity_reward/mean": 0.9081054329872131, "rewards/code_complexity_reward/std": 0.1321866512298584, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 565, "step_time": 42.189630473032594 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 100.25390625, "completions/mean_terminated_length": 100.25390625, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.24230550415813923, "epoch": 0.645381984036488, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.0517401285469532, "kl": 0.2009236824233085, "learning_rate": 1.6934563116836555e-06, "loss": 0.0010047026444226503, "num_tokens": 91995612.0, "reward": 2.2833008766174316, "reward_std": 0.47609126567840576, "rewards/code_complexity_reward/mean": 0.924511730670929, "rewards/code_complexity_reward/std": 0.09752191603183746, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 566, "step_time": 35.85233470611274 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 99.650390625, "completions/mean_terminated_length": 99.650390625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2376681286841631, "epoch": 0.6465222348916762, "frac_reward_zero_std": 0.640625, "grad_norm": 0.04403993487358093, "kl": 0.20281842141412199, "learning_rate": 1.6840406455180801e-06, "loss": 0.0010142857208848, "num_tokens": 92114309.0, "reward": 2.303222894668579, "reward_std": 0.47095105051994324, "rewards/code_complexity_reward/mean": 0.9249023199081421, "rewards/code_complexity_reward/std": 0.08329086005687714, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 567, "step_time": 34.56841514073312 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 333.0, "completions/max_terminated_length": 333.0, "completions/mean_length": 102.3203125, "completions/mean_terminated_length": 102.3203125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2392811558675021, "epoch": 0.6476624857468644, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05538072809576988, "kl": 0.21914294664748013, "learning_rate": 1.6746379157601062e-06, "loss": 0.0010955771431326866, "num_tokens": 92234365.0, "reward": 2.2921388149261475, "reward_std": 0.5292962789535522, "rewards/code_complexity_reward/mean": 0.9033203125, "rewards/code_complexity_reward/std": 0.14958752691745758, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 568, "step_time": 36.613317688927054 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 98.16015625, "completions/mean_terminated_length": 98.16015625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2403176266234368, "epoch": 0.6488027366020525, "frac_reward_zero_std": 0.65625, "grad_norm": 0.05407093092799187, "kl": 0.1976576643064618, "learning_rate": 1.6652482714827783e-06, "loss": 0.0009882851736620069, "num_tokens": 92353927.0, "reward": 2.3067383766174316, "reward_std": 0.4576871395111084, "rewards/code_complexity_reward/mean": 0.9289062023162842, "rewards/code_complexity_reward/std": 0.058640625327825546, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 569, "step_time": 47.2462450126186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 270.0, "completions/max_terminated_length": 270.0, "completions/mean_length": 97.47265625, "completions/mean_terminated_length": 97.47265625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24854010227136314, "epoch": 0.6499429874572406, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.05666280537843704, "kl": 0.24243081314489245, "learning_rate": 1.6558718615516787e-06, "loss": 0.0012112932745367289, "num_tokens": 92472597.0, "reward": 2.354297161102295, "reward_std": 0.4895194172859192, "rewards/code_complexity_reward/mean": 0.9271484613418579, "rewards/code_complexity_reward/std": 0.07904316484928131, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 570, "step_time": 32.61239338107407 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 104.814453125, "completions/mean_terminated_length": 104.814453125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24840159621089697, "epoch": 0.6510832383124288, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.05506712570786476, "kl": 0.20693664206191897, "learning_rate": 1.6465088346225719e-06, "loss": 0.001034505432471633, "num_tokens": 92596074.0, "reward": 2.25439453125, "reward_std": 0.4796292781829834, "rewards/code_complexity_reward/mean": 0.9161132574081421, "rewards/code_complexity_reward/std": 0.12158900499343872, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 571, "step_time": 38.9134063180536 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 104.44140625, "completions/mean_terminated_length": 104.44140625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24935511266812682, "epoch": 0.6522234891676169, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05506172403693199, "kl": 0.20324248424731195, "learning_rate": 1.6371593391390427e-06, "loss": 0.0010160437086597085, "num_tokens": 92719928.0, "reward": 2.2252931594848633, "reward_std": 0.4825425446033478, "rewards/code_complexity_reward/mean": 0.908496081829071, "rewards/code_complexity_reward/std": 0.13588544726371765, "rewards/code_execution_reward/mean": 0.2265625, "rewards/code_execution_reward/std": 0.4190165400505066, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 572, "step_time": 47.55440848786384 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 305.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 99.8828125, "completions/mean_terminated_length": 99.8828125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24210550566203892, "epoch": 0.6533637400228051, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.05480341613292694, "kl": 0.20838238368742168, "learning_rate": 1.6278235233301482e-06, "loss": 0.0010418846504762769, "num_tokens": 92840088.0, "reward": 2.2936525344848633, "reward_std": 0.4697750210762024, "rewards/code_complexity_reward/mean": 0.923144519329071, "rewards/code_complexity_reward/std": 0.08507205545902252, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 573, "step_time": 52.128611135296524 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 99.896484375, "completions/mean_terminated_length": 99.896484375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24534063204191625, "epoch": 0.6545039908779932, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.054458606988191605, "kl": 0.1993200103752315, "learning_rate": 1.6185015352080614e-06, "loss": 0.0009964695200324059, "num_tokens": 92959943.0, "reward": 2.2357423305511475, "reward_std": 0.47733548283576965, "rewards/code_complexity_reward/mean": 0.9072265625, "rewards/code_complexity_reward/std": 0.13240408897399902, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 574, "step_time": 41.517266077920794 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 261.0, "completions/max_terminated_length": 261.0, "completions/mean_length": 101.666015625, "completions/mean_terminated_length": 101.666015625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24404830066487193, "epoch": 0.6556442417331813, "frac_reward_zero_std": 0.6640625, "grad_norm": 0.04702678695321083, "kl": 0.2119805586989969, "learning_rate": 1.6091935225657312e-06, "loss": 0.0010601052781566978, "num_tokens": 93082172.0, "reward": 2.324462890625, "reward_std": 0.48057007789611816, "rewards/code_complexity_reward/mean": 0.9258788824081421, "rewards/code_complexity_reward/std": 0.09033103287220001, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 575, "step_time": 33.328372513875365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 101.400390625, "completions/mean_terminated_length": 101.400390625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24111807951703668, "epoch": 0.6567844925883695, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.0608195960521698, "kl": 0.20879123266786337, "learning_rate": 1.599899632974535e-06, "loss": 0.0010439150501042604, "num_tokens": 93202121.0, "reward": 2.2924318313598633, "reward_std": 0.4987485110759735, "rewards/code_complexity_reward/mean": 0.913378894329071, "rewards/code_complexity_reward/std": 0.1207515150308609, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 576, "step_time": 38.22622794192284 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 350.0, "completions/max_terminated_length": 350.0, "completions/mean_length": 101.724609375, "completions/mean_terminated_length": 101.724609375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23644531867466867, "epoch": 0.6579247434435576, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.0456993468105793, "kl": 0.1983619388192892, "learning_rate": 1.5906200137819378e-06, "loss": 0.0009916280396282673, "num_tokens": 93323268.0, "reward": 2.379394769668579, "reward_std": 0.5397689938545227, "rewards/code_complexity_reward/mean": 0.9151366949081421, "rewards/code_complexity_reward/std": 0.12980633974075317, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 577, "step_time": 39.018898765556514 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 103.626953125, "completions/mean_terminated_length": 103.626953125, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.24202684639021754, "epoch": 0.6590649942987458, "frac_reward_zero_std": 0.609375, "grad_norm": 0.05488641932606697, "kl": 0.20814355206675828, "learning_rate": 1.581354812109162e-06, "loss": 0.001040792092680931, "num_tokens": 93445953.0, "reward": 2.2269532680511475, "reward_std": 0.4461359679698944, "rewards/code_complexity_reward/mean": 0.9208984375, "rewards/code_complexity_reward/std": 0.10474447160959244, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 578, "step_time": 54.81660872139037 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 305.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 100.546875, "completions/mean_terminated_length": 100.546875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24189731711521745, "epoch": 0.6602052451539339, "frac_reward_zero_std": 0.578125, "grad_norm": 0.04921170324087143, "kl": 0.21701286570169032, "learning_rate": 1.572104174848848e-06, "loss": 0.0010854019783437252, "num_tokens": 93567429.0, "reward": 2.2706055641174316, "reward_std": 0.4722538888454437, "rewards/code_complexity_reward/mean": 0.9201171398162842, "rewards/code_complexity_reward/std": 0.09854908287525177, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 579, "step_time": 34.99321747664362 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 338.0, "completions/max_terminated_length": 338.0, "completions/mean_length": 101.37890625, "completions/mean_terminated_length": 101.37890625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23840440809726715, "epoch": 0.661345496009122, "frac_reward_zero_std": 0.59375, "grad_norm": 0.050326909869909286, "kl": 0.20573327015154064, "learning_rate": 1.562868248662732e-06, "loss": 0.0010287256445735693, "num_tokens": 93688187.0, "reward": 2.278125047683716, "reward_std": 0.46942734718322754, "rewards/code_complexity_reward/mean": 0.9242187738418579, "rewards/code_complexity_reward/std": 0.09497849643230438, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 580, "step_time": 45.62173507269472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 99.91015625, "completions/mean_terminated_length": 99.10371398925781, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23158197524026036, "epoch": 0.6624857468643102, "frac_reward_zero_std": 0.625, "grad_norm": 0.0462837815284729, "kl": 0.2040328870061785, "learning_rate": 1.553647179979314e-06, "loss": 0.001020395546220243, "num_tokens": 93807201.0, "reward": 2.2930665016174316, "reward_std": 0.4925355315208435, "rewards/code_complexity_reward/mean": 0.9176757335662842, "rewards/code_complexity_reward/std": 0.11362669616937637, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 581, "step_time": 48.70501857902855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 102.8359375, "completions/mean_terminated_length": 102.8359375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23714206158183515, "epoch": 0.6636259977194983, "frac_reward_zero_std": 0.578125, "grad_norm": 0.050388503819704056, "kl": 0.2096622285898775, "learning_rate": 1.5444411149915427e-06, "loss": 0.0010482609504833817, "num_tokens": 93929029.0, "reward": 2.320605754852295, "reward_std": 0.4987952709197998, "rewards/code_complexity_reward/mean": 0.9168945550918579, "rewards/code_complexity_reward/std": 0.10622836649417877, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 582, "step_time": 46.31583148986101 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 102.64453125, "completions/mean_terminated_length": 102.64453125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24231922393664718, "epoch": 0.6647662485746865, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.053388264030218124, "kl": 0.21128435782156885, "learning_rate": 1.5352501996544935e-06, "loss": 0.0010564030380919576, "num_tokens": 94050451.0, "reward": 2.253467082977295, "reward_std": 0.4659532904624939, "rewards/code_complexity_reward/mean": 0.9193359017372131, "rewards/code_complexity_reward/std": 0.1122930720448494, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 583, "step_time": 38.80612906254828 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 102.70703125, "completions/mean_terminated_length": 102.70703125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.2393487044610083, "epoch": 0.6659064994298746, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.05171298235654831, "kl": 0.20611621881835163, "learning_rate": 1.5260745796830545e-06, "loss": 0.0010305530158802867, "num_tokens": 94172249.0, "reward": 2.3453612327575684, "reward_std": 0.49732083082199097, "rewards/code_complexity_reward/mean": 0.92333984375, "rewards/code_complexity_reward/std": 0.09128982573747635, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 584, "step_time": 47.70530119538307 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 101.806640625, "completions/mean_terminated_length": 101.806640625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24703040649183095, "epoch": 0.6670467502850627, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.0450909286737442, "kl": 0.2018586255144328, "learning_rate": 1.51691440054962e-06, "loss": 0.00100924470461905, "num_tokens": 94292882.0, "reward": 2.228320360183716, "reward_std": 0.46504920721054077, "rewards/code_complexity_reward/mean": 0.9134765267372131, "rewards/code_complexity_reward/std": 0.1236434131860733, "rewards/code_execution_reward/mean": 0.22265625, "rewards/code_execution_reward/std": 0.41643625497817993, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 585, "step_time": 49.05031272023916 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 423.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 100.8515625, "completions/mean_terminated_length": 100.8515625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23878978635184467, "epoch": 0.6681870011402509, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.052338857203722, "kl": 0.21141735091805458, "learning_rate": 1.5077698074817793e-06, "loss": 0.0010573549661785364, "num_tokens": 94414294.0, "reward": 2.364990472793579, "reward_std": 0.5244428515434265, "rewards/code_complexity_reward/mean": 0.9224609136581421, "rewards/code_complexity_reward/std": 0.11655272543430328, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 586, "step_time": 52.32290252111852 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 102.873046875, "completions/mean_terminated_length": 102.07240295410156, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24443615088239312, "epoch": 0.669327251995439, "frac_reward_zero_std": 0.59375, "grad_norm": 0.048426304012537, "kl": 0.21376398764550686, "learning_rate": 1.49864094546002e-06, "loss": 0.0010690137278288603, "num_tokens": 94534825.0, "reward": 2.30908203125, "reward_std": 0.4778386652469635, "rewards/code_complexity_reward/mean": 0.920703113079071, "rewards/code_complexity_reward/std": 0.09358637034893036, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 587, "step_time": 48.74785809777677 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 324.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 100.9140625, "completions/mean_terminated_length": 100.9140625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.237269596895203, "epoch": 0.6704675028506272, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.0650099441409111, "kl": 0.2490114215761423, "learning_rate": 1.489527959215421e-06, "loss": 0.0012447629123926163, "num_tokens": 94654173.0, "reward": 2.3746094703674316, "reward_std": 0.5024011135101318, "rewards/code_complexity_reward/mean": 0.9210937023162842, "rewards/code_complexity_reward/std": 0.0917559415102005, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 588, "step_time": 36.50703972671181 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 245.0, "completions/max_terminated_length": 245.0, "completions/mean_length": 96.9921875, "completions/mean_terminated_length": 96.9921875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23521485831588507, "epoch": 0.6716077537058153, "frac_reward_zero_std": 0.59375, "grad_norm": 0.04687383025884628, "kl": 0.20754020288586617, "learning_rate": 1.4804309932273669e-06, "loss": 0.0010377021972090006, "num_tokens": 94772037.0, "reward": 2.34521484375, "reward_std": 0.48720571398735046, "rewards/code_complexity_reward/mean": 0.9278320074081421, "rewards/code_complexity_reward/std": 0.08235635608434677, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 589, "step_time": 31.82078389171511 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 317.0, "completions/max_terminated_length": 317.0, "completions/mean_length": 102.2109375, "completions/mean_terminated_length": 102.2109375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23878747317939997, "epoch": 0.6727480045610034, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.04636867716908455, "kl": 0.1984395303297788, "learning_rate": 1.471350191721254e-06, "loss": 0.000992167741060257, "num_tokens": 94893505.0, "reward": 2.280078411102295, "reward_std": 0.4551333487033844, "rewards/code_complexity_reward/mean": 0.9271484613418579, "rewards/code_complexity_reward/std": 0.08142128586769104, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 590, "step_time": 42.14853657782078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 104.88671875, "completions/mean_terminated_length": 104.09001922607422, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24960414553061128, "epoch": 0.6738882554161916, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.05515638366341591, "kl": 0.23405302804894745, "learning_rate": 1.462285698666199e-06, "loss": 0.0011702073970809579, "num_tokens": 95015699.0, "reward": 2.273486375808716, "reward_std": 0.5083995461463928, "rewards/code_complexity_reward/mean": 0.9100586175918579, "rewards/code_complexity_reward/std": 0.13603051006793976, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 591, "step_time": 55.291026601567864 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 289.0, "completions/max_terminated_length": 289.0, "completions/mean_length": 102.5546875, "completions/mean_terminated_length": 102.5546875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24928809236735106, "epoch": 0.6750285062713797, "frac_reward_zero_std": 0.6875, "grad_norm": 0.04262048006057739, "kl": 0.2162347340490669, "learning_rate": 1.4532376577727662e-06, "loss": 0.0010815239511430264, "num_tokens": 95136991.0, "reward": 2.26318359375, "reward_std": 0.4485539197921753, "rewards/code_complexity_reward/mean": 0.9219726324081421, "rewards/code_complexity_reward/std": 0.08012108504772186, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 592, "step_time": 34.110939600504935 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 103.166015625, "completions/mean_terminated_length": 103.166015625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24079820979386568, "epoch": 0.6761687571265679, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.047517839819192886, "kl": 0.20676472154445946, "learning_rate": 1.4442062124906764e-06, "loss": 0.0010337713174521923, "num_tokens": 95259952.0, "reward": 2.308887004852295, "reward_std": 0.4824352562427521, "rewards/code_complexity_reward/mean": 0.9237304329872131, "rewards/code_complexity_reward/std": 0.0916704535484314, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 593, "step_time": 40.530859797261655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 278.0, "completions/max_terminated_length": 278.0, "completions/mean_length": 98.376953125, "completions/mean_terminated_length": 98.376953125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24067919980734587, "epoch": 0.677309007981756, "frac_reward_zero_std": 0.6875, "grad_norm": 0.04412622004747391, "kl": 0.20286630117334425, "learning_rate": 1.4351915060065488e-06, "loss": 0.0010143695399165154, "num_tokens": 95377753.0, "reward": 2.3133790493011475, "reward_std": 0.46978330612182617, "rewards/code_complexity_reward/mean": 0.92431640625, "rewards/code_complexity_reward/std": 0.06788329780101776, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 594, "step_time": 33.696400593966246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 437.0, "completions/mean_length": 106.216796875, "completions/mean_terminated_length": 105.42269897460938, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24543360387906432, "epoch": 0.6784492588369442, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.0461430698633194, "kl": 0.21403010794892907, "learning_rate": 1.4261936812416124e-06, "loss": 0.0010703590232878923, "num_tokens": 95499760.0, "reward": 2.2757325172424316, "reward_std": 0.5138190388679504, "rewards/code_complexity_reward/mean": 0.9113280773162842, "rewards/code_complexity_reward/std": 0.1409660279750824, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 595, "step_time": 57.09528720472008 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 106.240234375, "completions/mean_terminated_length": 104.6490249633789, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24606430414132774, "epoch": 0.6795895096921323, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.05144377797842026, "kl": 0.20824243081733584, "learning_rate": 1.4172128808494572e-06, "loss": 0.0010410714894533157, "num_tokens": 95622439.0, "reward": 2.2729005813598633, "reward_std": 0.5209850072860718, "rewards/code_complexity_reward/mean": 0.9052734375, "rewards/code_complexity_reward/std": 0.15037956833839417, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 596, "step_time": 75.62934380583465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 260.0, "completions/max_terminated_length": 260.0, "completions/mean_length": 98.16015625, "completions/mean_terminated_length": 98.16015625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24516392778605223, "epoch": 0.6807297605473204, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05066615715622902, "kl": 0.206049676053226, "learning_rate": 1.408249247213762e-06, "loss": 0.0010302297305315733, "num_tokens": 95740597.0, "reward": 2.321777582168579, "reward_std": 0.5246370434761047, "rewards/code_complexity_reward/mean": 0.9170898199081421, "rewards/code_complexity_reward/std": 0.13280773162841797, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 597, "step_time": 41.213763911277056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 458.0, "completions/max_terminated_length": 458.0, "completions/mean_length": 102.87890625, "completions/mean_terminated_length": 102.87890625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2424487101379782, "epoch": 0.6818700114025086, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05313912779092789, "kl": 0.20232841605320573, "learning_rate": 1.3993029224460364e-06, "loss": 0.0010114414617419243, "num_tokens": 95862843.0, "reward": 2.27294921875, "reward_std": 0.495809942483902, "rewards/code_complexity_reward/mean": 0.9151367545127869, "rewards/code_complexity_reward/std": 0.123905710875988, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 598, "step_time": 47.35081245098263 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 446.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 106.119140625, "completions/mean_terminated_length": 106.119140625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2353483068291098, "epoch": 0.6830102622576967, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.05669548735022545, "kl": 0.19921730272471905, "learning_rate": 1.390374048383379e-06, "loss": 0.000995927257463336, "num_tokens": 95985648.0, "reward": 2.337402582168579, "reward_std": 0.4903596043586731, "rewards/code_complexity_reward/mean": 0.9200195074081421, "rewards/code_complexity_reward/std": 0.08959901332855225, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 599, "step_time": 43.68393323197961 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 104.55859375, "completions/mean_terminated_length": 104.55859375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24701581988483667, "epoch": 0.6841505131128849, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.05010548233985901, "kl": 0.20554535882547498, "learning_rate": 1.3814627665862112e-06, "loss": 0.0010277156252413988, "num_tokens": 96107906.0, "reward": 2.287841796875, "reward_std": 0.5143345594406128, "rewards/code_complexity_reward/mean": 0.9107421636581421, "rewards/code_complexity_reward/std": 0.1332707405090332, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 600, "step_time": 42.95379906427115 }, { "epoch": 0.6841505131128849, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 157.26, "eval_completions/max_terminated_length": 157.26, "eval_completions/mean_length": 105.535, "eval_completions/mean_terminated_length": 105.535, "eval_completions/min_length": 72.96, "eval_completions/min_terminated_length": 72.96, "eval_entropy": 0.23598769932985306, "eval_frac_reward_zero_std": 0.55, "eval_kl": 0.1953857347369194, "eval_loss": 0.0009752993355505168, "eval_num_tokens": 96107906.0, "eval_reward": 2.2758126258850098, "eval_reward_std": 0.3584836133569479, "eval_rewards/code_complexity_reward/mean": 0.9154999804496765, "eval_rewards/code_complexity_reward/std": 0.059037979394197464, "eval_rewards/code_execution_reward/mean": 0.27, "eval_rewards/code_execution_reward/std": 0.29325084686279296, "eval_rewards/code_syntax_reward/mean": 0.49125, "eval_rewards/code_syntax_reward/std": 0.022306769788265228, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.4990625, "eval_rewards/xmlcount_reward_func/std": 0.002651650309562683, "eval_runtime": 353.3015, "eval_samples_per_second": 0.283, "eval_steps_per_second": 0.037, "step": 600 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 102.533203125, "completions/mean_terminated_length": 102.533203125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24771644454449415, "epoch": 0.685290763968073, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.047582611441612244, "kl": 0.20537267765030265, "learning_rate": 1.3725692183360528e-06, "loss": 0.0010267526376992464, "num_tokens": 96228375.0, "reward": 2.2674806118011475, "reward_std": 0.4959642291069031, "rewards/code_complexity_reward/mean": 0.91259765625, "rewards/code_complexity_reward/std": 0.12741787731647491, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 601, "step_time": 40.27438442502171 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 445.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 109.388671875, "completions/mean_terminated_length": 109.388671875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2439700576942414, "epoch": 0.6864310148232611, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.04521101340651512, "kl": 0.19612436811439693, "learning_rate": 1.3636935446332628e-06, "loss": 0.000980766722932458, "num_tokens": 96359102.0, "reward": 2.280956983566284, "reward_std": 0.49724650382995605, "rewards/code_complexity_reward/mean": 0.910449206829071, "rewards/code_complexity_reward/std": 0.1207980364561081, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 602, "step_time": 46.18675137963146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 329.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 103.111328125, "completions/mean_terminated_length": 103.111328125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24747656285762787, "epoch": 0.6875712656784493, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.04866265505552292, "kl": 0.2192029650323093, "learning_rate": 1.3548358861948196e-06, "loss": 0.0010960651561617851, "num_tokens": 96481183.0, "reward": 2.2506346702575684, "reward_std": 0.4563274085521698, "rewards/code_complexity_reward/mean": 0.9196288585662842, "rewards/code_complexity_reward/std": 0.09985536336898804, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 603, "step_time": 46.07686768844724 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 371.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 101.41015625, "completions/mean_terminated_length": 101.41015625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23228014865890145, "epoch": 0.6887115165336374, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.05201301351189613, "kl": 0.1950598380062729, "learning_rate": 1.3459963834520806e-06, "loss": 0.0009750532335601747, "num_tokens": 96601537.0, "reward": 2.2972657680511475, "reward_std": 0.4932442903518677, "rewards/code_complexity_reward/mean": 0.9150390625, "rewards/code_complexity_reward/std": 0.11364103108644485, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 604, "step_time": 38.802519978024065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 105.115234375, "completions/mean_terminated_length": 105.115234375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23647242062725127, "epoch": 0.6898517673888256, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05475208908319473, "kl": 0.1902446097228676, "learning_rate": 1.3371751765485568e-06, "loss": 0.0009512893157079816, "num_tokens": 96723728.0, "reward": 2.3214356899261475, "reward_std": 0.4967268109321594, "rewards/code_complexity_reward/mean": 0.91796875, "rewards/code_complexity_reward/std": 0.10341230034828186, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 605, "step_time": 37.8190534170717 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 102.79296875, "completions/mean_terminated_length": 101.99217224121094, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24661161773838103, "epoch": 0.6909920182440137, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.05112626776099205, "kl": 0.20521409600041807, "learning_rate": 1.3283724053376985e-06, "loss": 0.0010259677655994892, "num_tokens": 96844762.0, "reward": 2.3270020484924316, "reward_std": 0.5033300518989563, "rewards/code_complexity_reward/mean": 0.920605480670929, "rewards/code_complexity_reward/std": 0.10731659829616547, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 606, "step_time": 54.26020245999098 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 104.88671875, "completions/mean_terminated_length": 104.88671875, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.23576552071608603, "epoch": 0.6921322690992018, "frac_reward_zero_std": 0.625, "grad_norm": 0.043700750917196274, "kl": 0.21599361975677311, "learning_rate": 1.319588209380664e-06, "loss": 0.001080215792171657, "num_tokens": 96965408.0, "reward": 2.2542481422424316, "reward_std": 0.47156190872192383, "rewards/code_complexity_reward/mean": 0.9166991710662842, "rewards/code_complexity_reward/std": 0.11010332405567169, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 607, "step_time": 64.75771021936089 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 221.0, "completions/max_terminated_length": 221.0, "completions/mean_length": 96.638671875, "completions/mean_terminated_length": 96.638671875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23823750670999289, "epoch": 0.69327251995439, "frac_reward_zero_std": 0.59375, "grad_norm": 0.050973404198884964, "kl": 0.21426887554116547, "learning_rate": 1.3108227279441243e-06, "loss": 0.0010714787058532238, "num_tokens": 97083699.0, "reward": 2.274658441543579, "reward_std": 0.4584108889102936, "rewards/code_complexity_reward/mean": 0.932910144329071, "rewards/code_complexity_reward/std": 0.08649472892284393, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 608, "step_time": 32.0719695314765 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 464.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 104.021484375, "completions/mean_terminated_length": 104.021484375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24080603127367795, "epoch": 0.6944127708095781, "frac_reward_zero_std": 0.625, "grad_norm": 0.05169383063912392, "kl": 0.23125243932008743, "learning_rate": 1.3020760999980386e-06, "loss": 0.001155742211267352, "num_tokens": 97205226.0, "reward": 2.31396484375, "reward_std": 0.505280613899231, "rewards/code_complexity_reward/mean": 0.9131835699081421, "rewards/code_complexity_reward/std": 0.12389243394136429, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 609, "step_time": 47.76495453249663 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 101.607421875, "completions/mean_terminated_length": 101.607421875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23810616019181907, "epoch": 0.6955530216647663, "frac_reward_zero_std": 0.65625, "grad_norm": 0.046436600387096405, "kl": 0.2010175040923059, "learning_rate": 1.2933484642134631e-06, "loss": 0.0010051175486296415, "num_tokens": 97325393.0, "reward": 2.3026368618011475, "reward_std": 0.5054186582565308, "rewards/code_complexity_reward/mean": 0.92138671875, "rewards/code_complexity_reward/std": 0.12428515404462814, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 610, "step_time": 35.471980806440115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 331.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 102.607421875, "completions/mean_terminated_length": 102.607421875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24739643116481602, "epoch": 0.6966932725199544, "frac_reward_zero_std": 0.625, "grad_norm": 0.04728738218545914, "kl": 0.21044443035498261, "learning_rate": 1.2846399589603453e-06, "loss": 0.0010521336225792766, "num_tokens": 97444784.0, "reward": 2.3241212368011475, "reward_std": 0.4899071455001831, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.09030061960220337, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 611, "step_time": 36.98652384802699 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 286.0, "completions/max_terminated_length": 286.0, "completions/mean_length": 102.81640625, "completions/mean_terminated_length": 102.81640625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2416261974722147, "epoch": 0.6978335233751425, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.045632146298885345, "kl": 0.1989265363663435, "learning_rate": 1.2759507223053341e-06, "loss": 0.000994640402495861, "num_tokens": 97567774.0, "reward": 2.2476563453674316, "reward_std": 0.5054152607917786, "rewards/code_complexity_reward/mean": 0.9054687023162842, "rewards/code_complexity_reward/std": 0.14817669987678528, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 612, "step_time": 43.20637665595859 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 100.384765625, "completions/mean_terminated_length": 100.384765625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24914766428992152, "epoch": 0.6989737742303307, "frac_reward_zero_std": 0.5625, "grad_norm": 0.04860232397913933, "kl": 0.20905036595650017, "learning_rate": 1.2672808920095914e-06, "loss": 0.0010451360139995813, "num_tokens": 97689923.0, "reward": 2.2464356422424316, "reward_std": 0.4539651572704315, "rewards/code_complexity_reward/mean": 0.9193359613418579, "rewards/code_complexity_reward/std": 0.1002303957939148, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 613, "step_time": 38.80267655290663 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 421.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 102.216796875, "completions/mean_terminated_length": 102.216796875, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.24533106270246208, "epoch": 0.7001140250855188, "frac_reward_zero_std": 0.515625, "grad_norm": 0.06846283376216888, "kl": 0.19415723043493927, "learning_rate": 1.2586306055266007e-06, "loss": 0.0009704953408800066, "num_tokens": 97811170.0, "reward": 2.35205078125, "reward_std": 0.5101328492164612, "rewards/code_complexity_reward/mean": 0.9170898199081421, "rewards/code_complexity_reward/std": 0.10315912961959839, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 614, "step_time": 42.28157492540777 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 310.0, "completions/max_terminated_length": 310.0, "completions/mean_length": 102.359375, "completions/mean_terminated_length": 102.359375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2534307560417801, "epoch": 0.701254275940707, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.04813406616449356, "kl": 0.22567394096404314, "learning_rate": 1.2500000000000007e-06, "loss": 0.0011282588820904493, "num_tokens": 97932862.0, "reward": 2.277880907058716, "reward_std": 0.5118700861930847, "rewards/code_complexity_reward/mean": 0.9097656011581421, "rewards/code_complexity_reward/std": 0.14056198298931122, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 615, "step_time": 39.08693247754127 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 106.15234375, "completions/mean_terminated_length": 106.15234375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24337985413149, "epoch": 0.7023945267958951, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05107182264328003, "kl": 0.2006391913164407, "learning_rate": 1.2413892122613968e-06, "loss": 0.0010030355770140886, "num_tokens": 98057944.0, "reward": 2.28466796875, "reward_std": 0.49532485008239746, "rewards/code_complexity_reward/mean": 0.9122070074081421, "rewards/code_complexity_reward/std": 0.1230420470237732, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 616, "step_time": 44.269765574485064 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 105.58984375, "completions/mean_terminated_length": 105.58984375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24822693597525358, "epoch": 0.7035347776510832, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.0664619654417038, "kl": 0.2386764141265303, "learning_rate": 1.2327983788282033e-06, "loss": 0.0011930952314287424, "num_tokens": 98181562.0, "reward": 2.2581543922424316, "reward_std": 0.4773983359336853, "rewards/code_complexity_reward/mean": 0.9171874523162842, "rewards/code_complexity_reward/std": 0.11602305620908737, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 617, "step_time": 54.55048869457096 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 412.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 104.828125, "completions/mean_terminated_length": 104.828125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24390678922645748, "epoch": 0.7046750285062714, "frac_reward_zero_std": 0.609375, "grad_norm": 0.04902678728103638, "kl": 0.20312066050246358, "learning_rate": 1.2242276359014724e-06, "loss": 0.0010155041236430407, "num_tokens": 98301934.0, "reward": 2.2701661586761475, "reward_std": 0.46047791838645935, "rewards/code_complexity_reward/mean": 0.92041015625, "rewards/code_complexity_reward/std": 0.09193732589483261, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 618, "step_time": 40.86728687398136 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 104.73828125, "completions/mean_terminated_length": 104.73828125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2458933494053781, "epoch": 0.7058152793614595, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.04961511120200157, "kl": 0.2003873810172081, "learning_rate": 1.215677119363736e-06, "loss": 0.0010017736349254847, "num_tokens": 98426472.0, "reward": 2.2803711891174316, "reward_std": 0.4626462757587433, "rewards/code_complexity_reward/mean": 0.9206054210662842, "rewards/code_complexity_reward/std": 0.081330806016922, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 619, "step_time": 37.85995785612613 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 285.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 99.486328125, "completions/mean_terminated_length": 99.486328125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2387530819978565, "epoch": 0.7069555302166477, "frac_reward_zero_std": 0.640625, "grad_norm": 0.05084627866744995, "kl": 0.21134246489964426, "learning_rate": 1.2071469647768564e-06, "loss": 0.001056535984389484, "num_tokens": 98546057.0, "reward": 2.2464356422424316, "reward_std": 0.4719201922416687, "rewards/code_complexity_reward/mean": 0.9162108898162842, "rewards/code_complexity_reward/std": 0.12001670151948929, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 620, "step_time": 33.961996871978045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 103.251953125, "completions/mean_terminated_length": 103.251953125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24746996304020286, "epoch": 0.7080957810718358, "frac_reward_zero_std": 0.625, "grad_norm": 0.04396074265241623, "kl": 0.20582228782586753, "learning_rate": 1.198637307379867e-06, "loss": 0.0010290637146681547, "num_tokens": 98667202.0, "reward": 2.234424114227295, "reward_std": 0.44765952229499817, "rewards/code_complexity_reward/mean": 0.9198242425918579, "rewards/code_complexity_reward/std": 0.10641170293092728, "rewards/code_execution_reward/mean": 0.220703125, "rewards/code_execution_reward/std": 0.4151262938976288, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 621, "step_time": 53.943612329661846 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 399.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 104.20703125, "completions/mean_terminated_length": 104.20703125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2322106973733753, "epoch": 0.7092360319270239, "frac_reward_zero_std": 0.640625, "grad_norm": 0.044262468814849854, "kl": 0.21053061308339238, "learning_rate": 1.190148282086837e-06, "loss": 0.0010526429396122694, "num_tokens": 98788284.0, "reward": 2.3312501907348633, "reward_std": 0.4767490327358246, "rewards/code_complexity_reward/mean": 0.926562488079071, "rewards/code_complexity_reward/std": 0.07743123918771744, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 622, "step_time": 42.60805843677372 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 492.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 103.919921875, "completions/mean_terminated_length": 103.919921875, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.23784553282894194, "epoch": 0.7103762827822121, "frac_reward_zero_std": 0.6875, "grad_norm": 0.04052217677235603, "kl": 0.20967737445607781, "learning_rate": 1.1816800234847304e-06, "loss": 0.001048439648002386, "num_tokens": 98910339.0, "reward": 2.2515625953674316, "reward_std": 0.4434095621109009, "rewards/code_complexity_reward/mean": 0.925976574420929, "rewards/code_complexity_reward/std": 0.08363689482212067, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 623, "step_time": 49.47667822800577 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 467.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 106.0625, "completions/mean_terminated_length": 106.0625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2392222359776497, "epoch": 0.7115165336374002, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.0530867837369442, "kl": 0.21304173092357814, "learning_rate": 1.1732326658312693e-06, "loss": 0.0010650595650076866, "num_tokens": 99032375.0, "reward": 2.3241212368011475, "reward_std": 0.5059520602226257, "rewards/code_complexity_reward/mean": 0.91943359375, "rewards/code_complexity_reward/std": 0.11107142269611359, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 624, "step_time": 56.425724162720144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 277.0, "completions/max_terminated_length": 277.0, "completions/mean_length": 98.228515625, "completions/mean_terminated_length": 98.228515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24131595576182008, "epoch": 0.7126567844925884, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.058433547616004944, "kl": 0.20301689812913537, "learning_rate": 1.1648063430528084e-06, "loss": 0.0010152023751288652, "num_tokens": 99150944.0, "reward": 2.328857421875, "reward_std": 0.5217515230178833, "rewards/code_complexity_reward/mean": 0.9175781011581421, "rewards/code_complexity_reward/std": 0.12872090935707092, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 625, "step_time": 34.610777080990374 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 99.158203125, "completions/mean_terminated_length": 99.158203125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23847824148833752, "epoch": 0.7137970353477765, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.04633753374218941, "kl": 0.2142067519016564, "learning_rate": 1.1564011887422098e-06, "loss": 0.001070762169547379, "num_tokens": 99267589.0, "reward": 2.3506836891174316, "reward_std": 0.5049505233764648, "rewards/code_complexity_reward/mean": 0.9264647960662842, "rewards/code_complexity_reward/std": 0.10560515522956848, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 626, "step_time": 73.4401007834822 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 296.0, "completions/max_terminated_length": 296.0, "completions/mean_length": 102.52734375, "completions/mean_terminated_length": 102.52734375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24411053978838027, "epoch": 0.7149372862029647, "frac_reward_zero_std": 0.59375, "grad_norm": 0.08344262838363647, "kl": 0.21271714055910707, "learning_rate": 1.1480173361567287e-06, "loss": 0.0010633094934746623, "num_tokens": 99387955.0, "reward": 2.300585985183716, "reward_std": 0.4994220435619354, "rewards/code_complexity_reward/mean": 0.9203124642372131, "rewards/code_complexity_reward/std": 0.1158120334148407, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 627, "step_time": 35.154137978330255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 306.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 99.14453125, "completions/mean_terminated_length": 99.14453125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24048170796595514, "epoch": 0.7160775370581528, "frac_reward_zero_std": 0.65625, "grad_norm": 0.045693784952163696, "kl": 0.21101492131128907, "learning_rate": 1.1396549182158933e-06, "loss": 0.0010551491286605597, "num_tokens": 99507229.0, "reward": 2.2801761627197266, "reward_std": 0.4698467254638672, "rewards/code_complexity_reward/mean": 0.925976574420929, "rewards/code_complexity_reward/std": 0.1000441387295723, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 628, "step_time": 44.27908998820931 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 264.0, "completions/max_terminated_length": 264.0, "completions/mean_length": 101.3515625, "completions/mean_terminated_length": 101.3515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.237303058616817, "epoch": 0.7172177879133409, "frac_reward_zero_std": 0.609375, "grad_norm": 0.0490560345351696, "kl": 0.19935694127343595, "learning_rate": 1.1313140674994053e-06, "loss": 0.0009967429796233773, "num_tokens": 99627477.0, "reward": 2.23974609375, "reward_std": 0.46345648169517517, "rewards/code_complexity_reward/mean": 0.9180663824081421, "rewards/code_complexity_reward/std": 0.11510554701089859, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 629, "step_time": 41.15408802218735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 316.0, "completions/max_terminated_length": 316.0, "completions/mean_length": 107.767578125, "completions/mean_terminated_length": 107.767578125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2545372375752777, "epoch": 0.7183580387685291, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05060083046555519, "kl": 0.19694163813255727, "learning_rate": 1.1229949162450331e-06, "loss": 0.0009847600013017654, "num_tokens": 99752726.0, "reward": 2.2547852993011475, "reward_std": 0.4834457039833069, "rewards/code_complexity_reward/mean": 0.91259765625, "rewards/code_complexity_reward/std": 0.12422948330640793, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 630, "step_time": 52.87888828292489 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 101.98046875, "completions/mean_terminated_length": 101.98046875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23857086361385882, "epoch": 0.7194982896237172, "frac_reward_zero_std": 0.5625, "grad_norm": 0.06769980490207672, "kl": 0.19395770854316652, "learning_rate": 1.1146975963465179e-06, "loss": 0.0009697970235720277, "num_tokens": 99872920.0, "reward": 2.3275392055511475, "reward_std": 0.4902507960796356, "rewards/code_complexity_reward/mean": 0.9228515625, "rewards/code_complexity_reward/std": 0.08932927995920181, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 631, "step_time": 46.236975154839456 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 102.68359375, "completions/mean_terminated_length": 102.68359375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24769207695499063, "epoch": 0.7206385404789054, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05842607840895653, "kl": 0.20844931644387543, "learning_rate": 1.106422239351481e-06, "loss": 0.0010420955950394273, "num_tokens": 99993578.0, "reward": 2.3033204078674316, "reward_std": 0.46267610788345337, "rewards/code_complexity_reward/mean": 0.927929699420929, "rewards/code_complexity_reward/std": 0.07127034664154053, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 632, "step_time": 40.367829573340714 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 300.0, "completions/max_terminated_length": 300.0, "completions/mean_length": 101.53125, "completions/mean_terminated_length": 101.53125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2401159016881138, "epoch": 0.7217787913340935, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.05068323761224747, "kl": 0.19603063841350377, "learning_rate": 1.0981689764593384e-06, "loss": 0.0009801456471905112, "num_tokens": 100114914.0, "reward": 2.27783203125, "reward_std": 0.4583662450313568, "rewards/code_complexity_reward/mean": 0.9253906011581421, "rewards/code_complexity_reward/std": 0.08114785701036453, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 633, "step_time": 55.055100072175264 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 101.67578125, "completions/mean_terminated_length": 101.67578125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2453055998776108, "epoch": 0.7229190421892816, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.04890409857034683, "kl": 0.2049515990074724, "learning_rate": 1.0899379385192222e-06, "loss": 0.001024888944812119, "num_tokens": 100234220.0, "reward": 2.295898675918579, "reward_std": 0.4995230436325073, "rewards/code_complexity_reward/mean": 0.9175781011581421, "rewards/code_complexity_reward/std": 0.11642755568027496, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 634, "step_time": 39.46819037664682 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 404.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 104.53125, "completions/mean_terminated_length": 104.53125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.25004396378062665, "epoch": 0.7240592930444698, "frac_reward_zero_std": 0.65625, "grad_norm": 0.047072187066078186, "kl": 0.19407648965716362, "learning_rate": 1.0817292560279038e-06, "loss": 0.0009702604147605598, "num_tokens": 100357556.0, "reward": 2.2851076126098633, "reward_std": 0.4758750796318054, "rewards/code_complexity_reward/mean": 0.9248046875, "rewards/code_complexity_reward/std": 0.09837207198143005, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 635, "step_time": 51.8761006295681 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 373.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 107.625, "completions/mean_terminated_length": 107.625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24001182592473924, "epoch": 0.7251995438996579, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.06667012721300125, "kl": 0.19450940913520753, "learning_rate": 1.0735430591277268e-06, "loss": 0.000972374458797276, "num_tokens": 100481676.0, "reward": 2.2926270961761475, "reward_std": 0.4730047285556793, "rewards/code_complexity_reward/mean": 0.918652355670929, "rewards/code_complexity_reward/std": 0.08752265572547913, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 636, "step_time": 38.96399190276861 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 107.13671875, "completions/mean_terminated_length": 105.54902648925781, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24490740220062435, "epoch": 0.7263397947548461, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05130777135491371, "kl": 0.19300097250379622, "learning_rate": 1.0653794776045435e-06, "loss": 0.0009649736457504332, "num_tokens": 100607318.0, "reward": 2.3050293922424316, "reward_std": 0.5224958062171936, "rewards/code_complexity_reward/mean": 0.9098632335662842, "rewards/code_complexity_reward/std": 0.1312866359949112, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 637, "step_time": 50.15120738465339 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 104.791015625, "completions/mean_terminated_length": 103.99412536621094, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24561276542954147, "epoch": 0.7274800456100342, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.048409152776002884, "kl": 0.20986288087442517, "learning_rate": 1.0572386408856553e-06, "loss": 0.001049357932060957, "num_tokens": 100728759.0, "reward": 2.2581543922424316, "reward_std": 0.5000275373458862, "rewards/code_complexity_reward/mean": 0.909375011920929, "rewards/code_complexity_reward/std": 0.14351607859134674, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 638, "step_time": 49.230547707527876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 309.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 104.984375, "completions/mean_terminated_length": 104.984375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2530134664848447, "epoch": 0.7286202964652223, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05438821017742157, "kl": 0.19237612537108362, "learning_rate": 1.0491206780377636e-06, "loss": 0.0009617629693821073, "num_tokens": 100850291.0, "reward": 2.273242235183716, "reward_std": 0.4879535436630249, "rewards/code_complexity_reward/mean": 0.9124999642372131, "rewards/code_complexity_reward/std": 0.11674817651510239, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 639, "step_time": 35.42666126880795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 399.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 102.384765625, "completions/mean_terminated_length": 102.384765625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24241920025087893, "epoch": 0.7297605473204105, "frac_reward_zero_std": 0.609375, "grad_norm": 0.0721849799156189, "kl": 0.20403268886730075, "learning_rate": 1.0410257177649217e-06, "loss": 0.001020221970975399, "num_tokens": 100969912.0, "reward": 2.2728519439697266, "reward_std": 0.4833785891532898, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.11587999761104584, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 640, "step_time": 41.33649511169642 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 311.0, "completions/mean_length": 105.908203125, "completions/mean_terminated_length": 105.1135025024414, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.2489117169752717, "epoch": 0.7309007981755986, "frac_reward_zero_std": 0.53125, "grad_norm": 0.060320839285850525, "kl": 0.20799240050837398, "learning_rate": 1.0329538884064948e-06, "loss": 0.0010400509927421808, "num_tokens": 101091961.0, "reward": 2.248095750808716, "reward_std": 0.4987730085849762, "rewards/code_complexity_reward/mean": 0.9068359136581421, "rewards/code_complexity_reward/std": 0.13820964097976685, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 641, "step_time": 50.26889856066555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 438.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 101.6171875, "completions/mean_terminated_length": 101.6171875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24662940483540297, "epoch": 0.7320410490307868, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05982089415192604, "kl": 0.1958824882749468, "learning_rate": 1.0249053179351257e-06, "loss": 0.0009794796351343393, "num_tokens": 101211169.0, "reward": 2.2770509719848633, "reward_std": 0.48672768473625183, "rewards/code_complexity_reward/mean": 0.916308581829071, "rewards/code_complexity_reward/std": 0.1151135116815567, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 642, "step_time": 43.43799520935863 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 341.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 103.47265625, "completions/mean_terminated_length": 103.47265625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24806336732581258, "epoch": 0.7331812998859749, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.04845692962408066, "kl": 0.19717499043326825, "learning_rate": 1.0168801339547046e-06, "loss": 0.0009858054108917713, "num_tokens": 101334967.0, "reward": 2.3213868141174316, "reward_std": 0.4761294424533844, "rewards/code_complexity_reward/mean": 0.927441418170929, "rewards/code_complexity_reward/std": 0.07903464138507843, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 643, "step_time": 43.96718425117433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 109.4765625, "completions/mean_terminated_length": 108.6888427734375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24121770914644003, "epoch": 0.734321550741163, "frac_reward_zero_std": 0.5625, "grad_norm": 0.0614871010184288, "kl": 0.18735941033810377, "learning_rate": 1.0088784636983472e-06, "loss": 0.0009367465972900391, "num_tokens": 101460307.0, "reward": 2.2500977516174316, "reward_std": 0.5160321593284607, "rewards/code_complexity_reward/mean": 0.8986327648162842, "rewards/code_complexity_reward/std": 0.15150286257266998, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 644, "step_time": 57.19447825755924 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 100.591796875, "completions/mean_terminated_length": 100.591796875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2605228456668556, "epoch": 0.7354618015963512, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.052852217108011246, "kl": 0.20970291714183986, "learning_rate": 1.0009004340263778e-06, "loss": 0.0010486319661140442, "num_tokens": 101582734.0, "reward": 2.2851076126098633, "reward_std": 0.4831497073173523, "rewards/code_complexity_reward/mean": 0.9245116710662842, "rewards/code_complexity_reward/std": 0.10726886987686157, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 645, "step_time": 50.14856502041221 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 291.0, "completions/max_terminated_length": 291.0, "completions/mean_length": 105.4921875, "completions/mean_terminated_length": 105.4921875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2453690911643207, "epoch": 0.7366020524515393, "frac_reward_zero_std": 0.625, "grad_norm": 0.05279261991381645, "kl": 0.2189419160131365, "learning_rate": 9.929461714243166e-07, "loss": 0.001094964798539877, "num_tokens": 101703886.0, "reward": 2.2035157680511475, "reward_std": 0.44728726148605347, "rewards/code_complexity_reward/mean": 0.916015625, "rewards/code_complexity_reward/std": 0.12282288819551468, "rewards/code_execution_reward/mean": 0.1953125, "rewards/code_execution_reward/std": 0.3968288004398346, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 646, "step_time": 43.40766884107143 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 102.38671875, "completions/mean_terminated_length": 101.58512878417969, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23730689333751798, "epoch": 0.7377423033067275, "frac_reward_zero_std": 0.640625, "grad_norm": 0.05245546996593475, "kl": 0.2372375160921365, "learning_rate": 9.850158020008757e-07, "loss": 0.0011865562992170453, "num_tokens": 101824900.0, "reward": 2.310058832168579, "reward_std": 0.49187520146369934, "rewards/code_complexity_reward/mean": 0.9175781011581421, "rewards/code_complexity_reward/std": 0.09918253123760223, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 647, "step_time": 88.92451393604279 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 406.0, "completions/max_terminated_length": 406.0, "completions/mean_length": 102.46875, "completions/mean_terminated_length": 102.46875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24806264555081725, "epoch": 0.7388825541619156, "frac_reward_zero_std": 0.6875, "grad_norm": 0.04581437259912491, "kl": 0.2247301833704114, "learning_rate": 9.771094514859587e-07, "loss": 0.001123942551203072, "num_tokens": 101944340.0, "reward": 2.2886719703674316, "reward_std": 0.47097623348236084, "rewards/code_complexity_reward/mean": 0.9269530773162842, "rewards/code_complexity_reward/std": 0.08894959092140198, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 648, "step_time": 40.57632548362017 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 362.0, "completions/max_terminated_length": 362.0, "completions/mean_length": 99.97265625, "completions/mean_terminated_length": 99.97265625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24036209378391504, "epoch": 0.7400228050171037, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.05631216987967491, "kl": 0.2126726631540805, "learning_rate": 9.692272452286686e-07, "loss": 0.0010633589699864388, "num_tokens": 102065274.0, "reward": 2.2321290969848633, "reward_std": 0.5151786208152771, "rewards/code_complexity_reward/mean": 0.906542956829071, "rewards/code_complexity_reward/std": 0.16218046844005585, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 649, "step_time": 48.267849031835794 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 414.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 104.263671875, "completions/mean_terminated_length": 104.263671875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2513801141176373, "epoch": 0.7411630558722919, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05306233465671539, "kl": 0.2071295592468232, "learning_rate": 9.613693081953195e-07, "loss": 0.0010356390848755836, "num_tokens": 102187957.0, "reward": 2.260547161102295, "reward_std": 0.4894145131111145, "rewards/code_complexity_reward/mean": 0.9144531488418579, "rewards/code_complexity_reward/std": 0.12313617765903473, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 650, "step_time": 42.871707025915384 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 102.41796875, "completions/mean_terminated_length": 102.41796875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24117300123907626, "epoch": 0.74230330672748, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.04981996491551399, "kl": 0.19677597586996853, "learning_rate": 9.535357649674554e-07, "loss": 0.0009840261191129684, "num_tokens": 102308187.0, "reward": 2.2638673782348633, "reward_std": 0.45819681882858276, "rewards/code_complexity_reward/mean": 0.9197266101837158, "rewards/code_complexity_reward/std": 0.0909910798072815, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 651, "step_time": 39.63305279798806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 101.630859375, "completions/mean_terminated_length": 101.630859375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.23431182373315096, "epoch": 0.7434435575826682, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.06025029718875885, "kl": 0.1918134898878634, "learning_rate": 9.457267397398756e-07, "loss": 0.0009592489222995937, "num_tokens": 102428654.0, "reward": 2.2923340797424316, "reward_std": 0.4990314245223999, "rewards/code_complexity_reward/mean": 0.919140636920929, "rewards/code_complexity_reward/std": 0.12120901793241501, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 652, "step_time": 36.91645043250173 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 105.056640625, "completions/mean_terminated_length": 105.056640625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24349431321024895, "epoch": 0.7445838084378563, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.04922100901603699, "kl": 0.20136195793747902, "learning_rate": 9.379423563186652e-07, "loss": 0.0010066409595310688, "num_tokens": 102551223.0, "reward": 2.337646484375, "reward_std": 0.5047959685325623, "rewards/code_complexity_reward/mean": 0.9166015386581421, "rewards/code_complexity_reward/std": 0.11355219781398773, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 653, "step_time": 44.19416109099984 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 100.37890625, "completions/mean_terminated_length": 100.37890625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24672292033210397, "epoch": 0.7457240592930444, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05430770292878151, "kl": 0.2246197247877717, "learning_rate": 9.301827381192321e-07, "loss": 0.0011229044757783413, "num_tokens": 102669745.0, "reward": 2.342578411102295, "reward_std": 0.5001503229141235, "rewards/code_complexity_reward/mean": 0.9251953363418579, "rewards/code_complexity_reward/std": 0.09732207655906677, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 654, "step_time": 64.94891529437155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 101.45703125, "completions/mean_terminated_length": 101.45703125, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.24849953269585967, "epoch": 0.7468643101482326, "frac_reward_zero_std": 0.515625, "grad_norm": 0.056615572422742844, "kl": 0.21669773757457733, "learning_rate": 9.224480081643516e-07, "loss": 0.0010835774010047317, "num_tokens": 102790227.0, "reward": 2.2752928733825684, "reward_std": 0.5151329636573792, "rewards/code_complexity_reward/mean": 0.90869140625, "rewards/code_complexity_reward/std": 0.1434398740530014, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 655, "step_time": 38.966042656451464 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 427.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 103.19140625, "completions/mean_terminated_length": 103.19140625, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.24686659686267376, "epoch": 0.7480045610034207, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05277343839406967, "kl": 0.20823741075582802, "learning_rate": 9.147382890822143e-07, "loss": 0.0010411273688077927, "num_tokens": 102914185.0, "reward": 2.292285442352295, "reward_std": 0.4883916974067688, "rewards/code_complexity_reward/mean": 0.9188476800918579, "rewards/code_complexity_reward/std": 0.10786687582731247, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 656, "step_time": 44.04513622354716 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 299.0, "completions/max_terminated_length": 299.0, "completions/mean_length": 101.53125, "completions/mean_terminated_length": 101.53125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24524473561905324, "epoch": 0.7491448118586089, "frac_reward_zero_std": 0.671875, "grad_norm": 0.043620362877845764, "kl": 0.1999505723360926, "learning_rate": 9.070537031044829e-07, "loss": 0.0009995028376579285, "num_tokens": 103034861.0, "reward": 2.2972168922424316, "reward_std": 0.46637243032455444, "rewards/code_complexity_reward/mean": 0.9291015267372131, "rewards/code_complexity_reward/std": 0.07740136981010437, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 657, "step_time": 34.57185492943972 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 284.0, "completions/max_terminated_length": 284.0, "completions/mean_length": 97.298828125, "completions/mean_terminated_length": 97.298828125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24062561499886215, "epoch": 0.750285062713797, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.048049867153167725, "kl": 0.20179948606528342, "learning_rate": 8.993943720643536e-07, "loss": 0.0010091033764183521, "num_tokens": 103153138.0, "reward": 2.369922161102295, "reward_std": 0.5098958611488342, "rewards/code_complexity_reward/mean": 0.9232422113418579, "rewards/code_complexity_reward/std": 0.09983761608600616, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 658, "step_time": 42.2707738392055 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 101.603515625, "completions/mean_terminated_length": 101.603515625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24602918303571641, "epoch": 0.7514253135689852, "frac_reward_zero_std": 0.625, "grad_norm": 0.04850039258599281, "kl": 0.20921710832044482, "learning_rate": 8.917604173946268e-07, "loss": 0.0010459923651069403, "num_tokens": 103273371.0, "reward": 2.339404582977295, "reward_std": 0.4954353868961334, "rewards/code_complexity_reward/mean": 0.9261718988418579, "rewards/code_complexity_reward/std": 0.09796657413244247, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 659, "step_time": 45.49170074611902 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 284.0, "completions/mean_length": 99.708984375, "completions/mean_terminated_length": 98.90215301513672, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23582958173938096, "epoch": 0.7525655644241733, "frac_reward_zero_std": 0.625, "grad_norm": 0.05018993839621544, "kl": 0.19474807963706553, "learning_rate": 8.841519601257756e-07, "loss": 0.0009737719665281475, "num_tokens": 103393982.0, "reward": 2.3020997047424316, "reward_std": 0.481585294008255, "rewards/code_complexity_reward/mean": 0.9240233898162842, "rewards/code_complexity_reward/std": 0.10019073635339737, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 660, "step_time": 75.63631588593125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 265.0, "completions/max_terminated_length": 265.0, "completions/mean_length": 98.736328125, "completions/mean_terminated_length": 98.736328125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24882838292978704, "epoch": 0.7537058152793614, "frac_reward_zero_std": 0.625, "grad_norm": 0.04994688183069229, "kl": 0.19773256173357368, "learning_rate": 8.765691208840374e-07, "loss": 0.0009886613115668297, "num_tokens": 103513795.0, "reward": 2.266113519668579, "reward_std": 0.48366579413414, "rewards/code_complexity_reward/mean": 0.9200195074081421, "rewards/code_complexity_reward/std": 0.12027610093355179, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 661, "step_time": 32.50380467250943 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 301.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 101.63671875, "completions/mean_terminated_length": 101.63671875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23465591669082642, "epoch": 0.7548460661345496, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.04885907471179962, "kl": 0.20761960744857788, "learning_rate": 8.690120198894919e-07, "loss": 0.0010379807790741324, "num_tokens": 103635773.0, "reward": 2.2750000953674316, "reward_std": 0.4539670944213867, "rewards/code_complexity_reward/mean": 0.925976574420929, "rewards/code_complexity_reward/std": 0.08433591574430466, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 662, "step_time": 42.7682537836954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 330.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 100.65625, "completions/mean_terminated_length": 100.65625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23952380986884236, "epoch": 0.7559863169897377, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.05477761849761009, "kl": 0.20409923372790217, "learning_rate": 8.614807769541589e-07, "loss": 0.001020289957523346, "num_tokens": 103755893.0, "reward": 2.2741212844848633, "reward_std": 0.485039621591568, "rewards/code_complexity_reward/mean": 0.919238269329071, "rewards/code_complexity_reward/std": 0.11410335451364517, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 663, "step_time": 37.04019216168672 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 305.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 101.84765625, "completions/mean_terminated_length": 101.84765625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24305976321920753, "epoch": 0.7571265678449259, "frac_reward_zero_std": 0.609375, "grad_norm": 0.05098705738782883, "kl": 0.19332504109479487, "learning_rate": 8.539755114800996e-07, "loss": 0.0009665852412581444, "num_tokens": 103875431.0, "reward": 2.2746095657348633, "reward_std": 0.49317243695259094, "rewards/code_complexity_reward/mean": 0.9167968034744263, "rewards/code_complexity_reward/std": 0.12155665457248688, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 664, "step_time": 35.48939212039113 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 104.541015625, "completions/mean_terminated_length": 104.541015625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23989511583931744, "epoch": 0.758266818700114, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.04177011922001839, "kl": 0.2016486101783812, "learning_rate": 8.464963424575215e-07, "loss": 0.0010081371292471886, "num_tokens": 103998920.0, "reward": 2.2433106899261475, "reward_std": 0.5039136409759521, "rewards/code_complexity_reward/mean": 0.9052734375, "rewards/code_complexity_reward/std": 0.14639009535312653, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 665, "step_time": 42.81748484168202 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 101.8828125, "completions/mean_terminated_length": 101.8828125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2441236018203199, "epoch": 0.7594070695553021, "frac_reward_zero_std": 0.609375, "grad_norm": 0.05494213104248047, "kl": 0.21631849254481494, "learning_rate": 8.390433884628948e-07, "loss": 0.001081593451090157, "num_tokens": 104119460.0, "reward": 2.27978515625, "reward_std": 0.451727032661438, "rewards/code_complexity_reward/mean": 0.9278320074081421, "rewards/code_complexity_reward/std": 0.06932570785284042, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 666, "step_time": 35.48479483183473 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 309.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 100.03515625, "completions/mean_terminated_length": 100.03515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23952190368436277, "epoch": 0.7605473204104903, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.049277666956186295, "kl": 0.1983214234933257, "learning_rate": 8.316167676570666e-07, "loss": 0.0009913727408275008, "num_tokens": 104238002.0, "reward": 2.333984375, "reward_std": 0.47401905059814453, "rewards/code_complexity_reward/mean": 0.9273437261581421, "rewards/code_complexity_reward/std": 0.069202721118927, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 667, "step_time": 35.66629256401211 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 97.75390625, "completions/mean_terminated_length": 97.75390625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2492623154539615, "epoch": 0.7616875712656784, "frac_reward_zero_std": 0.6640625, "grad_norm": 0.043956317007541656, "kl": 0.20585578470490873, "learning_rate": 8.242165977833974e-07, "loss": 0.0010291829239577055, "num_tokens": 104356372.0, "reward": 2.291308641433716, "reward_std": 0.48326966166496277, "rewards/code_complexity_reward/mean": 0.9217773675918579, "rewards/code_complexity_reward/std": 0.10970234125852585, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 668, "step_time": 35.49989858083427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 98.142578125, "completions/mean_terminated_length": 98.142578125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23563682427629828, "epoch": 0.7628278221208666, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.06012844294309616, "kl": 0.20087731815874577, "learning_rate": 8.168429961658822e-07, "loss": 0.0010042220819741488, "num_tokens": 104475229.0, "reward": 2.3446290493011475, "reward_std": 0.513349711894989, "rewards/code_complexity_reward/mean": 0.92138671875, "rewards/code_complexity_reward/std": 0.11268225312232971, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 669, "step_time": 34.047235652804375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 101.79296875, "completions/mean_terminated_length": 101.79296875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2502584692556411, "epoch": 0.7639680729760547, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.04741930961608887, "kl": 0.20699286996386945, "learning_rate": 8.094960797073023e-07, "loss": 0.0010349510703235865, "num_tokens": 104597791.0, "reward": 2.2359862327575684, "reward_std": 0.45335254073143005, "rewards/code_complexity_reward/mean": 0.91943359375, "rewards/code_complexity_reward/std": 0.10611569136381149, "rewards/code_execution_reward/mean": 0.22265625, "rewards/code_execution_reward/std": 0.41643625497817993, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 670, "step_time": 43.21869311481714 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 101.521484375, "completions/mean_terminated_length": 101.521484375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23956805653870106, "epoch": 0.7651083238312428, "frac_reward_zero_std": 0.640625, "grad_norm": 0.052971623837947845, "kl": 0.19555161870084703, "learning_rate": 8.021759648873642e-07, "loss": 0.0009777500526979566, "num_tokens": 104718826.0, "reward": 2.2975587844848633, "reward_std": 0.5099968910217285, "rewards/code_complexity_reward/mean": 0.913378894329071, "rewards/code_complexity_reward/std": 0.13197916746139526, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 671, "step_time": 48.70804080273956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 285.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 100.63671875, "completions/mean_terminated_length": 100.63671875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24610372865572572, "epoch": 0.766248574686431, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.046418529003858566, "kl": 0.21790832420811057, "learning_rate": 7.94882767760852e-07, "loss": 0.0010898385662585497, "num_tokens": 104840912.0, "reward": 2.291796922683716, "reward_std": 0.49786344170570374, "rewards/code_complexity_reward/mean": 0.9164062738418579, "rewards/code_complexity_reward/std": 0.12047839909791946, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 672, "step_time": 34.34748478513211 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 328.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 101.849609375, "completions/mean_terminated_length": 101.849609375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24576640012674034, "epoch": 0.7673888255416191, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.04432780295610428, "kl": 0.21484419354237616, "learning_rate": 7.876166039557967e-07, "loss": 0.0010742529993876815, "num_tokens": 104960251.0, "reward": 2.2713868618011475, "reward_std": 0.48459044098854065, "rewards/code_complexity_reward/mean": 0.91552734375, "rewards/code_complexity_reward/std": 0.11694963276386261, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 673, "step_time": 36.39842011593282 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 280.0, "completions/max_terminated_length": 280.0, "completions/mean_length": 100.177734375, "completions/mean_terminated_length": 100.177734375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24881549971178174, "epoch": 0.7685290763968073, "frac_reward_zero_std": 0.546875, "grad_norm": 0.055204037576913834, "kl": 0.20818720431998372, "learning_rate": 7.8037758867163e-07, "loss": 0.0010410221293568611, "num_tokens": 105081626.0, "reward": 2.333300828933716, "reward_std": 0.474299818277359, "rewards/code_complexity_reward/mean": 0.9227539300918579, "rewards/code_complexity_reward/std": 0.07088229060173035, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 674, "step_time": 42.94864602573216 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 323.0, "completions/max_terminated_length": 323.0, "completions/mean_length": 101.708984375, "completions/mean_terminated_length": 101.708984375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2407880825921893, "epoch": 0.7696693272519954, "frac_reward_zero_std": 0.640625, "grad_norm": 0.05280233919620514, "kl": 0.1953259091824293, "learning_rate": 7.731658366773717e-07, "loss": 0.0009764598216861486, "num_tokens": 105202885.0, "reward": 2.23291015625, "reward_std": 0.49959465861320496, "rewards/code_complexity_reward/mean": 0.9024413824081421, "rewards/code_complexity_reward/std": 0.14814209938049316, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 675, "step_time": 45.57671476621181 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 311.0, "completions/max_terminated_length": 311.0, "completions/mean_length": 100.232421875, "completions/mean_terminated_length": 100.232421875, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.24582877522334456, "epoch": 0.7708095781071835, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.05966838076710701, "kl": 0.21190296462737024, "learning_rate": 7.659814623097958e-07, "loss": 0.0010594548657536507, "num_tokens": 105322912.0, "reward": 2.293261766433716, "reward_std": 0.4891469180583954, "rewards/code_complexity_reward/mean": 0.9227539300918579, "rewards/code_complexity_reward/std": 0.11263102293014526, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 676, "step_time": 35.786146306432784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 327.0, "completions/mean_length": 101.8515625, "completions/mean_terminated_length": 101.04891967773438, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24693343695253134, "epoch": 0.7719498289623717, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.06015755981206894, "kl": 0.2623262226115912, "learning_rate": 7.588245794716315e-07, "loss": 0.0013112672604620457, "num_tokens": 105443144.0, "reward": 2.2748048305511475, "reward_std": 0.47772032022476196, "rewards/code_complexity_reward/mean": 0.92138671875, "rewards/code_complexity_reward/std": 0.10532138496637344, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 677, "step_time": 48.94601328950375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 406.0, "completions/max_terminated_length": 406.0, "completions/mean_length": 101.255859375, "completions/mean_terminated_length": 101.255859375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23821218986995518, "epoch": 0.7730900798175598, "frac_reward_zero_std": 0.578125, "grad_norm": 0.04563581943511963, "kl": 0.20270211063325405, "learning_rate": 7.516953016297479e-07, "loss": 0.001013528206385672, "num_tokens": 105564987.0, "reward": 2.3775391578674316, "reward_std": 0.5144543647766113, "rewards/code_complexity_reward/mean": 0.9196288585662842, "rewards/code_complexity_reward/std": 0.09722398221492767, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 678, "step_time": 53.3261805344373 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 104.263671875, "completions/mean_terminated_length": 104.263671875, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.2512403135187924, "epoch": 0.774230330672748, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.060301389545202255, "kl": 0.20902026630938053, "learning_rate": 7.445937418133564e-07, "loss": 0.0010450008558109403, "num_tokens": 105688794.0, "reward": 2.228076457977295, "reward_std": 0.48189231753349304, "rewards/code_complexity_reward/mean": 0.9100586175918579, "rewards/code_complexity_reward/std": 0.13781704008579254, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 679, "step_time": 48.94384649023414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 105.65234375, "completions/mean_terminated_length": 105.65234375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24796868325211108, "epoch": 0.7753705815279361, "frac_reward_zero_std": 0.609375, "grad_norm": 0.04963986575603485, "kl": 0.19969322136603296, "learning_rate": 7.375200126122256e-07, "loss": 0.0009984263451769948, "num_tokens": 105812488.0, "reward": 2.2718751430511475, "reward_std": 0.49644532799720764, "rewards/code_complexity_reward/mean": 0.9111328125, "rewards/code_complexity_reward/std": 0.1325022131204605, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 680, "step_time": 45.82249296922237 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 394.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 104.04296875, "completions/mean_terminated_length": 104.04296875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23661412042565644, "epoch": 0.7765108323831242, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.05440690740942955, "kl": 0.22408254793845117, "learning_rate": 7.304742261748848e-07, "loss": 0.001120448810979724, "num_tokens": 105932418.0, "reward": 2.3345704078674316, "reward_std": 0.5353421568870544, "rewards/code_complexity_reward/mean": 0.9103515148162842, "rewards/code_complexity_reward/std": 0.13566632568836212, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 681, "step_time": 40.551005645655096 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 105.1875, "completions/mean_terminated_length": 104.39138793945312, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.23851076909340918, "epoch": 0.7776510832383124, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.04984547942876816, "kl": 0.21021209377795458, "learning_rate": 7.23456494206859e-07, "loss": 0.0010509021813049912, "num_tokens": 106055334.0, "reward": 2.320849895477295, "reward_std": 0.506828248500824, "rewards/code_complexity_reward/mean": 0.9173828363418579, "rewards/code_complexity_reward/std": 0.11450815945863724, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 682, "step_time": 55.46475271321833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 352.0, "completions/max_terminated_length": 352.0, "completions/mean_length": 104.658203125, "completions/mean_terminated_length": 104.658203125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24280061409808695, "epoch": 0.7787913340935005, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.05195563659071922, "kl": 0.1958033146802336, "learning_rate": 7.164669279688846e-07, "loss": 0.0009788157185539603, "num_tokens": 106177507.0, "reward": 2.2798829078674316, "reward_std": 0.5024067759513855, "rewards/code_complexity_reward/mean": 0.9152343273162842, "rewards/code_complexity_reward/std": 0.13138708472251892, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 683, "step_time": 38.51279195211828 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 467.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 106.666015625, "completions/mean_terminated_length": 106.666015625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24669361766427755, "epoch": 0.7799315849486887, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05414969101548195, "kl": 0.20149118825793266, "learning_rate": 7.095056382751559e-07, "loss": 0.0010073366574943066, "num_tokens": 106301012.0, "reward": 2.2543458938598633, "reward_std": 0.4610588550567627, "rewards/code_complexity_reward/mean": 0.921191394329071, "rewards/code_complexity_reward/std": 0.1013854444026947, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 684, "step_time": 46.03433496132493 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 104.54296875, "completions/mean_terminated_length": 104.54296875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.25388529943302274, "epoch": 0.7810718358038768, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.06584639847278595, "kl": 0.19804327958263457, "learning_rate": 7.025727354915655e-07, "loss": 0.0009901986923068762, "num_tokens": 106425122.0, "reward": 2.3270020484924316, "reward_std": 0.5299767255783081, "rewards/code_complexity_reward/mean": 0.9079101085662842, "rewards/code_complexity_reward/std": 0.137708842754364, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 685, "step_time": 36.12821339350194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 401.0, "completions/max_terminated_length": 401.0, "completions/mean_length": 102.990234375, "completions/mean_terminated_length": 102.990234375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23828665167093277, "epoch": 0.7822120866590649, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.05732176452875137, "kl": 0.21939418255351484, "learning_rate": 6.956683295339483e-07, "loss": 0.0010971357114613056, "num_tokens": 106545717.0, "reward": 2.2904298305511475, "reward_std": 0.5166562795639038, "rewards/code_complexity_reward/mean": 0.9111328125, "rewards/code_complexity_reward/std": 0.13509830832481384, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 686, "step_time": 42.475438771769404 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 327.0, "completions/max_terminated_length": 327.0, "completions/mean_length": 99.345703125, "completions/mean_terminated_length": 99.345703125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24409410054795444, "epoch": 0.7833523375142531, "frac_reward_zero_std": 0.6875, "grad_norm": 0.05899887531995773, "kl": 0.20699134352616966, "learning_rate": 6.887925298663506e-07, "loss": 0.001035055611282587, "num_tokens": 106664418.0, "reward": 2.28173828125, "reward_std": 0.4983568787574768, "rewards/code_complexity_reward/mean": 0.9161132574081421, "rewards/code_complexity_reward/std": 0.12370317429304123, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 687, "step_time": 37.19725839421153 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 426.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 101.080078125, "completions/mean_terminated_length": 101.080078125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24072536104358733, "epoch": 0.7844925883694412, "frac_reward_zero_std": 0.578125, "grad_norm": 0.06016572564840317, "kl": 0.20932246651500463, "learning_rate": 6.81945445499281e-07, "loss": 0.0010466128587722778, "num_tokens": 106786951.0, "reward": 2.3644533157348633, "reward_std": 0.5261820554733276, "rewards/code_complexity_reward/mean": 0.919238269329071, "rewards/code_complexity_reward/std": 0.12097129970788956, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 688, "step_time": 45.063663891516626 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 250.0, "completions/max_terminated_length": 250.0, "completions/mean_length": 101.416015625, "completions/mean_terminated_length": 101.416015625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2472120299935341, "epoch": 0.7856328392246295, "frac_reward_zero_std": 0.5625, "grad_norm": 0.06879277527332306, "kl": 0.20329043362289667, "learning_rate": 6.751271849879959e-07, "loss": 0.0010164622217416763, "num_tokens": 106908812.0, "reward": 2.3475587368011475, "reward_std": 0.5048385262489319, "rewards/code_complexity_reward/mean": 0.92236328125, "rewards/code_complexity_reward/std": 0.10064800083637238, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 689, "step_time": 34.174394665285945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 275.0, "completions/max_terminated_length": 275.0, "completions/mean_length": 102.572265625, "completions/mean_terminated_length": 102.572265625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.25066910590976477, "epoch": 0.7867730900798175, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.04834514483809471, "kl": 0.20019438839517534, "learning_rate": 6.68337856430766e-07, "loss": 0.0010008974932134151, "num_tokens": 107029573.0, "reward": 2.2920899391174316, "reward_std": 0.492190420627594, "rewards/code_complexity_reward/mean": 0.921582043170929, "rewards/code_complexity_reward/std": 0.11351025104522705, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 690, "step_time": 34.767037304118276 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 271.0, "completions/max_terminated_length": 271.0, "completions/mean_length": 103.09375, "completions/mean_terminated_length": 103.09375, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.24359893100336194, "epoch": 0.7879133409350056, "frac_reward_zero_std": 0.546875, "grad_norm": 0.08438313752412796, "kl": 0.2630957511719316, "learning_rate": 6.615775674671706e-07, "loss": 0.0013153948821127415, "num_tokens": 107150945.0, "reward": 2.299072265625, "reward_std": 0.48499035835266113, "rewards/code_complexity_reward/mean": 0.9258788824081421, "rewards/code_complexity_reward/std": 0.10510086268186569, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 691, "step_time": 34.38354547228664 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 427.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 102.875, "completions/mean_terminated_length": 102.875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.248255972051993, "epoch": 0.7890535917901939, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.043279778212308884, "kl": 0.19400394102558494, "learning_rate": 6.548464252763906e-07, "loss": 0.0009698772337287664, "num_tokens": 107274009.0, "reward": 2.2909178733825684, "reward_std": 0.5050533413887024, "rewards/code_complexity_reward/mean": 0.91259765625, "rewards/code_complexity_reward/std": 0.12856459617614746, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 692, "step_time": 52.84984053298831 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 98.287109375, "completions/mean_terminated_length": 98.287109375, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.25179400155320764, "epoch": 0.790193842645382, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.05790230631828308, "kl": 0.20191916660405695, "learning_rate": 6.481445365755021e-07, "loss": 0.0010095520410686731, "num_tokens": 107392120.0, "reward": 2.280761957168579, "reward_std": 0.5374943614006042, "rewards/code_complexity_reward/mean": 0.9083007574081421, "rewards/code_complexity_reward/std": 0.16194887459278107, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 693, "step_time": 44.00715931970626 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 103.212890625, "completions/mean_terminated_length": 103.212890625, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.24306918121874332, "epoch": 0.7913340935005702, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05552439019083977, "kl": 0.19198064785450697, "learning_rate": 6.414720076177958e-07, "loss": 0.0009597428143024445, "num_tokens": 107514261.0, "reward": 2.3083009719848633, "reward_std": 0.5428637862205505, "rewards/code_complexity_reward/mean": 0.9084960222244263, "rewards/code_complexity_reward/std": 0.15432791411876678, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 694, "step_time": 43.39767027460039 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 287.0, "completions/max_terminated_length": 287.0, "completions/mean_length": 101.83984375, "completions/mean_terminated_length": 101.83984375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23378529795445502, "epoch": 0.7924743443557583, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.049058061093091965, "kl": 0.20689011714421213, "learning_rate": 6.348289441910807e-07, "loss": 0.0010342153254896402, "num_tokens": 107635543.0, "reward": 2.2549805641174316, "reward_std": 0.4867416322231293, "rewards/code_complexity_reward/mean": 0.915722668170929, "rewards/code_complexity_reward/std": 0.12726178765296936, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 695, "step_time": 42.86570569500327 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 453.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 107.357421875, "completions/mean_terminated_length": 107.357421875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2430480478797108, "epoch": 0.7936145952109465, "frac_reward_zero_std": 0.59375, "grad_norm": 0.051718901842832565, "kl": 0.20211503258906305, "learning_rate": 6.282154516160157e-07, "loss": 0.0010105764959007502, "num_tokens": 107759754.0, "reward": 2.295459270477295, "reward_std": 0.5160062313079834, "rewards/code_complexity_reward/mean": 0.9115234613418579, "rewards/code_complexity_reward/std": 0.13360872864723206, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 696, "step_time": 52.717568319290876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 371.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 106.619140625, "completions/mean_terminated_length": 106.619140625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.25399909913539886, "epoch": 0.7947548460661346, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.05242061987519264, "kl": 0.19909470481798053, "learning_rate": 6.216316347444362e-07, "loss": 0.0009953530970960855, "num_tokens": 107883363.0, "reward": 2.2099609375, "reward_std": 0.4639851450920105, "rewards/code_complexity_reward/mean": 0.9107421636581421, "rewards/code_complexity_reward/std": 0.13792505860328674, "rewards/code_execution_reward/mean": 0.208984375, "rewards/code_execution_reward/std": 0.40698084235191345, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 697, "step_time": 47.48484544362873 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 103.5234375, "completions/mean_terminated_length": 103.5234375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24276013765484095, "epoch": 0.7958950969213227, "frac_reward_zero_std": 0.625, "grad_norm": 0.0524413175880909, "kl": 0.20975171274039894, "learning_rate": 6.150775979576906e-07, "loss": 0.0010487439576536417, "num_tokens": 108007839.0, "reward": 2.313281536102295, "reward_std": 0.49715372920036316, "rewards/code_complexity_reward/mean": 0.9164062738418579, "rewards/code_complexity_reward/std": 0.1137528121471405, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 698, "step_time": 43.723106822930276 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 107.11328125, "completions/mean_terminated_length": 107.11328125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2503434296231717, "epoch": 0.7970353477765109, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05265442281961441, "kl": 0.19784958777017891, "learning_rate": 6.085534451649905e-07, "loss": 0.0009892585221678019, "num_tokens": 108131205.0, "reward": 2.2554688453674316, "reward_std": 0.4941729009151459, "rewards/code_complexity_reward/mean": 0.9132812023162842, "rewards/code_complexity_reward/std": 0.13573519885540009, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 699, "step_time": 51.68148906622082 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 101.37890625, "completions/mean_terminated_length": 101.37890625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24243175284937024, "epoch": 0.798175598631699, "frac_reward_zero_std": 0.59375, "grad_norm": 0.04741673916578293, "kl": 0.20587884401902556, "learning_rate": 6.020592798017554e-07, "loss": 0.0010293842060491443, "num_tokens": 108252015.0, "reward": 2.3122072219848633, "reward_std": 0.4896106719970703, "rewards/code_complexity_reward/mean": 0.918261706829071, "rewards/code_complexity_reward/std": 0.10451409220695496, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 700, "step_time": 36.499849361367524 }, { "epoch": 0.798175598631699, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 148.32, "eval_completions/max_terminated_length": 148.32, "eval_completions/mean_length": 104.0625, "eval_completions/mean_terminated_length": 104.0625, "eval_completions/min_length": 72.98, "eval_completions/min_terminated_length": 72.98, "eval_entropy": 0.2448738557100296, "eval_frac_reward_zero_std": 0.51, "eval_kl": 0.20988993167877198, "eval_loss": 0.001050163758918643, "eval_num_tokens": 108252015.0, "eval_reward": 2.2619376087188723, "eval_reward_std": 0.3678464848548174, "eval_rewards/code_complexity_reward/mean": 0.9172499740123748, "eval_rewards/code_complexity_reward/std": 0.05878296665847302, "eval_rewards/code_execution_reward/mean": 0.2525, "eval_rewards/code_execution_reward/std": 0.3116739410161972, "eval_rewards/code_syntax_reward/mean": 0.4925, "eval_rewards/code_syntax_reward/std": 0.021213203072547912, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.4996875, "eval_rewards/xmlcount_reward_func/std": 0.000883883461356163, "eval_runtime": 343.3801, "eval_samples_per_second": 0.291, "eval_steps_per_second": 0.038, "step": 700 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 108.474609375, "completions/mean_terminated_length": 107.68492889404297, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24301723181270063, "epoch": 0.7993158494868872, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05271249637007713, "kl": 0.20316160446964204, "learning_rate": 5.955952048279795e-07, "loss": 0.0010157802607864141, "num_tokens": 108379242.0, "reward": 2.262012004852295, "reward_std": 0.48320093750953674, "rewards/code_complexity_reward/mean": 0.9156249761581421, "rewards/code_complexity_reward/std": 0.11471908539533615, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 701, "step_time": 54.609510853886604 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 106.546875, "completions/mean_terminated_length": 106.546875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.25252994755283, "epoch": 0.8004561003420753, "frac_reward_zero_std": 0.609375, "grad_norm": 0.04571513459086418, "kl": 0.196490183705464, "learning_rate": 5.891613227265971e-07, "loss": 0.0009825568413361907, "num_tokens": 108502722.0, "reward": 2.292187452316284, "reward_std": 0.4957525432109833, "rewards/code_complexity_reward/mean": 0.919726550579071, "rewards/code_complexity_reward/std": 0.1199377030134201, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 702, "step_time": 41.060339806601405 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 101.828125, "completions/mean_terminated_length": 101.828125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24352534534409642, "epoch": 0.8015963511972634, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.06089961901307106, "kl": 0.23261151392944157, "learning_rate": 5.827577355018577e-07, "loss": 0.001162544242106378, "num_tokens": 108622758.0, "reward": 2.263476848602295, "reward_std": 0.4806572198867798, "rewards/code_complexity_reward/mean": 0.9183593988418579, "rewards/code_complexity_reward/std": 0.11851511895656586, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 703, "step_time": 40.73301642946899 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 359.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 104.1953125, "completions/mean_terminated_length": 104.1953125, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.2541432923171669, "epoch": 0.8027366020524516, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.07721440494060516, "kl": 0.21571039338596165, "learning_rate": 5.763845446777077e-07, "loss": 0.001078541623428464, "num_tokens": 108744058.0, "reward": 2.2216796875, "reward_std": 0.4828554689884186, "rewards/code_complexity_reward/mean": 0.9058593511581421, "rewards/code_complexity_reward/std": 0.14356668293476105, "rewards/code_execution_reward/mean": 0.2265625, "rewards/code_execution_reward/std": 0.4190165400505066, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 704, "step_time": 38.13861049618572 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 101.5546875, "completions/mean_terminated_length": 101.5546875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24813633295707405, "epoch": 0.8038768529076397, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.053264446556568146, "kl": 0.20035768998786807, "learning_rate": 5.700418512961825e-07, "loss": 0.0010016849264502525, "num_tokens": 108863726.0, "reward": 2.3074707984924316, "reward_std": 0.5079351663589478, "rewards/code_complexity_reward/mean": 0.917675793170929, "rewards/code_complexity_reward/std": 0.12015591561794281, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 705, "step_time": 44.14198759943247 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 382.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 106.18359375, "completions/mean_terminated_length": 106.18359375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24840082740411162, "epoch": 0.8050171037628279, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.04572787508368492, "kl": 0.2187145217321813, "learning_rate": 5.637297559158067e-07, "loss": 0.0010935761965811253, "num_tokens": 108986204.0, "reward": 2.3028321266174316, "reward_std": 0.4708288908004761, "rewards/code_complexity_reward/mean": 0.926464855670929, "rewards/code_complexity_reward/std": 0.0794292464852333, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 706, "step_time": 39.657482699491084 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 350.0, "completions/max_terminated_length": 350.0, "completions/mean_length": 102.73046875, "completions/mean_terminated_length": 102.73046875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24915859196335077, "epoch": 0.806157354618016, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.04730569198727608, "kl": 0.21218205895274878, "learning_rate": 5.574483586099924e-07, "loss": 0.001060984330251813, "num_tokens": 109108530.0, "reward": 2.302539110183716, "reward_std": 0.5145471096038818, "rewards/code_complexity_reward/mean": 0.9144531488418579, "rewards/code_complexity_reward/std": 0.13046692311763763, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 707, "step_time": 37.56925188563764 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 314.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 101.939453125, "completions/mean_terminated_length": 101.939453125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2357964210677892, "epoch": 0.8072976054732041, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05092599242925644, "kl": 0.19224520563147962, "learning_rate": 5.511977589654601e-07, "loss": 0.0009611106943339109, "num_tokens": 109229679.0, "reward": 2.2736330032348633, "reward_std": 0.5095930695533752, "rewards/code_complexity_reward/mean": 0.9080077409744263, "rewards/code_complexity_reward/std": 0.13708002865314484, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 708, "step_time": 36.92702592909336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 100.755859375, "completions/mean_terminated_length": 100.755859375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24391159531660378, "epoch": 0.8084378563283923, "frac_reward_zero_std": 0.65625, "grad_norm": 0.058646310120821, "kl": 0.21903261565603316, "learning_rate": 5.449780560806572e-07, "loss": 0.001095225801691413, "num_tokens": 109350238.0, "reward": 2.339648485183716, "reward_std": 0.5068226456642151, "rewards/code_complexity_reward/mean": 0.9232422113418579, "rewards/code_complexity_reward/std": 0.10789225250482559, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 709, "step_time": 38.488920538686216 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 335.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 103.126953125, "completions/mean_terminated_length": 103.126953125, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.24605153850279748, "epoch": 0.8095781071835804, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.06817112863063812, "kl": 0.21711121150292456, "learning_rate": 5.387893485641862e-07, "loss": 0.0010853046551346779, "num_tokens": 109469531.0, "reward": 2.319385051727295, "reward_std": 0.4866379499435425, "rewards/code_complexity_reward/mean": 0.9208008050918579, "rewards/code_complexity_reward/std": 0.09907568246126175, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 710, "step_time": 45.09415065776557 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 283.0, "completions/mean_length": 105.34765625, "completions/mean_terminated_length": 103.75294494628906, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24701650231145322, "epoch": 0.8107183580387686, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05504023656249046, "kl": 0.20434214896522462, "learning_rate": 5.326317345332416e-07, "loss": 0.00102146971039474, "num_tokens": 109590141.0, "reward": 2.2587890625, "reward_std": 0.5042368769645691, "rewards/code_complexity_reward/mean": 0.9122070074081421, "rewards/code_complexity_reward/std": 0.1402835100889206, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 711, "step_time": 55.74226041324437 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 317.0, "completions/max_terminated_length": 317.0, "completions/mean_length": 100.341796875, "completions/mean_terminated_length": 100.341796875, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.24975518183782697, "epoch": 0.8118586088939567, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05834927409887314, "kl": 0.23118917155079544, "learning_rate": 5.26505311612055e-07, "loss": 0.0011563901789486408, "num_tokens": 109710204.0, "reward": 2.2959961891174316, "reward_std": 0.466938853263855, "rewards/code_complexity_reward/mean": 0.925488293170929, "rewards/code_complexity_reward/std": 0.08264082670211792, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 712, "step_time": 36.410879210568964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 103.529296875, "completions/mean_terminated_length": 102.72994232177734, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24925118358805776, "epoch": 0.8129988597491448, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.052306223660707474, "kl": 0.20511854765936732, "learning_rate": 5.204101769303474e-07, "loss": 0.0010256670648232102, "num_tokens": 109830751.0, "reward": 2.321044921875, "reward_std": 0.5197566747665405, "rewards/code_complexity_reward/mean": 0.9141601324081421, "rewards/code_complexity_reward/std": 0.13145162165164948, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 713, "step_time": 65.34411282278597 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 197.0, "completions/max_terminated_length": 197.0, "completions/mean_length": 96.748046875, "completions/mean_terminated_length": 96.748046875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2458487378899008, "epoch": 0.814139110604333, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.054434821009635925, "kl": 0.21854134555906057, "learning_rate": 5.143464271217876e-07, "loss": 0.001092600403353572, "num_tokens": 109947014.0, "reward": 2.33984375, "reward_std": 0.4964299201965332, "rewards/code_complexity_reward/mean": 0.9302734136581421, "rewards/code_complexity_reward/std": 0.0963701456785202, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 714, "step_time": 30.293210967443883 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 287.0, "completions/max_terminated_length": 287.0, "completions/mean_length": 100.296875, "completions/mean_terminated_length": 100.296875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23536292696371675, "epoch": 0.8152793614595211, "frac_reward_zero_std": 0.671875, "grad_norm": 0.046376582235097885, "kl": 0.2038064857479185, "learning_rate": 5.083141583224627e-07, "loss": 0.0010189954191446304, "num_tokens": 110066062.0, "reward": 2.29736328125, "reward_std": 0.48654890060424805, "rewards/code_complexity_reward/mean": 0.9219726324081421, "rewards/code_complexity_reward/std": 0.10649465024471283, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 715, "step_time": 35.62514796666801 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 108.462890625, "completions/mean_terminated_length": 108.462890625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2521156568545848, "epoch": 0.8164196123147093, "frac_reward_zero_std": 0.6796875, "grad_norm": 0.04489089921116829, "kl": 0.20079332310706377, "learning_rate": 5.023134661693518e-07, "loss": 0.0010039483895525336, "num_tokens": 110194087.0, "reward": 2.2245118618011475, "reward_std": 0.4570065140724182, "rewards/code_complexity_reward/mean": 0.91455078125, "rewards/code_complexity_reward/std": 0.11598380655050278, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 716, "step_time": 49.86968694161624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 103.353515625, "completions/mean_terminated_length": 103.353515625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24281382816843688, "epoch": 0.8175598631698974, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05278758332133293, "kl": 0.21260083327069879, "learning_rate": 4.963444457988109e-07, "loss": 0.0010629248572513461, "num_tokens": 110316036.0, "reward": 2.2450196743011475, "reward_std": 0.4782138466835022, "rewards/code_complexity_reward/mean": 0.91943359375, "rewards/code_complexity_reward/std": 0.1278616487979889, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 717, "step_time": 46.82540684659034 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 280.0, "completions/max_terminated_length": 280.0, "completions/mean_length": 100.595703125, "completions/mean_terminated_length": 100.595703125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24684189655818045, "epoch": 0.8187001140250855, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.04914984107017517, "kl": 0.1960437116213143, "learning_rate": 4.904071918450643e-07, "loss": 0.0009801870910450816, "num_tokens": 110436493.0, "reward": 2.3460936546325684, "reward_std": 0.5042426586151123, "rewards/code_complexity_reward/mean": 0.923828125, "rewards/code_complexity_reward/std": 0.10528404265642166, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 718, "step_time": 41.72013191320002 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 103.6015625, "completions/mean_terminated_length": 103.6015625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23729397263377905, "epoch": 0.8198403648802737, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05835561454296112, "kl": 0.1981233104597777, "learning_rate": 4.84501798438704e-07, "loss": 0.0009907083585858345, "num_tokens": 110558305.0, "reward": 2.280566453933716, "reward_std": 0.47736257314682007, "rewards/code_complexity_reward/mean": 0.9208008050918579, "rewards/code_complexity_reward/std": 0.10677601397037506, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 719, "step_time": 61.57203767821193 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 101.623046875, "completions/mean_terminated_length": 101.623046875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24333718628622591, "epoch": 0.8209806157354618, "frac_reward_zero_std": 0.53125, "grad_norm": 0.057371966540813446, "kl": 0.2052890881896019, "learning_rate": 4.78628359205198e-07, "loss": 0.0010262478608638048, "num_tokens": 110682096.0, "reward": 2.25634765625, "reward_std": 0.47016024589538574, "rewards/code_complexity_reward/mean": 0.9200195074081421, "rewards/code_complexity_reward/std": 0.10833466053009033, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 720, "step_time": 38.2635377375409 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 279.0, "completions/max_terminated_length": 279.0, "completions/mean_length": 99.58984375, "completions/mean_terminated_length": 99.58984375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2467581802047789, "epoch": 0.82212086659065, "frac_reward_zero_std": 0.609375, "grad_norm": 0.05422898009419441, "kl": 0.20660232938826084, "learning_rate": 4.727869672634044e-07, "loss": 0.0010328821372240782, "num_tokens": 110801450.0, "reward": 2.3102540969848633, "reward_std": 0.4914109408855438, "rewards/code_complexity_reward/mean": 0.921191394329071, "rewards/code_complexity_reward/std": 0.10517502576112747, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 721, "step_time": 44.08141899295151 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 106.220703125, "completions/mean_terminated_length": 106.220703125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24037073249928653, "epoch": 0.8232611174458381, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05391684174537659, "kl": 0.2070755665190518, "learning_rate": 4.669777152240976e-07, "loss": 0.0010353944962844253, "num_tokens": 110924835.0, "reward": 2.2695798873901367, "reward_std": 0.48121634125709534, "rewards/code_complexity_reward/mean": 0.9154297113418579, "rewards/code_complexity_reward/std": 0.11810726672410965, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 722, "step_time": 51.00942743103951 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 401.0, "completions/max_terminated_length": 401.0, "completions/mean_length": 102.5703125, "completions/mean_terminated_length": 102.5703125, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.24549208139069378, "epoch": 0.8244013683010262, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.06590178608894348, "kl": 0.20375408767722547, "learning_rate": 4.612006951884973e-07, "loss": 0.0010186012368649244, "num_tokens": 111047455.0, "reward": 2.294921875, "reward_std": 0.5046104788780212, "rewards/code_complexity_reward/mean": 0.9136718511581421, "rewards/code_complexity_reward/std": 0.12318582832813263, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 723, "step_time": 48.83771118707955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 100.337890625, "completions/mean_terminated_length": 99.53228759765625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2391695014666766, "epoch": 0.8255416191562144, "frac_reward_zero_std": 0.609375, "grad_norm": 0.047957196831703186, "kl": 0.2052069460041821, "learning_rate": 4.5545599874681076e-07, "loss": 0.0010260913986712694, "num_tokens": 111168100.0, "reward": 2.310302734375, "reward_std": 0.5079048275947571, "rewards/code_complexity_reward/mean": 0.9126952886581421, "rewards/code_complexity_reward/std": 0.11970674991607666, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 724, "step_time": 49.40463504195213 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 319.0, "completions/max_terminated_length": 319.0, "completions/mean_length": 103.6875, "completions/mean_terminated_length": 103.6875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24652637145482004, "epoch": 0.8266818700114025, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05596977844834328, "kl": 0.2231048702960834, "learning_rate": 4.4974371697677847e-07, "loss": 0.0011152348015457392, "num_tokens": 111288292.0, "reward": 2.32080078125, "reward_std": 0.5100669264793396, "rewards/code_complexity_reward/mean": 0.9219726324081421, "rewards/code_complexity_reward/std": 0.1218762919306755, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 725, "step_time": 36.84105319343507 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 353.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 100.58984375, "completions/mean_terminated_length": 100.58984375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24415309051983058, "epoch": 0.8278221208665907, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.053048472851514816, "kl": 0.21271071932278574, "learning_rate": 4.4406394044223174e-07, "loss": 0.0010634780628606677, "num_tokens": 111408686.0, "reward": 2.264697551727295, "reward_std": 0.4992167055606842, "rewards/code_complexity_reward/mean": 0.9110351800918579, "rewards/code_complexity_reward/std": 0.13377808034420013, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 726, "step_time": 40.755146680399776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 102.29296875, "completions/mean_terminated_length": 102.29296875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.24966220906935632, "epoch": 0.8289623717217788, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.053513024002313614, "kl": 0.20755530218593776, "learning_rate": 4.384167591916566e-07, "loss": 0.0010376188438385725, "num_tokens": 111528256.0, "reward": 2.3271484375, "reward_std": 0.5224746465682983, "rewards/code_complexity_reward/mean": 0.9195312261581421, "rewards/code_complexity_reward/std": 0.12702132761478424, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 727, "step_time": 49.82554463110864 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 105.658203125, "completions/mean_terminated_length": 105.658203125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2426525808405131, "epoch": 0.830102622576967, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.0643443688750267, "kl": 0.2150881071574986, "learning_rate": 4.328022627567657e-07, "loss": 0.0010751127265393734, "num_tokens": 111650865.0, "reward": 2.236328125, "reward_std": 0.47701430320739746, "rewards/code_complexity_reward/mean": 0.9117187261581421, "rewards/code_complexity_reward/std": 0.1289706826210022, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 728, "step_time": 48.77645061071962 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 378.0, "completions/max_terminated_length": 378.0, "completions/mean_length": 109.494140625, "completions/mean_terminated_length": 109.494140625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23554831836372614, "epoch": 0.8312428734321551, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.05378365516662598, "kl": 0.2197300810366869, "learning_rate": 4.2722054015107874e-07, "loss": 0.001098728971555829, "num_tokens": 111775642.0, "reward": 2.3001465797424316, "reward_std": 0.5500722527503967, "rewards/code_complexity_reward/mean": 0.8957030773162842, "rewards/code_complexity_reward/std": 0.16570165753364563, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 729, "step_time": 41.596001693978906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 105.3125, "completions/mean_terminated_length": 105.3125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24965189583599567, "epoch": 0.8323831242873432, "frac_reward_zero_std": 0.59375, "grad_norm": 0.052754998207092285, "kl": 0.2150946136098355, "learning_rate": 4.216716798685125e-07, "loss": 0.0010755739640444517, "num_tokens": 111897526.0, "reward": 2.3373045921325684, "reward_std": 0.4916175901889801, "rewards/code_complexity_reward/mean": 0.923828125, "rewards/code_complexity_reward/std": 0.08863276988267899, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 730, "step_time": 41.25046293158084 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 371.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 104.263671875, "completions/mean_terminated_length": 104.263671875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2370694950222969, "epoch": 0.8335233751425314, "frac_reward_zero_std": 0.578125, "grad_norm": 0.06542468816041946, "kl": 0.21138844965025783, "learning_rate": 4.161557698819757e-07, "loss": 0.0010565794073045254, "num_tokens": 112018865.0, "reward": 2.3439455032348633, "reward_std": 0.529226541519165, "rewards/code_complexity_reward/mean": 0.914355456829071, "rewards/code_complexity_reward/std": 0.1304589807987213, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.024685947224497795, "step": 731, "step_time": 40.32876432873309 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 101.498046875, "completions/mean_terminated_length": 101.498046875, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.23904610937461257, "epoch": 0.8346636259977195, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.051389992237091064, "kl": 0.20211225026287138, "learning_rate": 4.106728976419763e-07, "loss": 0.0010106090921908617, "num_tokens": 112137204.0, "reward": 2.3357911109924316, "reward_std": 0.48637449741363525, "rewards/code_complexity_reward/mean": 0.9286133050918579, "rewards/code_complexity_reward/std": 0.08148950338363647, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 732, "step_time": 48.769389348104596 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 103.67578125, "completions/mean_terminated_length": 103.67578125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2495288390200585, "epoch": 0.8358038768529077, "frac_reward_zero_std": 0.484375, "grad_norm": 0.0686315968632698, "kl": 0.20191925461404026, "learning_rate": 4.0522315007523486e-07, "loss": 0.0010092363227158785, "num_tokens": 112260494.0, "reward": 2.279541015625, "reward_std": 0.5009117722511292, "rewards/code_complexity_reward/mean": 0.9122070074081421, "rewards/code_complexity_reward/std": 0.1262604296207428, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 733, "step_time": 50.709065459668636 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 101.638671875, "completions/mean_terminated_length": 101.638671875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2383534589316696, "epoch": 0.8369441277080958, "frac_reward_zero_std": 0.578125, "grad_norm": 0.0565103255212307, "kl": 0.2035371728707105, "learning_rate": 3.998066135833031e-07, "loss": 0.001017640344798565, "num_tokens": 112380769.0, "reward": 2.292285203933716, "reward_std": 0.4837820529937744, "rewards/code_complexity_reward/mean": 0.9198241829872131, "rewards/code_complexity_reward/std": 0.10361644625663757, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 734, "step_time": 39.487962127663195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 101.767578125, "completions/mean_terminated_length": 101.767578125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23470477596856654, "epoch": 0.8380843785632839, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.054557058960199356, "kl": 0.20834392309188843, "learning_rate": 3.944233740412001e-07, "loss": 0.0010417391313239932, "num_tokens": 112501454.0, "reward": 2.2630860805511475, "reward_std": 0.48899298906326294, "rewards/code_complexity_reward/mean": 0.916015625, "rewards/code_complexity_reward/std": 0.11487824469804764, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.04406425356864929, "step": 735, "step_time": 37.77514402382076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 289.0, "completions/max_terminated_length": 289.0, "completions/mean_length": 100.234375, "completions/mean_terminated_length": 100.234375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24793117842637002, "epoch": 0.8392246294184721, "frac_reward_zero_std": 0.65625, "grad_norm": 0.04835471510887146, "kl": 0.2018408733420074, "learning_rate": 3.890735167960455e-07, "loss": 0.0010092142038047314, "num_tokens": 112620598.0, "reward": 2.2466797828674316, "reward_std": 0.4596393406391144, "rewards/code_complexity_reward/mean": 0.922558605670929, "rewards/code_complexity_reward/std": 0.10696807503700256, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 736, "step_time": 34.20434008538723 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 231.0, "completions/max_terminated_length": 231.0, "completions/mean_length": 97.263671875, "completions/mean_terminated_length": 97.263671875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24163653934374452, "epoch": 0.8403648802736602, "frac_reward_zero_std": 0.546875, "grad_norm": 0.06909804046154022, "kl": 0.21280664601363242, "learning_rate": 3.8375712666570865e-07, "loss": 0.0010641436092555523, "num_tokens": 112741169.0, "reward": 2.3174805641174316, "reward_std": 0.5323048233985901, "rewards/code_complexity_reward/mean": 0.914746105670929, "rewards/code_complexity_reward/std": 0.14461073279380798, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 737, "step_time": 38.030350690707564 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 104.916015625, "completions/mean_terminated_length": 104.916015625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2550750400405377, "epoch": 0.8415051311288484, "frac_reward_zero_std": 0.65625, "grad_norm": 0.045421574264764786, "kl": 0.2012863770360127, "learning_rate": 3.784742879374631e-07, "loss": 0.0010065321112051606, "num_tokens": 112862906.0, "reward": 2.285937786102295, "reward_std": 0.461711585521698, "rewards/code_complexity_reward/mean": 0.9232422113418579, "rewards/code_complexity_reward/std": 0.08356557786464691, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 738, "step_time": 41.602230842225254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 341.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 101.6640625, "completions/mean_terminated_length": 101.6640625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24946099054068327, "epoch": 0.8426453819840365, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.054068371653556824, "kl": 0.2037520012818277, "learning_rate": 3.7322508436665184e-07, "loss": 0.001018680166453123, "num_tokens": 112982354.0, "reward": 2.313769817352295, "reward_std": 0.5086346864700317, "rewards/code_complexity_reward/mean": 0.9168944954872131, "rewards/code_complexity_reward/std": 0.12164368480443954, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 739, "step_time": 38.04164169728756 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 389.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 108.697265625, "completions/mean_terminated_length": 108.697265625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2488474368583411, "epoch": 0.8437856328392246, "frac_reward_zero_std": 0.609375, "grad_norm": 0.07380837947130203, "kl": 0.21451147296465933, "learning_rate": 3.680095991753577e-07, "loss": 0.001072798972018063, "num_tokens": 113106539.0, "reward": 2.262988328933716, "reward_std": 0.4999728202819824, "rewards/code_complexity_reward/mean": 0.9061523675918579, "rewards/code_complexity_reward/std": 0.13762110471725464, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 740, "step_time": 44.0015436289832 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 260.0, "completions/max_terminated_length": 260.0, "completions/mean_length": 99.86328125, "completions/mean_terminated_length": 99.86328125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2436896124854684, "epoch": 0.8449258836944128, "frac_reward_zero_std": 0.6953125, "grad_norm": 0.05165558680891991, "kl": 0.20987687772139907, "learning_rate": 3.6282791505108327e-07, "loss": 0.0010493537411093712, "num_tokens": 113226245.0, "reward": 2.2724609375, "reward_std": 0.4386906921863556, "rewards/code_complexity_reward/mean": 0.9278320074081421, "rewards/code_complexity_reward/std": 0.05623360723257065, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.033087924122810364, "step": 741, "step_time": 41.65017924364656 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 103.416015625, "completions/mean_terminated_length": 103.416015625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24076948408037424, "epoch": 0.8460661345496009, "frac_reward_zero_std": 0.609375, "grad_norm": 0.049424413591623306, "kl": 0.19844770641066134, "learning_rate": 3.576801141454439e-07, "loss": 0.000992216751910746, "num_tokens": 113349122.0, "reward": 2.2940430641174316, "reward_std": 0.4973793029785156, "rewards/code_complexity_reward/mean": 0.9137694835662842, "rewards/code_complexity_reward/std": 0.11650004237890244, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 742, "step_time": 39.28315735794604 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 101.830078125, "completions/mean_terminated_length": 101.830078125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23714406974613667, "epoch": 0.8472063854047891, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.046750057488679886, "kl": 0.20218539983034134, "learning_rate": 3.5256627807286086e-07, "loss": 0.0010109159629791975, "num_tokens": 113472299.0, "reward": 2.3099610805511475, "reward_std": 0.47791582345962524, "rewards/code_complexity_reward/mean": 0.92529296875, "rewards/code_complexity_reward/std": 0.09000930190086365, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 743, "step_time": 48.206519547849894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 431.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 98.68359375, "completions/mean_terminated_length": 98.68359375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2393394608516246, "epoch": 0.8483466362599772, "frac_reward_zero_std": 0.53125, "grad_norm": 0.07370144873857498, "kl": 0.2097292224643752, "learning_rate": 3.474864879092693e-07, "loss": 0.001048503676429391, "num_tokens": 113590485.0, "reward": 2.3392577171325684, "reward_std": 0.5334338545799255, "rewards/code_complexity_reward/mean": 0.9111328125, "rewards/code_complexity_reward/std": 0.13397099077701569, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 744, "step_time": 43.13458985462785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 266.0, "completions/max_terminated_length": 266.0, "completions/mean_length": 100.359375, "completions/mean_terminated_length": 100.359375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24465679237619042, "epoch": 0.8494868871151653, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.0801507756114006, "kl": 0.3154769414104521, "learning_rate": 3.424408241908336e-07, "loss": 0.0015771833714097738, "num_tokens": 113711461.0, "reward": 2.250244140625, "reward_std": 0.4793037474155426, "rewards/code_complexity_reward/mean": 0.9190429449081421, "rewards/code_complexity_reward/std": 0.12680655717849731, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 745, "step_time": 33.20736639108509 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 389.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 101.947265625, "completions/mean_terminated_length": 101.947265625, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.23472591768950224, "epoch": 0.8506271379703535, "frac_reward_zero_std": 0.5625, "grad_norm": 0.059989091008901596, "kl": 0.20830512954853475, "learning_rate": 3.374293669126669e-07, "loss": 0.0010413227137178183, "num_tokens": 113831426.0, "reward": 2.249316692352295, "reward_std": 0.48572030663490295, "rewards/code_complexity_reward/mean": 0.9120116829872131, "rewards/code_complexity_reward/std": 0.1278960257768631, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 746, "step_time": 41.326008228585124 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 101.595703125, "completions/mean_terminated_length": 101.595703125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23316759010776877, "epoch": 0.8517673888255416, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.05281998962163925, "kl": 0.2129888627678156, "learning_rate": 3.324521955275697e-07, "loss": 0.0010649259202182293, "num_tokens": 113951763.0, "reward": 2.3037109375, "reward_std": 0.4733152389526367, "rewards/code_complexity_reward/mean": 0.9175781011581421, "rewards/code_complexity_reward/std": 0.09169843792915344, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 747, "step_time": 42.99918904993683 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 400.0, "completions/max_terminated_length": 400.0, "completions/mean_length": 102.1484375, "completions/mean_terminated_length": 102.1484375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23316948069259524, "epoch": 0.8529076396807298, "frac_reward_zero_std": 0.53125, "grad_norm": 0.055326562374830246, "kl": 0.2036168398335576, "learning_rate": 3.2750938894476223e-07, "loss": 0.001018046517856419, "num_tokens": 114072891.0, "reward": 2.295215129852295, "reward_std": 0.472493439912796, "rewards/code_complexity_reward/mean": 0.9217773675918579, "rewards/code_complexity_reward/std": 0.09022349119186401, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 748, "step_time": 70.93384928256273 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 273.0, "completions/max_terminated_length": 273.0, "completions/mean_length": 102.169921875, "completions/mean_terminated_length": 102.169921875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2489883026573807, "epoch": 0.8540478905359179, "frac_reward_zero_std": 0.6640625, "grad_norm": 0.05180979147553444, "kl": 0.21239809156395495, "learning_rate": 3.2260102552863994e-07, "loss": 0.0010619076201692224, "num_tokens": 114194990.0, "reward": 2.3136720657348633, "reward_std": 0.4796021282672882, "rewards/code_complexity_reward/mean": 0.928515613079071, "rewards/code_complexity_reward/std": 0.08734697848558426, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 749, "step_time": 43.653443502262235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 106.30859375, "completions/mean_terminated_length": 106.30859375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24320506514050066, "epoch": 0.855188141391106, "frac_reward_zero_std": 0.65625, "grad_norm": 0.04282500967383385, "kl": 0.20675740763545036, "learning_rate": 3.177271830975276e-07, "loss": 0.0010335331317037344, "num_tokens": 114316724.0, "reward": 2.3078126907348633, "reward_std": 0.47118279337882996, "rewards/code_complexity_reward/mean": 0.9255858659744263, "rewards/code_complexity_reward/std": 0.08126717060804367, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 750, "step_time": 47.340795022435486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 102.23046875, "completions/mean_terminated_length": 102.23046875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2507036773022264, "epoch": 0.8563283922462942, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.05745412036776543, "kl": 0.2028441766742617, "learning_rate": 3.1288793892244427e-07, "loss": 0.0010139462538063526, "num_tokens": 114438306.0, "reward": 2.285937786102295, "reward_std": 0.4553632140159607, "rewards/code_complexity_reward/mean": 0.9281250238418579, "rewards/code_complexity_reward/std": 0.07552814483642578, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 751, "step_time": 37.15405931789428 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 341.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 103.326171875, "completions/mean_terminated_length": 103.326171875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24513232964091003, "epoch": 0.8574686431014823, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05901392549276352, "kl": 0.23172475886531174, "learning_rate": 3.080833697258842e-07, "loss": 0.0011588532943278551, "num_tokens": 114560625.0, "reward": 2.2694826126098633, "reward_std": 0.48202279210090637, "rewards/code_complexity_reward/mean": 0.9216796159744263, "rewards/code_complexity_reward/std": 0.1225891262292862, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 752, "step_time": 37.56013802345842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 319.0, "completions/max_terminated_length": 319.0, "completions/mean_length": 105.04296875, "completions/mean_terminated_length": 105.04296875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24918866553343832, "epoch": 0.8586088939566705, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.05586810037493706, "kl": 0.2110057296231389, "learning_rate": 3.0331355168059214e-07, "loss": 0.0010550360893830657, "num_tokens": 114683615.0, "reward": 2.2758302688598633, "reward_std": 0.5042850375175476, "rewards/code_complexity_reward/mean": 0.910937488079071, "rewards/code_complexity_reward/std": 0.14085806906223297, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 753, "step_time": 43.541185586713254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 102.109375, "completions/mean_terminated_length": 102.109375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24618348595686257, "epoch": 0.8597491448118586, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05379674583673477, "kl": 0.20177338714711368, "learning_rate": 2.985785604083649e-07, "loss": 0.0010087359696626663, "num_tokens": 114805755.0, "reward": 2.242676019668579, "reward_std": 0.48602795600891113, "rewards/code_complexity_reward/mean": 0.9122070074081421, "rewards/code_complexity_reward/std": 0.1376785784959793, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 754, "step_time": 40.794543566182256 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 321.0, "completions/max_terminated_length": 321.0, "completions/mean_length": 102.810546875, "completions/mean_terminated_length": 102.810546875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2466066989582032, "epoch": 0.8608893956670467, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.049047186970710754, "kl": 0.19667268684133887, "learning_rate": 2.9387847097884254e-07, "loss": 0.000983318081125617, "num_tokens": 114926058.0, "reward": 2.2774415016174316, "reward_std": 0.4596049189567566, "rewards/code_complexity_reward/mean": 0.925488293170929, "rewards/code_complexity_reward/std": 0.09087396413087845, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 755, "step_time": 42.01257076486945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 287.0, "completions/max_terminated_length": 287.0, "completions/mean_length": 101.041015625, "completions/mean_terminated_length": 101.041015625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24255974381230772, "epoch": 0.8620296465222349, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.05274321138858795, "kl": 0.2045296966098249, "learning_rate": 2.8921335790832757e-07, "loss": 0.0010226145386695862, "num_tokens": 115047035.0, "reward": 2.279492139816284, "reward_std": 0.4917412996292114, "rewards/code_complexity_reward/mean": 0.915820300579071, "rewards/code_complexity_reward/std": 0.1205974593758583, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 756, "step_time": 33.55184879805893 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 315.0, "completions/max_terminated_length": 315.0, "completions/mean_length": 98.259765625, "completions/mean_terminated_length": 98.259765625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23385960329324007, "epoch": 0.863169897377423, "frac_reward_zero_std": 0.546875, "grad_norm": 0.06056765094399452, "kl": 0.2178947392385453, "learning_rate": 2.845832951585969e-07, "loss": 0.0010896958410739899, "num_tokens": 115165848.0, "reward": 2.3184571266174316, "reward_std": 0.5022709369659424, "rewards/code_complexity_reward/mean": 0.9196288585662842, "rewards/code_complexity_reward/std": 0.10872170329093933, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 757, "step_time": 42.81713387928903 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 99.18359375, "completions/mean_terminated_length": 99.18359375, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.24495337810367346, "epoch": 0.8643101482326112, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.055173903703689575, "kl": 0.20322436257265508, "learning_rate": 2.799883561357314e-07, "loss": 0.001015923568047583, "num_tokens": 115286390.0, "reward": 2.2979979515075684, "reward_std": 0.4985125958919525, "rewards/code_complexity_reward/mean": 0.920605480670929, "rewards/code_complexity_reward/std": 0.11394982784986496, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 758, "step_time": 38.10828482825309 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 330.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 101.984375, "completions/mean_terminated_length": 101.984375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2416523394640535, "epoch": 0.8654503990877993, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05988035351037979, "kl": 0.20332020428031683, "learning_rate": 2.7542861368895444e-07, "loss": 0.0010164221748709679, "num_tokens": 115406670.0, "reward": 2.3043947219848633, "reward_std": 0.47357285022735596, "rewards/code_complexity_reward/mean": 0.925097644329071, "rewards/code_complexity_reward/std": 0.09087523072957993, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 759, "step_time": 37.415085134096444 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 102.47265625, "completions/mean_terminated_length": 102.47265625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24230832769535482, "epoch": 0.8665906499429875, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.055318184196949005, "kl": 0.2009070268832147, "learning_rate": 2.709041401094717e-07, "loss": 0.0010045133531093597, "num_tokens": 115530160.0, "reward": 2.260791063308716, "reward_std": 0.4920075833797455, "rewards/code_complexity_reward/mean": 0.9198241829872131, "rewards/code_complexity_reward/std": 0.12871702015399933, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 760, "step_time": 45.3333844570443 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 262.0, "completions/max_terminated_length": 262.0, "completions/mean_length": 103.271484375, "completions/mean_terminated_length": 103.271484375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24883301742374897, "epoch": 0.8677309007981756, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.05072956532239914, "kl": 0.2053956207819283, "learning_rate": 2.664150071293314e-07, "loss": 0.0010269053746014833, "num_tokens": 115649667.0, "reward": 2.2857422828674316, "reward_std": 0.4719032049179077, "rewards/code_complexity_reward/mean": 0.9254882335662842, "rewards/code_complexity_reward/std": 0.09671591222286224, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 761, "step_time": 33.0412312168628 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 105.0, "completions/mean_terminated_length": 105.0, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.24399837898090482, "epoch": 0.8688711516533637, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.05061091482639313, "kl": 0.21421594195999205, "learning_rate": 2.619612859202808e-07, "loss": 0.0010707555338740349, "num_tokens": 115770879.0, "reward": 2.3111817836761475, "reward_std": 0.48234304785728455, "rewards/code_complexity_reward/mean": 0.92236328125, "rewards/code_complexity_reward/std": 0.09180128574371338, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 762, "step_time": 35.22180560324341 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 407.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 104.9765625, "completions/mean_terminated_length": 104.9765625, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.24383034952916205, "epoch": 0.8700114025085519, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.0585455484688282, "kl": 0.1999966655857861, "learning_rate": 2.575430470926421e-07, "loss": 0.0009999093599617481, "num_tokens": 115892039.0, "reward": 2.2655274868011475, "reward_std": 0.47938740253448486, "rewards/code_complexity_reward/mean": 0.91650390625, "rewards/code_complexity_reward/std": 0.11656399816274643, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 763, "step_time": 41.51932050008327 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 238.0, "completions/max_terminated_length": 238.0, "completions/mean_length": 100.525390625, "completions/mean_terminated_length": 100.525390625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.25318253808654845, "epoch": 0.87115165336374, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.05070706084370613, "kl": 0.2112223682925105, "learning_rate": 2.531603606941929e-07, "loss": 0.0010560562368482351, "num_tokens": 116012084.0, "reward": 2.2674806118011475, "reward_std": 0.4676544666290283, "rewards/code_complexity_reward/mean": 0.92333984375, "rewards/code_complexity_reward/std": 0.10476503521203995, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 764, "step_time": 39.094721800647676 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 419.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 104.71875, "completions/mean_terminated_length": 104.71875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2450142395682633, "epoch": 0.8722919042189282, "frac_reward_zero_std": 0.6640625, "grad_norm": 0.04495328664779663, "kl": 0.2074656041804701, "learning_rate": 2.4881329620905144e-07, "loss": 0.0010374176781624556, "num_tokens": 116134740.0, "reward": 2.2437500953674316, "reward_std": 0.4451192021369934, "rewards/code_complexity_reward/mean": 0.9210937023162842, "rewards/code_complexity_reward/std": 0.09928499162197113, "rewards/code_execution_reward/mean": 0.2265625, "rewards/code_execution_reward/std": 0.4190165400505066, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 765, "step_time": 41.49885568302125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 106.73046875, "completions/mean_terminated_length": 106.73046875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24341208674013615, "epoch": 0.8734321550741163, "frac_reward_zero_std": 0.59375, "grad_norm": 0.048358432948589325, "kl": 0.20584576204419136, "learning_rate": 2.4450192255658115e-07, "loss": 0.0010291950311511755, "num_tokens": 116259290.0, "reward": 2.2445311546325684, "reward_std": 0.463236927986145, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.10896053910255432, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 766, "step_time": 57.818505223840475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 279.0, "completions/max_terminated_length": 279.0, "completions/mean_length": 103.58984375, "completions/mean_terminated_length": 103.58984375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2495675932150334, "epoch": 0.8745724059293044, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.054982610046863556, "kl": 0.1993230115622282, "learning_rate": 2.402263080902917e-07, "loss": 0.0009964328492060304, "num_tokens": 116380352.0, "reward": 2.2796387672424316, "reward_std": 0.46874603629112244, "rewards/code_complexity_reward/mean": 0.927441418170929, "rewards/code_complexity_reward/std": 0.0971909612417221, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 767, "step_time": 34.116027442738414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 105.228515625, "completions/mean_terminated_length": 105.228515625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24812098545953631, "epoch": 0.8757126567844926, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05340832099318504, "kl": 0.2038407945074141, "learning_rate": 2.359865205967618e-07, "loss": 0.0010192570043727756, "num_tokens": 116502201.0, "reward": 2.2618165016174316, "reward_std": 0.503035843372345, "rewards/code_complexity_reward/mean": 0.9108397960662842, "rewards/code_complexity_reward/std": 0.13593439757823944, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 768, "step_time": 46.86800542566925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 274.0, "completions/max_terminated_length": 274.0, "completions/mean_length": 103.263671875, "completions/mean_terminated_length": 103.263671875, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.2430135477334261, "epoch": 0.8768529076396807, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.05781649053096771, "kl": 0.21327010542154312, "learning_rate": 2.317826272945578e-07, "loss": 0.0010663452558219433, "num_tokens": 116624564.0, "reward": 2.331787109375, "reward_std": 0.5206232070922852, "rewards/code_complexity_reward/mean": 0.9175781011581421, "rewards/code_complexity_reward/std": 0.12200859934091568, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 769, "step_time": 42.558622932992876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 457.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 106.322265625, "completions/mean_terminated_length": 106.322265625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23847284144721925, "epoch": 0.8779931584948689, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05510120466351509, "kl": 0.20642531127668917, "learning_rate": 2.2761469483317257e-07, "loss": 0.0010323310270905495, "num_tokens": 116748981.0, "reward": 2.2903809547424316, "reward_std": 0.49097681045532227, "rewards/code_complexity_reward/mean": 0.9152343273162842, "rewards/code_complexity_reward/std": 0.11477171629667282, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 770, "step_time": 45.15606455691159 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 101.837890625, "completions/mean_terminated_length": 101.837890625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24379343376494944, "epoch": 0.879133409350057, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.056797124445438385, "kl": 0.19907205761410296, "learning_rate": 2.2348278929196886e-07, "loss": 0.0009953127009794116, "num_tokens": 116868146.0, "reward": 2.2529296875, "reward_std": 0.49532178044319153, "rewards/code_complexity_reward/mean": 0.9117187261581421, "rewards/code_complexity_reward/std": 0.1392226666212082, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 771, "step_time": 58.63336842413992 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 441.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 104.322265625, "completions/mean_terminated_length": 104.322265625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2544490231666714, "epoch": 0.8802736602052451, "frac_reward_zero_std": 0.625, "grad_norm": 0.04764336347579956, "kl": 0.20285178627818823, "learning_rate": 2.1938697617912759e-07, "loss": 0.0010142853716388345, "num_tokens": 116990607.0, "reward": 2.307910442352295, "reward_std": 0.47347545623779297, "rewards/code_complexity_reward/mean": 0.9276366829872131, "rewards/code_complexity_reward/std": 0.08104552328586578, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 772, "step_time": 47.197410385124385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 102.201171875, "completions/mean_terminated_length": 102.201171875, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.2543066954240203, "epoch": 0.8814139110604333, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.05708806589245796, "kl": 0.20567588391713798, "learning_rate": 2.1532732043061527e-07, "loss": 0.0010282526491209865, "num_tokens": 117112634.0, "reward": 2.2437989711761475, "reward_std": 0.46472102403640747, "rewards/code_complexity_reward/mean": 0.9186522960662842, "rewards/code_complexity_reward/std": 0.11394345015287399, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 773, "step_time": 42.77503878157586 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 421.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 106.326171875, "completions/mean_terminated_length": 106.326171875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24138084589503706, "epoch": 0.8825541619156214, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.05858087167143822, "kl": 0.21810262417420745, "learning_rate": 2.1130388640914794e-07, "loss": 0.001090589095838368, "num_tokens": 117235857.0, "reward": 2.24853515625, "reward_std": 0.46400031447410583, "rewards/code_complexity_reward/mean": 0.9161132574081421, "rewards/code_complexity_reward/std": 0.11543811857700348, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 774, "step_time": 49.199674397706985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 280.0, "completions/max_terminated_length": 280.0, "completions/mean_length": 97.98046875, "completions/mean_terminated_length": 97.98046875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2325193677097559, "epoch": 0.8836944127708096, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.052042294293642044, "kl": 0.19390096981078386, "learning_rate": 2.0731673790317596e-07, "loss": 0.0009692884050309658, "num_tokens": 117353031.0, "reward": 2.366259813308716, "reward_std": 0.4960958659648895, "rewards/code_complexity_reward/mean": 0.9276366829872131, "rewards/code_complexity_reward/std": 0.07915206253528595, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 775, "step_time": 41.98956793732941 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 406.0, "completions/max_terminated_length": 406.0, "completions/mean_length": 102.796875, "completions/mean_terminated_length": 102.796875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2393335378728807, "epoch": 0.8848346636259977, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05462901294231415, "kl": 0.18649066390935332, "learning_rate": 2.0336593812587096e-07, "loss": 0.0009322575060650706, "num_tokens": 117473543.0, "reward": 2.3716797828674316, "reward_std": 0.5269625782966614, "rewards/code_complexity_reward/mean": 0.9171874523162842, "rewards/code_complexity_reward/std": 0.11560061573982239, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 776, "step_time": 50.107173272408545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 388.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 106.8046875, "completions/mean_terminated_length": 106.8046875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24283042713068426, "epoch": 0.8859749144811858, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05454649403691292, "kl": 0.2128400495275855, "learning_rate": 1.9945154971412168e-07, "loss": 0.0010641596745699644, "num_tokens": 117596423.0, "reward": 2.3074707984924316, "reward_std": 0.48041293025016785, "rewards/code_complexity_reward/mean": 0.9225585460662842, "rewards/code_complexity_reward/std": 0.09254974871873856, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 777, "step_time": 41.0915355309844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 104.494140625, "completions/mean_terminated_length": 104.494140625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23806692706421018, "epoch": 0.887115165336374, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.051752325147390366, "kl": 0.20155888888984919, "learning_rate": 1.9557363472754582e-07, "loss": 0.0010077442275360227, "num_tokens": 117718432.0, "reward": 2.306933879852295, "reward_std": 0.5018028020858765, "rewards/code_complexity_reward/mean": 0.9168945550918579, "rewards/code_complexity_reward/std": 0.11796846240758896, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 778, "step_time": 38.96075674891472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 100.45703125, "completions/mean_terminated_length": 100.45703125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24521340755745769, "epoch": 0.8882554161915621, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.06817879527807236, "kl": 0.24501844751648605, "learning_rate": 1.9173225464749867e-07, "loss": 0.0012246984988451004, "num_tokens": 117837442.0, "reward": 2.302734375, "reward_std": 0.47907519340515137, "rewards/code_complexity_reward/mean": 0.9250977039337158, "rewards/code_complexity_reward/std": 0.09737247973680496, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 779, "step_time": 37.64857401326299 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 335.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 103.208984375, "completions/mean_terminated_length": 103.208984375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23365635401569307, "epoch": 0.8893956670467503, "frac_reward_zero_std": 0.640625, "grad_norm": 0.049248505383729935, "kl": 0.20848369062878191, "learning_rate": 1.879274703761072e-07, "loss": 0.0010424023494124413, "num_tokens": 117959613.0, "reward": 2.2603516578674316, "reward_std": 0.47390857338905334, "rewards/code_complexity_reward/mean": 0.918164074420929, "rewards/code_complexity_reward/std": 0.11104444414377213, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 780, "step_time": 46.23451016936451 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 321.0, "completions/max_terminated_length": 321.0, "completions/mean_length": 100.41015625, "completions/mean_terminated_length": 100.41015625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24502743617631495, "epoch": 0.8905359179019384, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.050728939473629, "kl": 0.20651052752509713, "learning_rate": 1.8415934223529665e-07, "loss": 0.0010324518661946058, "num_tokens": 118078919.0, "reward": 2.2350587844848633, "reward_std": 0.478752464056015, "rewards/code_complexity_reward/mean": 0.914355456829071, "rewards/code_complexity_reward/std": 0.1320246309041977, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 781, "step_time": 44.11142007354647 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 101.099609375, "completions/mean_terminated_length": 101.099609375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24161975481547415, "epoch": 0.8916761687571265, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05984211340546608, "kl": 0.20458688796497881, "learning_rate": 1.8042792996583902e-07, "loss": 0.0010228054597973824, "num_tokens": 118199282.0, "reward": 2.241455078125, "reward_std": 0.49373915791511536, "rewards/code_complexity_reward/mean": 0.9083007574081421, "rewards/code_complexity_reward/std": 0.1435653269290924, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 782, "step_time": 46.09325705841184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 296.0, "completions/max_terminated_length": 296.0, "completions/mean_length": 102.814453125, "completions/mean_terminated_length": 102.814453125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.23942517675459385, "epoch": 0.8928164196123147, "frac_reward_zero_std": 0.578125, "grad_norm": 0.07185967266559601, "kl": 0.21094024810008705, "learning_rate": 1.76733292726404e-07, "loss": 0.0010544888209551573, "num_tokens": 118320395.0, "reward": 2.2933597564697266, "reward_std": 0.4860227108001709, "rewards/code_complexity_reward/mean": 0.92236328125, "rewards/code_complexity_reward/std": 0.10723751038312912, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 783, "step_time": 34.891953041777015 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 366.0, "completions/max_terminated_length": 366.0, "completions/mean_length": 102.0859375, "completions/mean_terminated_length": 102.0859375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24357536341995, "epoch": 0.8939566704675028, "frac_reward_zero_std": 0.640625, "grad_norm": 0.0462709479033947, "kl": 0.20909320726059377, "learning_rate": 1.7307548909262118e-07, "loss": 0.0010455182055011392, "num_tokens": 118441623.0, "reward": 2.2538576126098633, "reward_std": 0.4460700750350952, "rewards/code_complexity_reward/mean": 0.9197264909744263, "rewards/code_complexity_reward/std": 0.07624711841344833, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 784, "step_time": 40.151994206011295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 105.27734375, "completions/mean_terminated_length": 104.48140716552734, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24127778434194624, "epoch": 0.895096921322691, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05282621085643768, "kl": 0.20827760337851942, "learning_rate": 1.694545770561537e-07, "loss": 0.0010412507690489292, "num_tokens": 118564693.0, "reward": 2.2767577171325684, "reward_std": 0.5054394602775574, "rewards/code_complexity_reward/mean": 0.91259765625, "rewards/code_complexity_reward/std": 0.1301906257867813, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 785, "step_time": 60.3175740018487 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 101.72265625, "completions/mean_terminated_length": 100.91976165771484, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24166113859973848, "epoch": 0.8962371721778791, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.05135417729616165, "kl": 0.2207891179714352, "learning_rate": 1.658706140237737e-07, "loss": 0.0011037427466362715, "num_tokens": 118684543.0, "reward": 2.3314943313598633, "reward_std": 0.5044814944267273, "rewards/code_complexity_reward/mean": 0.921191394329071, "rewards/code_complexity_reward/std": 0.10941557586193085, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 786, "step_time": 49.52517205942422 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 319.0, "completions/max_terminated_length": 319.0, "completions/mean_length": 99.60546875, "completions/mean_terminated_length": 99.60546875, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.24266806757077575, "epoch": 0.8973774230330672, "frac_reward_zero_std": 0.578125, "grad_norm": 0.06238792836666107, "kl": 0.19647406111471355, "learning_rate": 1.623236568164574e-07, "loss": 0.0009820754639804363, "num_tokens": 118802917.0, "reward": 2.2892580032348633, "reward_std": 0.5042401552200317, "rewards/code_complexity_reward/mean": 0.9167969226837158, "rewards/code_complexity_reward/std": 0.12722037732601166, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 787, "step_time": 35.74121873360127 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 404.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 106.3984375, "completions/mean_terminated_length": 106.3984375, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.24473963002674282, "epoch": 0.8985176738882554, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.05587523803114891, "kl": 0.2002555716317147, "learning_rate": 1.5881376166848151e-07, "loss": 0.0010014282306656241, "num_tokens": 118925465.0, "reward": 2.2694337368011475, "reward_std": 0.4933989942073822, "rewards/code_complexity_reward/mean": 0.91748046875, "rewards/code_complexity_reward/std": 0.12442470341920853, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 788, "step_time": 42.02771863061935 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 107.546875, "completions/mean_terminated_length": 107.546875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24817639216780663, "epoch": 0.8996579247434435, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.059891920536756516, "kl": 0.20290119131095707, "learning_rate": 1.5534098422653243e-07, "loss": 0.0010144348489120603, "num_tokens": 119050121.0, "reward": 2.2401368618011475, "reward_std": 0.46271488070487976, "rewards/code_complexity_reward/mean": 0.92236328125, "rewards/code_complexity_reward/std": 0.11387534439563751, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 789, "step_time": 44.98208007682115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 103.32421875, "completions/mean_terminated_length": 103.32421875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24720612866804004, "epoch": 0.9007981755986317, "frac_reward_zero_std": 0.5, "grad_norm": 0.052877940237522125, "kl": 0.19526389660313725, "learning_rate": 1.5190537954882373e-07, "loss": 0.0009763809503056109, "num_tokens": 119171927.0, "reward": 2.304980754852295, "reward_std": 0.5134298205375671, "rewards/code_complexity_reward/mean": 0.9129881858825684, "rewards/code_complexity_reward/std": 0.12814442813396454, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 790, "step_time": 45.896696587093174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 243.0, "completions/max_terminated_length": 243.0, "completions/mean_length": 104.466796875, "completions/mean_terminated_length": 104.466796875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2604955052956939, "epoch": 0.9019384264538198, "frac_reward_zero_std": 0.625, "grad_norm": 0.05178629234433174, "kl": 0.19957288540899754, "learning_rate": 1.485070021042237e-07, "loss": 0.0009977947920560837, "num_tokens": 119294342.0, "reward": 2.2447755336761475, "reward_std": 0.4859859347343445, "rewards/code_complexity_reward/mean": 0.91064453125, "rewards/code_complexity_reward/std": 0.1293857842683792, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 791, "step_time": 39.510312176309526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 101.91015625, "completions/mean_terminated_length": 101.91015625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2415813086554408, "epoch": 0.9030786773090079, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.052129071205854416, "kl": 0.20883699925616384, "learning_rate": 1.4514590577139136e-07, "loss": 0.0010440812911838293, "num_tokens": 119416784.0, "reward": 2.34521484375, "reward_std": 0.48193544149398804, "rewards/code_complexity_reward/mean": 0.9307616949081421, "rewards/code_complexity_reward/std": 0.07019685208797455, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 792, "step_time": 37.47027028631419 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 307.0, "completions/mean_length": 101.080078125, "completions/mean_terminated_length": 100.27593231201172, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2512553979177028, "epoch": 0.9042189281641961, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.055593352764844894, "kl": 0.21847413387149572, "learning_rate": 1.4182214383792192e-07, "loss": 0.0010922932997345924, "num_tokens": 119538117.0, "reward": 2.2694337368011475, "reward_std": 0.4733409285545349, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.11266906559467316, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 793, "step_time": 48.20929859019816 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 505.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 104.953125, "completions/mean_terminated_length": 104.953125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2467286209575832, "epoch": 0.9053591790193842, "frac_reward_zero_std": 0.6640625, "grad_norm": 0.04459979012608528, "kl": 0.19367323839105666, "learning_rate": 1.3853576899950343e-07, "loss": 0.0009681254159659147, "num_tokens": 119660093.0, "reward": 2.199267864227295, "reward_std": 0.4191720187664032, "rewards/code_complexity_reward/mean": 0.9178711175918579, "rewards/code_complexity_reward/std": 0.09741251170635223, "rewards/code_execution_reward/mean": 0.185546875, "rewards/code_execution_reward/std": 0.38912075757980347, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 794, "step_time": 55.05862649343908 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 273.0, "completions/max_terminated_length": 273.0, "completions/mean_length": 102.796875, "completions/mean_terminated_length": 102.796875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23573476797901094, "epoch": 0.9064994298745724, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.06422534584999084, "kl": 0.2108434047549963, "learning_rate": 1.3528683335907928e-07, "loss": 0.001053863437846303, "num_tokens": 119783013.0, "reward": 2.2635743618011475, "reward_std": 0.5029816627502441, "rewards/code_complexity_reward/mean": 0.91064453125, "rewards/code_complexity_reward/std": 0.1394670307636261, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 795, "step_time": 47.38901784364134 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 101.162109375, "completions/mean_terminated_length": 101.162109375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24093269067816436, "epoch": 0.9076396807297605, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.05423051491379738, "kl": 0.20082319155335426, "learning_rate": 1.3207538842602396e-07, "loss": 0.0010039161425083876, "num_tokens": 119902588.0, "reward": 2.3519532680511475, "reward_std": 0.5109429359436035, "rewards/code_complexity_reward/mean": 0.9225585460662842, "rewards/code_complexity_reward/std": 0.10655564069747925, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 796, "step_time": 45.00699697993696 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 422.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 102.6015625, "completions/mean_terminated_length": 102.6015625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24480484472587705, "epoch": 0.9087799315849487, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.049472931772470474, "kl": 0.2060973341576755, "learning_rate": 1.289014851153253e-07, "loss": 0.0010304739698767662, "num_tokens": 120022664.0, "reward": 2.3397462368011475, "reward_std": 0.4963342547416687, "rewards/code_complexity_reward/mean": 0.92529296875, "rewards/code_complexity_reward/std": 0.09441296011209488, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 797, "step_time": 51.601228991523385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 242.0, "completions/max_terminated_length": 242.0, "completions/mean_length": 98.208984375, "completions/mean_terminated_length": 98.208984375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2411960638128221, "epoch": 0.9099201824401368, "frac_reward_zero_std": 0.578125, "grad_norm": 0.04846816509962082, "kl": 0.20721523906104267, "learning_rate": 1.2576517374677745e-07, "loss": 0.0010360705200582743, "num_tokens": 120139799.0, "reward": 2.3133790493011475, "reward_std": 0.4643048346042633, "rewards/code_complexity_reward/mean": 0.93408203125, "rewards/code_complexity_reward/std": 0.0663231834769249, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 798, "step_time": 30.87582625914365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 105.583984375, "completions/mean_terminated_length": 105.583984375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23814302100799978, "epoch": 0.9110604332953249, "frac_reward_zero_std": 0.578125, "grad_norm": 0.04715698957443237, "kl": 0.19690573099069297, "learning_rate": 1.2266650404418378e-07, "loss": 0.0009846820030361414, "num_tokens": 120262770.0, "reward": 2.321094036102295, "reward_std": 0.5179407000541687, "rewards/code_complexity_reward/mean": 0.9125000238418579, "rewards/code_complexity_reward/std": 0.12425905466079712, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 799, "step_time": 44.21871352382004 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 295.0, "completions/max_terminated_length": 295.0, "completions/mean_length": 100.505859375, "completions/mean_terminated_length": 100.505859375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24759164708666503, "epoch": 0.9122006841505131, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.05688778683543205, "kl": 0.20100104552693665, "learning_rate": 1.1960552513456764e-07, "loss": 0.0010049648117274046, "num_tokens": 120382709.0, "reward": 2.3091797828674316, "reward_std": 0.5030375123023987, "rewards/code_complexity_reward/mean": 0.919140636920929, "rewards/code_complexity_reward/std": 0.11461175978183746, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 800, "step_time": 45.57687596138567 }, { "epoch": 0.9122006841505131, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 145.98, "eval_completions/max_terminated_length": 145.98, "eval_completions/mean_length": 101.8325, "eval_completions/mean_terminated_length": 101.8325, "eval_completions/min_length": 71.34, "eval_completions/min_terminated_length": 71.34, "eval_entropy": 0.23909246921539307, "eval_frac_reward_zero_std": 0.5, "eval_kl": 0.19546649783849715, "eval_loss": 0.0009763582493178546, "eval_num_tokens": 120382709.0, "eval_reward": 2.298750112056732, "eval_reward_std": 0.3572192522138357, "eval_rewards/code_complexity_reward/mean": 0.9199999749660492, "eval_rewards/code_complexity_reward/std": 0.05470168549567461, "eval_rewards/code_execution_reward/mean": 0.285, "eval_rewards/code_execution_reward/std": 0.31344565451145173, "eval_rewards/code_syntax_reward/mean": 0.49375, "eval_rewards/code_syntax_reward/std": 0.01767766922712326, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.5, "eval_rewards/xmlcount_reward_func/std": 0.0, "eval_runtime": 338.2847, "eval_samples_per_second": 0.296, "eval_steps_per_second": 0.038, "step": 800 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 285.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 101.984375, "completions/mean_terminated_length": 101.984375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24693063274025917, "epoch": 0.9133409350057012, "frac_reward_zero_std": 0.5625, "grad_norm": 0.0544855110347271, "kl": 0.20878716022707522, "learning_rate": 1.1658228554739359e-07, "loss": 0.001043748576194048, "num_tokens": 120504297.0, "reward": 2.3462891578674316, "reward_std": 0.48078882694244385, "rewards/code_complexity_reward/mean": 0.9225585460662842, "rewards/code_complexity_reward/std": 0.07066842913627625, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 801, "step_time": 53.81441994383931 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 102.7421875, "completions/mean_terminated_length": 102.7421875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24430433474481106, "epoch": 0.9144811858608894, "frac_reward_zero_std": 0.6875, "grad_norm": 0.04427551478147507, "kl": 0.21276699472218752, "learning_rate": 1.1359683321379878e-07, "loss": 0.0010636652586981654, "num_tokens": 120625321.0, "reward": 2.2331056594848633, "reward_std": 0.4329536557197571, "rewards/code_complexity_reward/mean": 0.926074206829071, "rewards/code_complexity_reward/std": 0.08802473545074463, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 802, "step_time": 43.96862359717488 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 101.2890625, "completions/mean_terminated_length": 101.2890625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2367626887280494, "epoch": 0.9156214367160775, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.055522818118333817, "kl": 0.20271275588311255, "learning_rate": 1.106492154658323e-07, "loss": 0.0010135790798813105, "num_tokens": 120743653.0, "reward": 2.3053712844848633, "reward_std": 0.49316808581352234, "rewards/code_complexity_reward/mean": 0.922167956829071, "rewards/code_complexity_reward/std": 0.10723251849412918, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 803, "step_time": 35.8613604484126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 102.91015625, "completions/mean_terminated_length": 102.10958862304688, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24682559072971344, "epoch": 0.9167616875712656, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05476384609937668, "kl": 0.21381500991992652, "learning_rate": 1.0773947903570503e-07, "loss": 0.0010689814807847142, "num_tokens": 120865011.0, "reward": 2.283935546875, "reward_std": 0.49780043959617615, "rewards/code_complexity_reward/mean": 0.9185547232627869, "rewards/code_complexity_reward/std": 0.12246433645486832, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 804, "step_time": 55.128502265550196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 316.0, "completions/max_terminated_length": 316.0, "completions/mean_length": 100.98046875, "completions/mean_terminated_length": 100.98046875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24126586457714438, "epoch": 0.9179019384264538, "frac_reward_zero_std": 0.640625, "grad_norm": 0.05099915713071823, "kl": 0.20719301362987608, "learning_rate": 1.0486767005504911e-07, "loss": 0.0010354304686188698, "num_tokens": 120985121.0, "reward": 2.2975587844848633, "reward_std": 0.47836655378341675, "rewards/code_complexity_reward/mean": 0.9231444597244263, "rewards/code_complexity_reward/std": 0.09835473448038101, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 805, "step_time": 38.09973023086786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 404.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 104.486328125, "completions/mean_terminated_length": 104.486328125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2474762888159603, "epoch": 0.9190421892816419, "frac_reward_zero_std": 0.609375, "grad_norm": 0.0478985421359539, "kl": 0.21673831879161298, "learning_rate": 1.0203383405418515e-07, "loss": 0.0010837967274710536, "num_tokens": 121108022.0, "reward": 2.2850584983825684, "reward_std": 0.5022030472755432, "rewards/code_complexity_reward/mean": 0.92041015625, "rewards/code_complexity_reward/std": 0.12682494521141052, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 806, "step_time": 41.07062632963061 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 103.47265625, "completions/mean_terminated_length": 102.67318725585938, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24213313683867455, "epoch": 0.9201824401368301, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.054059505462646484, "kl": 0.19617363950237632, "learning_rate": 9.923801596140258e-08, "loss": 0.000980721553787589, "num_tokens": 121229288.0, "reward": 2.288623094558716, "reward_std": 0.5041497945785522, "rewards/code_complexity_reward/mean": 0.9149413704872131, "rewards/code_complexity_reward/std": 0.12793098390102386, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 807, "step_time": 58.851042496971786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 423.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 106.982421875, "completions/mean_terminated_length": 106.982421875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23319072485901415, "epoch": 0.9213226909920182, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05406786501407623, "kl": 0.19665061216801405, "learning_rate": 9.64802601022452e-08, "loss": 0.000983122969046235, "num_tokens": 121352895.0, "reward": 2.3140625953674316, "reward_std": 0.5054183006286621, "rewards/code_complexity_reward/mean": 0.9201172590255737, "rewards/code_complexity_reward/std": 0.11685534566640854, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 808, "step_time": 50.781579117290676 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 101.759765625, "completions/mean_terminated_length": 101.759765625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.241801920812577, "epoch": 0.9224629418472063, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.0556652694940567, "kl": 0.20586044481024146, "learning_rate": 9.376061019881006e-08, "loss": 0.0010290800128132105, "num_tokens": 121473640.0, "reward": 2.3428711891174316, "reward_std": 0.5246403813362122, "rewards/code_complexity_reward/mean": 0.9166991710662842, "rewards/code_complexity_reward/std": 0.12195184826850891, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 809, "step_time": 35.04772181902081 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 345.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 105.60546875, "completions/mean_terminated_length": 105.60546875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24940583831630647, "epoch": 0.9236031927023945, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.05153829604387283, "kl": 0.21395044610835612, "learning_rate": 9.107910936905301e-08, "loss": 0.0010696889366954565, "num_tokens": 121596442.0, "reward": 2.201171875, "reward_std": 0.46474650502204895, "rewards/code_complexity_reward/mean": 0.9104492664337158, "rewards/code_complexity_reward/std": 0.13810710608959198, "rewards/code_execution_reward/mean": 0.201171875, "rewards/code_execution_reward/std": 0.4012683033943176, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 810, "step_time": 37.06829050090164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 104.0, "completions/mean_terminated_length": 104.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24983719852752984, "epoch": 0.9247434435575826, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.04624596983194351, "kl": 0.20695660659112036, "learning_rate": 8.843580012610625e-08, "loss": 0.0010345734190195799, "num_tokens": 121718206.0, "reward": 2.2458009719848633, "reward_std": 0.443805992603302, "rewards/code_complexity_reward/mean": 0.923144519329071, "rewards/code_complexity_reward/std": 0.09074854105710983, "rewards/code_execution_reward/mean": 0.2265625, "rewards/code_execution_reward/std": 0.4190165400505066, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 811, "step_time": 47.12692367378622 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 255.0, "completions/max_terminated_length": 255.0, "completions/mean_length": 100.2421875, "completions/mean_terminated_length": 100.2421875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.25583585794083774, "epoch": 0.9258836944127709, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.13168072700500488, "kl": 0.30559187196195126, "learning_rate": 8.583072437760381e-08, "loss": 0.0015272139571607113, "num_tokens": 121840782.0, "reward": 2.2763671875, "reward_std": 0.49520018696784973, "rewards/code_complexity_reward/mean": 0.9195312261581421, "rewards/code_complexity_reward/std": 0.1272137463092804, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 812, "step_time": 31.7911094147712 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 321.0, "completions/mean_length": 106.109375, "completions/mean_terminated_length": 105.31507110595703, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24541813204996288, "epoch": 0.927023945267959, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.06838209927082062, "kl": 0.20154074765741825, "learning_rate": 8.3263923425016e-08, "loss": 0.0010074449237436056, "num_tokens": 121966058.0, "reward": 2.3160645961761475, "reward_std": 0.49168428778648376, "rewards/code_complexity_reward/mean": 0.91943359375, "rewards/code_complexity_reward/std": 0.09989363700151443, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 813, "step_time": 56.256505249999464 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 450.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 100.91015625, "completions/mean_terminated_length": 100.91015625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.233826226554811, "epoch": 0.928164196123147, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.05552414059638977, "kl": 0.20672540669329464, "learning_rate": 8.07354379629971e-08, "loss": 0.0010336164850741625, "num_tokens": 122085320.0, "reward": 2.3064942359924316, "reward_std": 0.4634203612804413, "rewards/code_complexity_reward/mean": 0.9293944835662842, "rewards/code_complexity_reward/std": 0.06531741470098495, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 814, "step_time": 44.74057548586279 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 105.64453125, "completions/mean_terminated_length": 105.64453125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24134204629808664, "epoch": 0.9293044469783353, "frac_reward_zero_std": 0.609375, "grad_norm": 0.053928177803754807, "kl": 0.19255767785944045, "learning_rate": 7.82453080787382e-08, "loss": 0.0009625913808122277, "num_tokens": 122205462.0, "reward": 2.275683879852295, "reward_std": 0.48637956380844116, "rewards/code_complexity_reward/mean": 0.9188476800918579, "rewards/code_complexity_reward/std": 0.11675337702035904, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 815, "step_time": 36.19075152184814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 277.0, "completions/max_terminated_length": 277.0, "completions/mean_length": 99.11328125, "completions/mean_terminated_length": 99.11328125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2497749957256019, "epoch": 0.9304446978335233, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.0489383228123188, "kl": 0.2088247323408723, "learning_rate": 7.579357325133208e-08, "loss": 0.0010441092308610678, "num_tokens": 122325560.0, "reward": 2.3134765625, "reward_std": 0.5096659064292908, "rewards/code_complexity_reward/mean": 0.9205077886581421, "rewards/code_complexity_reward/std": 0.11980632692575455, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 816, "step_time": 82.03753112256527 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 277.0, "completions/max_terminated_length": 277.0, "completions/mean_length": 102.91015625, "completions/mean_terminated_length": 102.91015625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24439573381096125, "epoch": 0.9315849486887116, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.04448368027806282, "kl": 0.2102453731931746, "learning_rate": 7.338027235114759e-08, "loss": 0.0010511998552829027, "num_tokens": 122445418.0, "reward": 2.289843797683716, "reward_std": 0.4813186824321747, "rewards/code_complexity_reward/mean": 0.9222656488418579, "rewards/code_complexity_reward/std": 0.10873018950223923, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 817, "step_time": 52.54528467357159 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 106.861328125, "completions/mean_terminated_length": 106.06848907470703, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2304356952663511, "epoch": 0.9327251995438997, "frac_reward_zero_std": 0.609375, "grad_norm": 0.0522712841629982, "kl": 0.20516197732649744, "learning_rate": 7.100544363921324e-08, "loss": 0.001025802455842495, "num_tokens": 122569815.0, "reward": 2.2804689407348633, "reward_std": 0.4727371037006378, "rewards/code_complexity_reward/mean": 0.923144519329071, "rewards/code_complexity_reward/std": 0.09506653994321823, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 818, "step_time": 67.14129808172584 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 282.0, "completions/max_terminated_length": 282.0, "completions/mean_length": 99.525390625, "completions/mean_terminated_length": 99.525390625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24609071272425354, "epoch": 0.9338654503990877, "frac_reward_zero_std": 0.59375, "grad_norm": 0.0632210448384285, "kl": 0.2067659164313227, "learning_rate": 6.866912476661075e-08, "loss": 0.0010339856380596757, "num_tokens": 122691824.0, "reward": 2.3694825172424316, "reward_std": 0.509251594543457, "rewards/code_complexity_reward/mean": 0.9291015863418579, "rewards/code_complexity_reward/std": 0.09698372334241867, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 819, "step_time": 33.354931293055415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 306.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 101.796875, "completions/mean_terminated_length": 101.796875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23372497083619237, "epoch": 0.935005701254276, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.06368544697761536, "kl": 0.20278119761496782, "learning_rate": 6.637135277387713e-08, "loss": 0.0010139276273548603, "num_tokens": 122810904.0, "reward": 2.3192381858825684, "reward_std": 0.4857620596885681, "rewards/code_complexity_reward/mean": 0.92041015625, "rewards/code_complexity_reward/std": 0.09220302104949951, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 820, "step_time": 44.12205119244754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 99.982421875, "completions/mean_terminated_length": 99.982421875, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.2347351920325309, "epoch": 0.936145952109464, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.05268912762403488, "kl": 0.20779918087646365, "learning_rate": 6.411216409041965e-08, "loss": 0.0010389751987531781, "num_tokens": 122932603.0, "reward": 2.329345703125, "reward_std": 0.49580374360084534, "rewards/code_complexity_reward/mean": 0.9241210222244263, "rewards/code_complexity_reward/std": 0.09866629540920258, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 821, "step_time": 35.580179903656244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 104.001953125, "completions/mean_terminated_length": 104.001953125, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.2409504095558077, "epoch": 0.9372862029646523, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.0561768114566803, "kl": 0.20155664952471852, "learning_rate": 6.189159453393573e-08, "loss": 0.0010076719336211681, "num_tokens": 123055256.0, "reward": 2.3023438453674316, "reward_std": 0.495384156703949, "rewards/code_complexity_reward/mean": 0.9171874523162842, "rewards/code_complexity_reward/std": 0.1123817041516304, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 822, "step_time": 34.28804542310536 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 108.052734375, "completions/mean_terminated_length": 108.052734375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2527662147767842, "epoch": 0.9384264538198404, "frac_reward_zero_std": 0.578125, "grad_norm": 0.0477706640958786, "kl": 0.19685496343299747, "learning_rate": 5.970967930984728e-08, "loss": 0.000984064070507884, "num_tokens": 123179547.0, "reward": 2.2710938453674316, "reward_std": 0.49383118748664856, "rewards/code_complexity_reward/mean": 0.9152343273162842, "rewards/code_complexity_reward/std": 0.12610507011413574, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 823, "step_time": 42.423948156647384 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 352.0, "completions/max_terminated_length": 352.0, "completions/mean_length": 104.77734375, "completions/mean_terminated_length": 104.77734375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2455215724185109, "epoch": 0.9395667046750285, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.04294752702116966, "kl": 0.20245797303505242, "learning_rate": 5.756645301074088e-08, "loss": 0.0010122403036803007, "num_tokens": 123303277.0, "reward": 2.285937786102295, "reward_std": 0.4808439016342163, "rewards/code_complexity_reward/mean": 0.9193359017372131, "rewards/code_complexity_reward/std": 0.102498859167099, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 824, "step_time": 43.98361643124372 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 366.0, "completions/max_terminated_length": 366.0, "completions/mean_length": 104.107421875, "completions/mean_terminated_length": 104.107421875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24107529502362013, "epoch": 0.9407069555302167, "frac_reward_zero_std": 0.609375, "grad_norm": 0.05344848334789276, "kl": 0.20785449515096843, "learning_rate": 5.546194961581958e-08, "loss": 0.0010390307288616896, "num_tokens": 123426032.0, "reward": 2.2608399391174316, "reward_std": 0.46215343475341797, "rewards/code_complexity_reward/mean": 0.9235351085662842, "rewards/code_complexity_reward/std": 0.09836134314537048, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 825, "step_time": 47.7901212759316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 105.46875, "completions/mean_terminated_length": 105.46875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.244854886084795, "epoch": 0.9418472063854048, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.044597089290618896, "kl": 0.21494625392369926, "learning_rate": 5.3396202490365586e-08, "loss": 0.0010745424078777432, "num_tokens": 123550148.0, "reward": 2.2833008766174316, "reward_std": 0.47943973541259766, "rewards/code_complexity_reward/mean": 0.9166991710662842, "rewards/code_complexity_reward/std": 0.10336530953645706, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 826, "step_time": 40.15753853786737 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 302.0, "completions/max_terminated_length": 302.0, "completions/mean_length": 103.40234375, "completions/mean_terminated_length": 103.40234375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24266860843636096, "epoch": 0.942987457240593, "frac_reward_zero_std": 0.5625, "grad_norm": 0.14103320240974426, "kl": 0.34241684759035707, "learning_rate": 5.136924438520902e-08, "loss": 0.0017101343255490065, "num_tokens": 123670438.0, "reward": 2.3151369094848633, "reward_std": 0.48696351051330566, "rewards/code_complexity_reward/mean": 0.916308581829071, "rewards/code_complexity_reward/std": 0.09678109735250473, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 827, "step_time": 41.41449430026114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 103.16796875, "completions/mean_terminated_length": 103.16796875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23762981337495148, "epoch": 0.9441277080957811, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.06365665048360825, "kl": 0.22545420820824802, "learning_rate": 4.9381107436211607e-08, "loss": 0.001126841176301241, "num_tokens": 123791040.0, "reward": 2.3473634719848633, "reward_std": 0.4999588131904602, "rewards/code_complexity_reward/mean": 0.9250977039337158, "rewards/code_complexity_reward/std": 0.09263478219509125, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 828, "step_time": 37.27905864920467 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 102.0078125, "completions/mean_terminated_length": 102.0078125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.2432017270475626, "epoch": 0.9452679589509693, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.06261098384857178, "kl": 0.202464759349823, "learning_rate": 4.743182316375439e-08, "loss": 0.0010121114319190383, "num_tokens": 123912572.0, "reward": 2.333984375, "reward_std": 0.49079328775405884, "rewards/code_complexity_reward/mean": 0.9234374761581421, "rewards/code_complexity_reward/std": 0.08977847546339035, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 829, "step_time": 39.941031967289746 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 105.310546875, "completions/mean_terminated_length": 103.71569061279297, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24213713267818093, "epoch": 0.9464082098061574, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05247661843895912, "kl": 0.20339006755966693, "learning_rate": 4.55214224722389e-08, "loss": 0.0010167864384129643, "num_tokens": 124035459.0, "reward": 2.3166017532348633, "reward_std": 0.49557265639305115, "rewards/code_complexity_reward/mean": 0.921191394329071, "rewards/code_complexity_reward/std": 0.10475554317235947, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 830, "step_time": 80.16567555442452 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 105.224609375, "completions/mean_terminated_length": 105.224609375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23841107659973204, "epoch": 0.9475484606613455, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.05745789781212807, "kl": 0.19849807093851268, "learning_rate": 4.364993564959841e-08, "loss": 0.0009925381746143103, "num_tokens": 124156138.0, "reward": 2.2646484375, "reward_std": 0.4924978017807007, "rewards/code_complexity_reward/mean": 0.9156250357627869, "rewards/code_complexity_reward/std": 0.12995557487010956, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 831, "step_time": 46.69776357430965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 107.64453125, "completions/mean_terminated_length": 107.64453125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24521560058929026, "epoch": 0.9486887115165337, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05413607507944107, "kl": 0.20884339860640466, "learning_rate": 4.1817392366815535e-08, "loss": 0.0010442410130053759, "num_tokens": 124281204.0, "reward": 2.2874512672424316, "reward_std": 0.47565487027168274, "rewards/code_complexity_reward/mean": 0.9183593988418579, "rewards/code_complexity_reward/std": 0.10007258504629135, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 832, "step_time": 39.923297378234565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 287.0, "completions/max_terminated_length": 287.0, "completions/mean_length": 103.375, "completions/mean_terminated_length": 103.375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24556369264610112, "epoch": 0.9498289623717218, "frac_reward_zero_std": 0.625, "grad_norm": 0.05468320474028587, "kl": 0.21760304854251444, "learning_rate": 4.002382167745428e-08, "loss": 0.0010881510097533464, "num_tokens": 124401932.0, "reward": 2.2628908157348633, "reward_std": 0.48066309094429016, "rewards/code_complexity_reward/mean": 0.916015625, "rewards/code_complexity_reward/std": 0.11080272495746613, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 833, "step_time": 34.93422294408083 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 286.0, "completions/max_terminated_length": 286.0, "completions/mean_length": 103.748046875, "completions/mean_terminated_length": 103.748046875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23780406033620238, "epoch": 0.95096921322691, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.04832856357097626, "kl": 0.21311818598769605, "learning_rate": 3.8269252017197056e-08, "loss": 0.0010655962396413088, "num_tokens": 124522779.0, "reward": 2.299853801727295, "reward_std": 0.4827263057231903, "rewards/code_complexity_reward/mean": 0.9219726324081421, "rewards/code_complexity_reward/std": 0.0997580960392952, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 834, "step_time": 35.27717292867601 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 474.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 103.03125, "completions/mean_terminated_length": 103.03125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2498533606994897, "epoch": 0.9521094640820981, "frac_reward_zero_std": 0.484375, "grad_norm": 0.17647972702980042, "kl": 0.3819098256062716, "learning_rate": 3.6553711203395906e-08, "loss": 0.001906347693875432, "num_tokens": 124644391.0, "reward": 2.281543254852295, "reward_std": 0.49580052495002747, "rewards/code_complexity_reward/mean": 0.9178711175918579, "rewards/code_complexity_reward/std": 0.12222640961408615, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 835, "step_time": 46.23299945052713 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 101.0390625, "completions/mean_terminated_length": 100.23483276367188, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24001458659768105, "epoch": 0.9532497149372862, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.05230312421917915, "kl": 0.2052074612583965, "learning_rate": 3.487722643463032e-08, "loss": 0.0010259209666401148, "num_tokens": 124765099.0, "reward": 2.3016114234924316, "reward_std": 0.4816938638687134, "rewards/code_complexity_reward/mean": 0.9293944835662842, "rewards/code_complexity_reward/std": 0.09676885604858398, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 836, "step_time": 49.39361329842359 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 99.443359375, "completions/mean_terminated_length": 99.443359375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24050398590043187, "epoch": 0.9543899657924744, "frac_reward_zero_std": 0.59375, "grad_norm": 0.0477035716176033, "kl": 0.19688865286298096, "learning_rate": 3.323982429027567e-08, "loss": 0.0009843719890341163, "num_tokens": 124884054.0, "reward": 2.369677782058716, "reward_std": 0.5271399021148682, "rewards/code_complexity_reward/mean": 0.9193359017372131, "rewards/code_complexity_reward/std": 0.11500509083271027, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 837, "step_time": 39.13465452659875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 427.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 104.873046875, "completions/mean_terminated_length": 104.873046875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.24404676351696253, "epoch": 0.9555302166476625, "frac_reward_zero_std": 0.5625, "grad_norm": 0.056383855640888214, "kl": 0.21875750133767724, "learning_rate": 3.164153073008297e-08, "loss": 0.0010936845792457461, "num_tokens": 125006933.0, "reward": 2.317187786102295, "reward_std": 0.4690207242965698, "rewards/code_complexity_reward/mean": 0.9291015863418579, "rewards/code_complexity_reward/std": 0.06608105450868607, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 838, "step_time": 41.43852844182402 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 465.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 101.767578125, "completions/mean_terminated_length": 101.767578125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24251903663389385, "epoch": 0.9566704675028507, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.05551822483539581, "kl": 0.20776184718124568, "learning_rate": 3.0082371093766435e-08, "loss": 0.001038663904182613, "num_tokens": 125127706.0, "reward": 2.2981934547424316, "reward_std": 0.4990626275539398, "rewards/code_complexity_reward/mean": 0.916210949420929, "rewards/code_complexity_reward/std": 0.11629008501768112, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 839, "step_time": 44.39953484851867 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 310.0, "completions/max_terminated_length": 310.0, "completions/mean_length": 102.17578125, "completions/mean_terminated_length": 102.17578125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2372551446314901, "epoch": 0.9578107183580388, "frac_reward_zero_std": 0.6953125, "grad_norm": 0.037782590836286545, "kl": 0.19887846149504185, "learning_rate": 2.856237010060242e-08, "loss": 0.0009944261983036995, "num_tokens": 125249312.0, "reward": 2.36474609375, "reward_std": 0.49892479181289673, "rewards/code_complexity_reward/mean": 0.9268554449081421, "rewards/code_complexity_reward/std": 0.08900666236877441, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 840, "step_time": 35.76893021725118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 106.76171875, "completions/mean_terminated_length": 105.96868896484375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24696301249787211, "epoch": 0.9589509692132269, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.07517043501138687, "kl": 0.24422083143144846, "learning_rate": 2.708155184903666e-08, "loss": 0.0012215982424095273, "num_tokens": 125372542.0, "reward": 2.2494142055511475, "reward_std": 0.4970235824584961, "rewards/code_complexity_reward/mean": 0.90966796875, "rewards/code_complexity_reward/std": 0.13638213276863098, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 841, "step_time": 49.09551933594048 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 300.0, "completions/max_terminated_length": 300.0, "completions/mean_length": 100.08203125, "completions/mean_terminated_length": 100.08203125, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.23844301514327526, "epoch": 0.9600912200684151, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.05338239297270775, "kl": 0.19681075075641274, "learning_rate": 2.5639939816302917e-08, "loss": 0.0009837883990257978, "num_tokens": 125491892.0, "reward": 2.2819337844848633, "reward_std": 0.45780980587005615, "rewards/code_complexity_reward/mean": 0.927050769329071, "rewards/code_complexity_reward/std": 0.07879780232906342, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 842, "step_time": 36.77471300307661 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 104.921875, "completions/mean_terminated_length": 104.921875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24062489182688296, "epoch": 0.9612314709236032, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.05022776871919632, "kl": 0.1926362479571253, "learning_rate": 2.4237556858050794e-08, "loss": 0.0009631087305024266, "num_tokens": 125613236.0, "reward": 2.29931640625, "reward_std": 0.49212440848350525, "rewards/code_complexity_reward/mean": 0.9200195074081421, "rewards/code_complexity_reward/std": 0.11034813523292542, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 843, "step_time": 36.25142548419535 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 100.98046875, "completions/mean_terminated_length": 100.98046875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24317993712611496, "epoch": 0.9623717217787914, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.049195222556591034, "kl": 0.20924274530261755, "learning_rate": 2.2874425207982386e-08, "loss": 0.0010462000500410795, "num_tokens": 125734598.0, "reward": 2.2697267532348633, "reward_std": 0.45350509881973267, "rewards/code_complexity_reward/mean": 0.928515613079071, "rewards/code_complexity_reward/std": 0.08250804245471954, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 844, "step_time": 40.93817859515548 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 240.0, "completions/max_terminated_length": 240.0, "completions/mean_length": 100.150390625, "completions/mean_terminated_length": 100.150390625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24572978937067091, "epoch": 0.9635119726339795, "frac_reward_zero_std": 0.6328125, "grad_norm": 0.05400016903877258, "kl": 0.21128877927549183, "learning_rate": 2.15505664775012e-08, "loss": 0.0010562522802501917, "num_tokens": 125854115.0, "reward": 2.3416993618011475, "reward_std": 0.5032909512519836, "rewards/code_complexity_reward/mean": 0.92529296875, "rewards/code_complexity_reward/std": 0.10552223771810532, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 845, "step_time": 32.088774509727955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 462.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 100.0390625, "completions/mean_terminated_length": 100.0390625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.25078777503222227, "epoch": 0.9646522234891676, "frac_reward_zero_std": 0.703125, "grad_norm": 0.04317848011851311, "kl": 0.23469420231413096, "learning_rate": 2.026600165536824e-08, "loss": 0.0011735802982002497, "num_tokens": 125973739.0, "reward": 2.241211175918579, "reward_std": 0.4559144675731659, "rewards/code_complexity_reward/mean": 0.923632800579071, "rewards/code_complexity_reward/std": 0.10462909191846848, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 846, "step_time": 46.532187038101256 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 103.87890625, "completions/mean_terminated_length": 103.08023071289062, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.240998818539083, "epoch": 0.9657924743443558, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.05218718573451042, "kl": 0.2020764984190464, "learning_rate": 1.9020751107370894e-08, "loss": 0.0010103669483214617, "num_tokens": 126094029.0, "reward": 2.2716312408447266, "reward_std": 0.49071332812309265, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.12079966068267822, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 847, "step_time": 56.24840477388352 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 303.0, "completions/max_terminated_length": 303.0, "completions/mean_length": 100.91796875, "completions/mean_terminated_length": 100.91796875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2382639276329428, "epoch": 0.9669327251995439, "frac_reward_zero_std": 0.546875, "grad_norm": 0.06576541066169739, "kl": 0.20107891666702926, "learning_rate": 1.7814834575997364e-08, "loss": 0.0010054492158815265, "num_tokens": 126215339.0, "reward": 2.2611329555511475, "reward_std": 0.4551973044872284, "rewards/code_complexity_reward/mean": 0.9228515625, "rewards/code_complexity_reward/std": 0.09095746278762817, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 848, "step_time": 35.618023046292365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 99.072265625, "completions/mean_terminated_length": 99.072265625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23492065723985434, "epoch": 0.9680729760547321, "frac_reward_zero_std": 0.625, "grad_norm": 0.05573631450533867, "kl": 0.199364154599607, "learning_rate": 1.664827118012663e-08, "loss": 0.000996717601083219, "num_tokens": 126334852.0, "reward": 2.2858400344848633, "reward_std": 0.46081236004829407, "rewards/code_complexity_reward/mean": 0.927050769329071, "rewards/code_complexity_reward/std": 0.07811184972524643, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 849, "step_time": 35.82902354747057 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 102.771484375, "completions/mean_terminated_length": 102.771484375, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.2510602311231196, "epoch": 0.9692132269099202, "frac_reward_zero_std": 0.671875, "grad_norm": 0.050890080630779266, "kl": 0.2276359610259533, "learning_rate": 1.5521079414723695e-08, "loss": 0.0011386030819267035, "num_tokens": 126456471.0, "reward": 2.289844036102295, "reward_std": 0.47316890954971313, "rewards/code_complexity_reward/mean": 0.9222656488418579, "rewards/code_complexity_reward/std": 0.09243574738502502, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 850, "step_time": 44.537183033302426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 432.0, "completions/max_terminated_length": 432.0, "completions/mean_length": 104.54296875, "completions/mean_terminated_length": 104.54296875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.25253123906441033, "epoch": 0.9703534777651083, "frac_reward_zero_std": 0.515625, "grad_norm": 0.07113254815340042, "kl": 0.23604706884361804, "learning_rate": 1.4433277150545932e-08, "loss": 0.0011795408790931106, "num_tokens": 126580253.0, "reward": 2.2603516578674316, "reward_std": 0.468895822763443, "rewards/code_complexity_reward/mean": 0.9220702648162842, "rewards/code_complexity_reward/std": 0.10930848866701126, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 851, "step_time": 52.555076740682125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 102.07421875, "completions/mean_terminated_length": 102.07421875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2384332255460322, "epoch": 0.9714937286202965, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.051364365965127945, "kl": 0.20290796202607453, "learning_rate": 1.3384881633861923e-08, "loss": 0.001014671172015369, "num_tokens": 126701075.0, "reward": 2.3667969703674316, "reward_std": 0.5158553719520569, "rewards/code_complexity_reward/mean": 0.9201171398162842, "rewards/code_complexity_reward/std": 0.10475759208202362, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 852, "step_time": 47.111714063212276 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 103.244140625, "completions/mean_terminated_length": 102.44422912597656, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24185312422923744, "epoch": 0.9726339794754846, "frac_reward_zero_std": 0.578125, "grad_norm": 0.24137720465660095, "kl": 0.4391664331778884, "learning_rate": 1.2375909486175008e-08, "loss": 0.0021953866817057133, "num_tokens": 126823888.0, "reward": 2.2562012672424316, "reward_std": 0.5004273056983948, "rewards/code_complexity_reward/mean": 0.9132812023162842, "rewards/code_complexity_reward/std": 0.13897645473480225, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 853, "step_time": 64.6231040628627 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 295.0, "completions/max_terminated_length": 295.0, "completions/mean_length": 102.166015625, "completions/mean_terminated_length": 102.166015625, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.24677378358319402, "epoch": 0.9737742303306728, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.046345677226781845, "kl": 0.20177258015610278, "learning_rate": 1.1406376703962384e-08, "loss": 0.0010087218834087253, "num_tokens": 126944325.0, "reward": 2.273486614227295, "reward_std": 0.4492480158805847, "rewards/code_complexity_reward/mean": 0.929003894329071, "rewards/code_complexity_reward/std": 0.0770898312330246, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 854, "step_time": 34.745357423089445 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 105.173828125, "completions/mean_terminated_length": 105.173828125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24493313813582063, "epoch": 0.9749144811858609, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05931542068719864, "kl": 0.21827948838472366, "learning_rate": 1.0476298658420036e-08, "loss": 0.001091197831556201, "num_tokens": 127066426.0, "reward": 2.2720704078674316, "reward_std": 0.4803794324398041, "rewards/code_complexity_reward/mean": 0.9191405773162842, "rewards/code_complexity_reward/std": 0.11689410358667374, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 855, "step_time": 40.75477554555982 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 102.51953125, "completions/mean_terminated_length": 101.71820068359375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24625724600628018, "epoch": 0.976054732041049, "frac_reward_zero_std": 0.609375, "grad_norm": 0.04971766471862793, "kl": 0.20200964633841068, "learning_rate": 9.58569009521959e-09, "loss": 0.0010099400533363223, "num_tokens": 127190776.0, "reward": 2.2821779251098633, "reward_std": 0.49041295051574707, "rewards/code_complexity_reward/mean": 0.9197264909744263, "rewards/code_complexity_reward/std": 0.11662878841161728, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 856, "step_time": 64.48235603421926 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 101.390625, "completions/mean_terminated_length": 101.390625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24486622563563287, "epoch": 0.9771949828962372, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.05786637216806412, "kl": 0.20202787755988538, "learning_rate": 8.73456513427462e-09, "loss": 0.0010100343497470021, "num_tokens": 127312876.0, "reward": 2.2841310501098633, "reward_std": 0.5017809867858887, "rewards/code_complexity_reward/mean": 0.919726550579071, "rewards/code_complexity_reward/std": 0.13014940917491913, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 857, "step_time": 44.745537389069796 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 317.0, "completions/max_terminated_length": 317.0, "completions/mean_length": 103.396484375, "completions/mean_terminated_length": 103.396484375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24536854191683233, "epoch": 0.9783352337514253, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.08367118239402771, "kl": 0.2988036659080535, "learning_rate": 7.922937269516096e-09, "loss": 0.0014934050850570202, "num_tokens": 127432523.0, "reward": 2.28173828125, "reward_std": 0.5315151214599609, "rewards/code_complexity_reward/mean": 0.9083007574081421, "rewards/code_complexity_reward/std": 0.15744549036026, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 858, "step_time": 36.9886094564572 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 103.021484375, "completions/mean_terminated_length": 103.021484375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2414942046161741, "epoch": 0.9794754846066135, "frac_reward_zero_std": 0.6640625, "grad_norm": 0.04493594542145729, "kl": 0.20113336062058806, "learning_rate": 7.150819368679229e-09, "loss": 0.0010055124294012785, "num_tokens": 127553874.0, "reward": 2.3273439407348633, "reward_std": 0.47575706243515015, "rewards/code_complexity_reward/mean": 0.926562488079071, "rewards/code_complexity_reward/std": 0.07219984382390976, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 859, "step_time": 49.27815591264516 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 376.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 99.0703125, "completions/mean_terminated_length": 99.0703125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24910847353748977, "epoch": 0.9806157354618016, "frac_reward_zero_std": 0.671875, "grad_norm": 0.04487563297152519, "kl": 0.21334302541799843, "learning_rate": 6.418223673099466e-09, "loss": 0.001066730823367834, "num_tokens": 127672710.0, "reward": 2.3119142055511475, "reward_std": 0.513815701007843, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.12641988694667816, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 860, "step_time": 39.175114469602704 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 103.3984375, "completions/mean_terminated_length": 103.3984375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24303746595978737, "epoch": 0.9817559863169898, "frac_reward_zero_std": 0.5625, "grad_norm": 0.053811296820640564, "kl": 0.19928916520439088, "learning_rate": 5.725161797517087e-09, "loss": 0.0009963905904442072, "num_tokens": 127793942.0, "reward": 2.2657227516174316, "reward_std": 0.49124613404273987, "rewards/code_complexity_reward/mean": 0.913769543170929, "rewards/code_complexity_reward/std": 0.12644797563552856, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 861, "step_time": 45.99060732964426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 105.306640625, "completions/mean_terminated_length": 105.306640625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.25616579153575003, "epoch": 0.9828962371721779, "frac_reward_zero_std": 0.578125, "grad_norm": 0.04930088669061661, "kl": 0.2028176395688206, "learning_rate": 5.071644729895408e-09, "loss": 0.0010140397353097796, "num_tokens": 127917411.0, "reward": 2.3060550689697266, "reward_std": 0.5029001235961914, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.11655354499816895, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 862, "step_time": 57.48633955232799 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 286.0, "completions/max_terminated_length": 286.0, "completions/mean_length": 103.703125, "completions/mean_terminated_length": 103.703125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2314486331306398, "epoch": 0.984036488027366, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.06253435462713242, "kl": 0.21031579852569848, "learning_rate": 4.4576828312442586e-09, "loss": 0.0010512593435123563, "num_tokens": 128039127.0, "reward": 2.2994141578674316, "reward_std": 0.5187197327613831, "rewards/code_complexity_reward/mean": 0.9142577648162842, "rewards/code_complexity_reward/std": 0.13484004139900208, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 863, "step_time": 33.697209098376334 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 104.115234375, "completions/mean_terminated_length": 104.115234375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24539354746229947, "epoch": 0.9851767388825542, "frac_reward_zero_std": 0.65625, "grad_norm": 0.053048618137836456, "kl": 0.19563305913470685, "learning_rate": 3.8832858354567736e-09, "loss": 0.000978159485384822, "num_tokens": 128161266.0, "reward": 2.268310785293579, "reward_std": 0.44945383071899414, "rewards/code_complexity_reward/mean": 0.9258788824081421, "rewards/code_complexity_reward/std": 0.07744203507900238, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.025282343849539757, "step": 864, "step_time": 50.733814591541886 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 324.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 100.6796875, "completions/mean_terminated_length": 100.6796875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24731322890147567, "epoch": 0.9863169897377423, "frac_reward_zero_std": 0.640625, "grad_norm": 0.06296393275260925, "kl": 0.2010595102328807, "learning_rate": 3.348462849155909e-09, "loss": 0.001005173078738153, "num_tokens": 128282050.0, "reward": 2.3257813453674316, "reward_std": 0.5066962838172913, "rewards/code_complexity_reward/mean": 0.920117199420929, "rewards/code_complexity_reward/std": 0.11685534566640854, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 865, "step_time": 36.972956960089505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 268.0, "completions/max_terminated_length": 268.0, "completions/mean_length": 99.505859375, "completions/mean_terminated_length": 99.505859375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2517169516067952, "epoch": 0.9874572405929305, "frac_reward_zero_std": 0.5625, "grad_norm": 0.061181727796792984, "kl": 0.2047175585757941, "learning_rate": 2.8532223515481683e-09, "loss": 0.0010234990622848272, "num_tokens": 128402057.0, "reward": 2.3341798782348633, "reward_std": 0.4799948036670685, "rewards/code_complexity_reward/mean": 0.928515613079071, "rewards/code_complexity_reward/std": 0.0796724185347557, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 866, "step_time": 40.9716827487573 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 229.0, "completions/max_terminated_length": 229.0, "completions/mean_length": 100.9140625, "completions/mean_terminated_length": 100.9140625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.25056519522331655, "epoch": 0.9885974914481186, "frac_reward_zero_std": 0.6171875, "grad_norm": 0.04520236328244209, "kl": 0.2020726641640067, "learning_rate": 2.3975721942903762e-09, "loss": 0.0010103408712893724, "num_tokens": 128523521.0, "reward": 2.277148723602295, "reward_std": 0.496489942073822, "rewards/code_complexity_reward/mean": 0.9124999642372131, "rewards/code_complexity_reward/std": 0.12713922560214996, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 867, "step_time": 33.824031102471054 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 314.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 102.9921875, "completions/mean_terminated_length": 102.9921875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2500294435303658, "epoch": 0.9897377423033067, "frac_reward_zero_std": 0.640625, "grad_norm": 0.04789186269044876, "kl": 0.19805417896714061, "learning_rate": 1.9815196013650563e-09, "loss": 0.0009900372242555022, "num_tokens": 128643657.0, "reward": 2.281738519668579, "reward_std": 0.46442267298698425, "rewards/code_complexity_reward/mean": 0.9258788824081421, "rewards/code_complexity_reward/std": 0.08973328024148941, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 868, "step_time": 36.26263254415244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 440.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 101.392578125, "completions/mean_terminated_length": 101.392578125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24261612305417657, "epoch": 0.9908779931584949, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05505850538611412, "kl": 0.2028319132514298, "learning_rate": 1.6050711689663544e-09, "loss": 0.0010140155209228396, "num_tokens": 128763262.0, "reward": 2.306640625, "reward_std": 0.5054641366004944, "rewards/code_complexity_reward/mean": 0.9214844107627869, "rewards/code_complexity_reward/std": 0.12085532397031784, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 869, "step_time": 43.98298567626625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 107.4921875, "completions/mean_terminated_length": 107.4921875, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.24672172125428915, "epoch": 0.992018244013683, "frac_reward_zero_std": 0.5859375, "grad_norm": 0.059129372239112854, "kl": 0.20282559003680944, "learning_rate": 1.2682328653940146e-09, "loss": 0.0010137974750250578, "num_tokens": 128884394.0, "reward": 2.2510743141174316, "reward_std": 0.48854175209999084, "rewards/code_complexity_reward/mean": 0.911816418170929, "rewards/code_complexity_reward/std": 0.12985020875930786, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 870, "step_time": 37.80213701445609 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 100.986328125, "completions/mean_terminated_length": 100.986328125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2449431070126593, "epoch": 0.9931584948688712, "frac_reward_zero_std": 0.609375, "grad_norm": 0.052398521453142166, "kl": 0.2076637598220259, "learning_rate": 9.710100309603954e-10, "loss": 0.0010384804336354136, "num_tokens": 129005279.0, "reward": 2.339648723602295, "reward_std": 0.5463612079620361, "rewards/code_complexity_reward/mean": 0.9120117425918579, "rewards/code_complexity_reward/std": 0.14820271730422974, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 871, "step_time": 37.944453430362046 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 100.302734375, "completions/mean_terminated_length": 100.302734375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2521348176524043, "epoch": 0.9942987457240593, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05441361665725708, "kl": 0.21031375066377223, "learning_rate": 7.134073779044293e-10, "loss": 0.0010517037007957697, "num_tokens": 129124538.0, "reward": 2.284960985183716, "reward_std": 0.48153620958328247, "rewards/code_complexity_reward/mean": 0.9212890267372131, "rewards/code_complexity_reward/std": 0.10592015087604523, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 872, "step_time": 45.36824326682836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 338.0, "completions/max_terminated_length": 338.0, "completions/mean_length": 100.86328125, "completions/mean_terminated_length": 100.86328125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24934915360063314, "epoch": 0.9954389965792474, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.05026278644800186, "kl": 0.20253438293002546, "learning_rate": 4.954289903180698e-10, "loss": 0.0010126831475645304, "num_tokens": 129244612.0, "reward": 2.308837890625, "reward_std": 0.5040552616119385, "rewards/code_complexity_reward/mean": 0.9190429449081421, "rewards/code_complexity_reward/std": 0.11344827711582184, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 873, "step_time": 47.45990268327296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 103.228515625, "completions/mean_terminated_length": 103.228515625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24032388348132372, "epoch": 0.9965792474344356, "frac_reward_zero_std": 0.6484375, "grad_norm": 0.043452490121126175, "kl": 0.2084324734751135, "learning_rate": 3.1707832408134354e-10, "loss": 0.0010421625338494778, "num_tokens": 129365517.0, "reward": 2.3524415493011475, "reward_std": 0.4832000732421875, "rewards/code_complexity_reward/mean": 0.92822265625, "rewards/code_complexity_reward/std": 0.06852775067090988, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 874, "step_time": 42.514279410243034 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 106.720703125, "completions/mean_terminated_length": 105.9275894165039, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23506706533953547, "epoch": 0.9977194982896237, "frac_reward_zero_std": 0.59375, "grad_norm": 0.051216620951890945, "kl": 0.21433608257211745, "learning_rate": 1.7835820680600634e-10, "loss": 0.001071437494829297, "num_tokens": 129488938.0, "reward": 2.3196778297424316, "reward_std": 0.5195985436439514, "rewards/code_complexity_reward/mean": 0.9126952886581421, "rewards/code_complexity_reward/std": 0.12360755354166031, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 875, "step_time": 55.24143814481795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 306.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 102.33984375, "completions/mean_terminated_length": 102.33984375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2407450689934194, "epoch": 0.9988597491448119, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.052528779953718185, "kl": 0.1945526758208871, "learning_rate": 7.927083779335487e-11, "loss": 0.0009726736461743712, "num_tokens": 129610236.0, "reward": 2.300586223602295, "reward_std": 0.46866557002067566, "rewards/code_complexity_reward/mean": 0.9261718392372131, "rewards/code_complexity_reward/std": 0.0827522724866867, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 876, "step_time": 43.048616603948176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 306.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 105.087890625, "completions/mean_terminated_length": 105.087890625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23941138037480414, "epoch": 1.0, "frac_reward_zero_std": 0.6015625, "grad_norm": 0.05000397562980652, "kl": 0.19697153195738792, "learning_rate": 1.98177879973116e-11, "loss": 0.0009848183253780007, "num_tokens": 129732413.0, "reward": 2.312549114227295, "reward_std": 0.5010702610015869, "rewards/code_complexity_reward/mean": 0.9168944954872131, "rewards/code_complexity_reward/std": 0.1162135973572731, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 877, "step_time": 41.75086905248463 }, { "epoch": 1.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 162.34, "eval_completions/max_terminated_length": 162.34, "eval_completions/mean_length": 105.585, "eval_completions/mean_terminated_length": 105.585, "eval_completions/min_length": 71.46, "eval_completions/min_terminated_length": 71.46, "eval_entropy": 0.24230012089014052, "eval_frac_reward_zero_std": 0.61, "eval_kl": 0.2040590050816536, "eval_loss": 0.001022406853735447, "eval_num_tokens": 129732413.0, "eval_reward": 2.2570001125335692, "eval_reward_std": 0.34441455591470005, "eval_rewards/code_complexity_reward/mean": 0.9157499778270721, "eval_rewards/code_complexity_reward/std": 0.056951567456126215, "eval_rewards/code_execution_reward/mean": 0.25, "eval_rewards/code_execution_reward/std": 0.29089601755142214, "eval_rewards/code_syntax_reward/mean": 0.49125, "eval_rewards/code_syntax_reward/std": 0.019864802658557893, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.5, "eval_rewards/xmlcount_reward_func/std": 0.0, "eval_runtime": 373.4101, "eval_samples_per_second": 0.268, "eval_steps_per_second": 0.035, "step": 877 } ], "logging_steps": 1, "max_steps": 877, "num_input_tokens_seen": 129732413, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 1, "trial_name": null, "trial_params": null }