{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 100, "global_step": 1755, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 304.072265625, "completions/mean_terminated_length": 288.3466491699219, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.26909572863951325, "epoch": 0.0005698005698005698, "frac_reward_zero_std": 0.0, "grad_norm": 0.058154359459877014, "kl": 0.0, "learning_rate": 0.0, "loss": -2.1973391994833946e-08, "num_tokens": 226181.0, "reward": 1.9341307878494263, "reward_std": 0.8174188733100891, "rewards/code_complexity_reward/mean": 0.6881835460662842, "rewards/code_complexity_reward/std": 0.31120941042900085, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4228515625, "rewards/code_syntax_reward/std": 0.18079319596290588, "rewards/reasoning_present_reward_func/mean": 0.09140625596046448, "rewards/reasoning_present_reward_func/std": 0.028054581955075264, "rewards/xmlcount_reward_func/mean": 0.425048828125, "rewards/xmlcount_reward_func/std": 0.09905990213155746, "step": 1, "step_time": 58.38009595498443 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.076171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 311.158203125, "completions/mean_terminated_length": 294.5982971191406, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.26130872941575944, "epoch": 0.0011396011396011395, "frac_reward_zero_std": 0.0, "grad_norm": 0.06200707331299782, "kl": 0.0, "learning_rate": 2.840909090909091e-08, "loss": -1.9615981727838516e-08, "num_tokens": 453446.0, "reward": 1.864013671875, "reward_std": 0.8421388268470764, "rewards/code_complexity_reward/mean": 0.655468761920929, "rewards/code_complexity_reward/std": 0.32235583662986755, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.41015625, "rewards/code_syntax_reward/std": 0.19215121865272522, "rewards/reasoning_present_reward_func/mean": 0.08964844048023224, "rewards/reasoning_present_reward_func/std": 0.030492907389998436, "rewards/xmlcount_reward_func/mean": 0.413818359375, "rewards/xmlcount_reward_func/std": 0.09978312999010086, "step": 2, "step_time": 70.5968795819208 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.056640625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 307.970703125, "completions/mean_terminated_length": 295.7204895019531, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.2675860656891018, "epoch": 0.0017094017094017094, "frac_reward_zero_std": 0.0, "grad_norm": 0.052059054374694824, "kl": 0.0013551432230087812, "learning_rate": 5.681818181818182e-08, "loss": 6.785623554605991e-06, "num_tokens": 679031.0, "reward": 1.8881347179412842, "reward_std": 0.7903958559036255, "rewards/code_complexity_reward/mean": 0.6783202886581421, "rewards/code_complexity_reward/std": 0.3113299310207367, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4189453125, "rewards/code_syntax_reward/std": 0.1844557821750641, "rewards/reasoning_present_reward_func/mean": 0.09140625596046448, "rewards/reasoning_present_reward_func/std": 0.028054583817720413, "rewards/xmlcount_reward_func/mean": 0.424072265625, "rewards/xmlcount_reward_func/std": 0.09287413954734802, "step": 3, "step_time": 59.134837133809924 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.076171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 313.93359375, "completions/mean_terminated_length": 297.6025390625, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.26574606634676456, "epoch": 0.002279202279202279, "frac_reward_zero_std": 0.0, "grad_norm": 0.04605681449174881, "kl": 0.0013815366110065952, "learning_rate": 8.522727272727273e-08, "loss": 6.945178029127419e-06, "num_tokens": 911381.0, "reward": 1.9944825172424316, "reward_std": 0.8305402398109436, "rewards/code_complexity_reward/mean": 0.69287109375, "rewards/code_complexity_reward/std": 0.30545106530189514, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.42578125, "rewards/code_syntax_reward/std": 0.17794041335582733, "rewards/reasoning_present_reward_func/mean": 0.08945312350988388, "rewards/reasoning_present_reward_func/std": 0.03074568510055542, "rewards/xmlcount_reward_func/mean": 0.423095703125, "rewards/xmlcount_reward_func/std": 0.09786135703325272, "step": 4, "step_time": 52.80495351366699 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 318.681640625, "completions/mean_terminated_length": 303.18353271484375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2672395780682564, "epoch": 0.002849002849002849, "frac_reward_zero_std": 0.0, "grad_norm": 0.03912132978439331, "kl": 0.0014235526341508375, "learning_rate": 1.1363636363636364e-07, "loss": 7.1394897531718016e-06, "num_tokens": 1144890.0, "reward": 1.864501953125, "reward_std": 0.7390746474266052, "rewards/code_complexity_reward/mean": 0.6953125, "rewards/code_complexity_reward/std": 0.2990812659263611, "rewards/code_execution_reward/mean": 0.22265625, "rewards/code_execution_reward/std": 0.41643625497817993, "rewards/code_syntax_reward/mean": 0.427734375, "rewards/code_syntax_reward/std": 0.17598573863506317, "rewards/reasoning_present_reward_func/mean": 0.0908203125, "rewards/reasoning_present_reward_func/std": 0.028902139514684677, "rewards/xmlcount_reward_func/mean": 0.427978515625, "rewards/xmlcount_reward_func/std": 0.09203699231147766, "step": 5, "step_time": 62.39822655264288 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.044921875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 293.82421875, "completions/mean_terminated_length": 283.5623779296875, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.268439564621076, "epoch": 0.003418803418803419, "frac_reward_zero_std": 0.0, "grad_norm": 0.05326269194483757, "kl": 0.0013797474521197728, "learning_rate": 1.4204545454545455e-07, "loss": 6.871094228699803e-06, "num_tokens": 1366048.0, "reward": 1.994970679283142, "reward_std": 0.8026270270347595, "rewards/code_complexity_reward/mean": 0.7162109613418579, "rewards/code_complexity_reward/std": 0.2943381369113922, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.43359375, "rewards/code_syntax_reward/std": 0.16985194385051727, "rewards/reasoning_present_reward_func/mean": 0.08906250447034836, "rewards/reasoning_present_reward_func/std": 0.031241435557603836, "rewards/xmlcount_reward_func/mean": 0.416259765625, "rewards/xmlcount_reward_func/std": 0.10214444994926453, "step": 6, "step_time": 61.523920251987875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 305.060546875, "completions/mean_terminated_length": 287.5233154296875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.26639376813545823, "epoch": 0.003988603988603989, "frac_reward_zero_std": 0.0, "grad_norm": 0.059001654386520386, "kl": 0.0013823104418406729, "learning_rate": 1.7045454545454545e-07, "loss": 6.948943337192759e-06, "num_tokens": 1590935.0, "reward": 1.9189453125, "reward_std": 0.7989214062690735, "rewards/code_complexity_reward/mean": 0.702832043170929, "rewards/code_complexity_reward/std": 0.30880236625671387, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.42578125, "rewards/code_syntax_reward/std": 0.17794041335582733, "rewards/reasoning_present_reward_func/mean": 0.08964844048023224, "rewards/reasoning_present_reward_func/std": 0.030492907389998436, "rewards/xmlcount_reward_func/mean": 0.42529296875, "rewards/xmlcount_reward_func/std": 0.09659016132354736, "step": 7, "step_time": 70.26412656530738 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.095703125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 312.8828125, "completions/mean_terminated_length": 291.8099365234375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2616435894742608, "epoch": 0.004558404558404558, "frac_reward_zero_std": 0.0, "grad_norm": 0.057591114193201065, "kl": 0.0013492748357748496, "learning_rate": 1.9886363636363638e-07, "loss": 6.7731598392128944e-06, "num_tokens": 1821747.0, "reward": 1.9371094703674316, "reward_std": 0.8028035759925842, "rewards/code_complexity_reward/mean": 0.6773437261581421, "rewards/code_complexity_reward/std": 0.3064926564693451, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.423828125, "rewards/code_syntax_reward/std": 0.17985260486602783, "rewards/reasoning_present_reward_func/mean": 0.091796875, "rewards/reasoning_present_reward_func/std": 0.02746807038784027, "rewards/xmlcount_reward_func/mean": 0.4296875, "rewards/xmlcount_reward_func/std": 0.08349625021219254, "step": 8, "step_time": 69.9122311854735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 308.666015625, "completions/mean_terminated_length": 292.3649597167969, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.25663648173213005, "epoch": 0.005128205128205128, "frac_reward_zero_std": 0.0, "grad_norm": 0.04516186937689781, "kl": 0.0013215701710578287, "learning_rate": 2.2727272727272729e-07, "loss": 6.617054168600589e-06, "num_tokens": 2047480.0, "reward": 1.9451661109924316, "reward_std": 0.7880263924598694, "rewards/code_complexity_reward/mean": 0.696093738079071, "rewards/code_complexity_reward/std": 0.3006587326526642, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4287109375, "rewards/code_syntax_reward/std": 0.17499202489852905, "rewards/reasoning_present_reward_func/mean": 0.09062500298023224, "rewards/reasoning_present_reward_func/std": 0.029176566749811172, "rewards/xmlcount_reward_func/mean": 0.425048828125, "rewards/xmlcount_reward_func/std": 0.09655893594026566, "step": 9, "step_time": 58.62356204353273 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.048828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 299.466796875, "completions/mean_terminated_length": 288.5564880371094, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.273256505606696, "epoch": 0.005698005698005698, "frac_reward_zero_std": 0.0, "grad_norm": 0.057338960468769073, "kl": 0.00142223120383278, "learning_rate": 2.556818181818182e-07, "loss": 7.100636139512062e-06, "num_tokens": 2266623.0, "reward": 1.8759278059005737, "reward_std": 0.8048427104949951, "rewards/code_complexity_reward/mean": 0.6729491949081421, "rewards/code_complexity_reward/std": 0.3165612816810608, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4169921875, "rewards/code_syntax_reward/std": 0.18622928857803345, "rewards/reasoning_present_reward_func/mean": 0.09042969346046448, "rewards/reasoning_present_reward_func/std": 0.02944713830947876, "rewards/xmlcount_reward_func/mean": 0.422119140625, "rewards/xmlcount_reward_func/std": 0.09613495320081711, "step": 10, "step_time": 48.51318749412894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 322.0234375, "completions/mean_terminated_length": 298.6929931640625, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2710467716678977, "epoch": 0.0062678062678062675, "frac_reward_zero_std": 0.0, "grad_norm": 0.05852578952908516, "kl": 0.0014334714487631572, "learning_rate": 2.840909090909091e-07, "loss": 7.169204764068127e-06, "num_tokens": 2503747.0, "reward": 1.733154296875, "reward_std": 0.8196302056312561, "rewards/code_complexity_reward/mean": 0.630566418170929, "rewards/code_complexity_reward/std": 0.34822484850883484, "rewards/code_execution_reward/mean": 0.20703125, "rewards/code_execution_reward/std": 0.40557438135147095, "rewards/code_syntax_reward/mean": 0.392578125, "rewards/code_syntax_reward/std": 0.20555779337882996, "rewards/reasoning_present_reward_func/mean": 0.08867187798023224, "rewards/reasoning_present_reward_func/std": 0.03172462433576584, "rewards/xmlcount_reward_func/mean": 0.414306640625, "rewards/xmlcount_reward_func/std": 0.0980442687869072, "step": 11, "step_time": 80.17032906971872 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 314.8203125, "completions/mean_terminated_length": 296.2820739746094, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.26291930908337235, "epoch": 0.006837606837606838, "frac_reward_zero_std": 0.0, "grad_norm": 0.042030218988657, "kl": 0.0013412324042292312, "learning_rate": 3.125e-07, "loss": 6.7484506871551275e-06, "num_tokens": 2734015.0, "reward": 1.967675805091858, "reward_std": 0.8047924637794495, "rewards/code_complexity_reward/mean": 0.6996093988418579, "rewards/code_complexity_reward/std": 0.30073270201683044, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4296875, "rewards/code_syntax_reward/std": 0.17398715019226074, "rewards/reasoning_present_reward_func/mean": 0.087890625, "rewards/reasoning_present_reward_func/std": 0.03265552595257759, "rewards/xmlcount_reward_func/mean": 0.42236328125, "rewards/xmlcount_reward_func/std": 0.09868449717760086, "step": 12, "step_time": 60.73664485104382 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.060546875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 303.4921875, "completions/mean_terminated_length": 290.0540466308594, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2628725725226104, "epoch": 0.007407407407407408, "frac_reward_zero_std": 0.0, "grad_norm": 0.05519591271877289, "kl": 0.0013953271209175, "learning_rate": 3.409090909090909e-07, "loss": 6.954127456992865e-06, "num_tokens": 2959331.0, "reward": 1.894384741783142, "reward_std": 0.8182656168937683, "rewards/code_complexity_reward/mean": 0.6864257454872131, "rewards/code_complexity_reward/std": 0.32139360904693604, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.416015625, "rewards/code_syntax_reward/std": 0.1871020793914795, "rewards/reasoning_present_reward_func/mean": 0.08613280951976776, "rewards/reasoning_present_reward_func/std": 0.034594181925058365, "rewards/xmlcount_reward_func/mean": 0.414794921875, "rewards/xmlcount_reward_func/std": 0.10242704302072525, "step": 13, "step_time": 59.753370475023985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 312.5859375, "completions/mean_terminated_length": 296.5991516113281, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.27724887523800135, "epoch": 0.007977207977207978, "frac_reward_zero_std": 0.0, "grad_norm": 0.05420304462313652, "kl": 0.001455250967410393, "learning_rate": 3.693181818181819e-07, "loss": 7.345282938331366e-06, "num_tokens": 3190519.0, "reward": 1.783935546875, "reward_std": 0.8215969204902649, "rewards/code_complexity_reward/mean": 0.6428711414337158, "rewards/code_complexity_reward/std": 0.33517375588417053, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.400390625, "rewards/code_syntax_reward/std": 0.19990174472332, "rewards/reasoning_present_reward_func/mean": 0.08906250447034836, "rewards/reasoning_present_reward_func/std": 0.031241435557603836, "rewards/xmlcount_reward_func/mean": 0.415283203125, "rewards/xmlcount_reward_func/std": 0.1046009510755539, "step": 14, "step_time": 61.68405893538147 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.076171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 303.546875, "completions/mean_terminated_length": 286.3594055175781, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.2602951454464346, "epoch": 0.008547008547008548, "frac_reward_zero_std": 0.0, "grad_norm": 0.060391996055841446, "kl": 0.0013536482183553744, "learning_rate": 3.9772727272727276e-07, "loss": 6.816779205109924e-06, "num_tokens": 3414631.0, "reward": 1.890527367591858, "reward_std": 0.8224665522575378, "rewards/code_complexity_reward/mean": 0.6753906011581421, "rewards/code_complexity_reward/std": 0.3261137008666992, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.416015625, "rewards/code_syntax_reward/std": 0.1871020793914795, "rewards/reasoning_present_reward_func/mean": 0.09062500298023224, "rewards/reasoning_present_reward_func/std": 0.029176566749811172, "rewards/xmlcount_reward_func/mean": 0.41748046875, "rewards/xmlcount_reward_func/std": 0.10027896612882614, "step": 15, "step_time": 80.34212884958833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.056640625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 301.904296875, "completions/mean_terminated_length": 289.28985595703125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24899061187170446, "epoch": 0.009116809116809116, "frac_reward_zero_std": 0.0, "grad_norm": 0.051769740879535675, "kl": 0.001339889792689064, "learning_rate": 4.2613636363636364e-07, "loss": 6.660935468971729e-06, "num_tokens": 3636134.0, "reward": 2.0060060024261475, "reward_std": 0.7957041263580322, "rewards/code_complexity_reward/mean": 0.702929675579071, "rewards/code_complexity_reward/std": 0.2927218973636627, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.435546875, "rewards/code_syntax_reward/std": 0.16771192848682404, "rewards/reasoning_present_reward_func/mean": 0.09238281846046448, "rewards/reasoning_present_reward_func/std": 0.026553234085440636, "rewards/xmlcount_reward_func/mean": 0.433349609375, "rewards/xmlcount_reward_func/std": 0.08906648308038712, "step": 16, "step_time": 56.14584970660508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.083984375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 314.3828125, "completions/mean_terminated_length": 296.264404296875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.26418807730078697, "epoch": 0.009686609686609686, "frac_reward_zero_std": 0.0, "grad_norm": 0.04935988411307335, "kl": 0.0014287417707237182, "learning_rate": 4.5454545454545457e-07, "loss": 7.100752554833889e-06, "num_tokens": 3865618.0, "reward": 1.8187987804412842, "reward_std": 0.8162350058555603, "rewards/code_complexity_reward/mean": 0.6507812738418579, "rewards/code_complexity_reward/std": 0.3320006728172302, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4052734375, "rewards/code_syntax_reward/std": 0.19612568616867065, "rewards/reasoning_present_reward_func/mean": 0.08964844048023224, "rewards/reasoning_present_reward_func/std": 0.030492907389998436, "rewards/xmlcount_reward_func/mean": 0.417236328125, "rewards/xmlcount_reward_func/std": 0.09360432624816895, "step": 17, "step_time": 67.54255327302963 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.060546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 314.2578125, "completions/mean_terminated_length": 301.5135192871094, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.25613101734779775, "epoch": 0.010256410256410256, "frac_reward_zero_std": 0.0, "grad_norm": 0.045039836317300797, "kl": 0.001367473058962787, "learning_rate": 4.829545454545455e-07, "loss": 6.808280886616558e-06, "num_tokens": 4094950.0, "reward": 1.891845703125, "reward_std": 0.8095765113830566, "rewards/code_complexity_reward/mean": 0.6727539300918579, "rewards/code_complexity_reward/std": 0.31256285309791565, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.41796875, "rewards/code_syntax_reward/std": 0.18534722924232483, "rewards/reasoning_present_reward_func/mean": 0.08750000596046448, "rewards/reasoning_present_reward_func/std": 0.03310423716902733, "rewards/xmlcount_reward_func/mean": 0.422607421875, "rewards/xmlcount_reward_func/std": 0.09460954368114471, "step": 18, "step_time": 58.67422782164067 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.091796875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 314.228515625, "completions/mean_terminated_length": 294.23870849609375, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.26919602416455746, "epoch": 0.010826210826210826, "frac_reward_zero_std": 0.0, "grad_norm": 0.058020979166030884, "kl": 0.0013952864101156592, "learning_rate": 5.113636363636364e-07, "loss": 6.926242349436507e-06, "num_tokens": 4326083.0, "reward": 1.8562500476837158, "reward_std": 0.7796268463134766, "rewards/code_complexity_reward/mean": 0.684765636920929, "rewards/code_complexity_reward/std": 0.3180978298187256, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.41796875, "rewards/code_syntax_reward/std": 0.18534722924232483, "rewards/reasoning_present_reward_func/mean": 0.09042969346046448, "rewards/reasoning_present_reward_func/std": 0.02944713830947876, "rewards/xmlcount_reward_func/mean": 0.4150390625, "rewards/xmlcount_reward_func/std": 0.09569192677736282, "step": 19, "step_time": 60.20371696539223 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.095703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 320.646484375, "completions/mean_terminated_length": 300.3952331542969, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.2627502048853785, "epoch": 0.011396011396011397, "frac_reward_zero_std": 0.0, "grad_norm": 0.05363831669092178, "kl": 0.001392078323988244, "learning_rate": 5.397727272727273e-07, "loss": 6.962683983147144e-06, "num_tokens": 4558318.0, "reward": 1.7906250953674316, "reward_std": 0.8358321189880371, "rewards/code_complexity_reward/mean": 0.626660168170929, "rewards/code_complexity_reward/std": 0.3363746702671051, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.400390625, "rewards/code_syntax_reward/std": 0.19990174472332, "rewards/reasoning_present_reward_func/mean": 0.09023437649011612, "rewards/reasoning_present_reward_func/std": 0.029713962227106094, "rewards/xmlcount_reward_func/mean": 0.41748046875, "rewards/xmlcount_reward_func/std": 0.1044606864452362, "step": 20, "step_time": 58.83533804491162 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 342.3203125, "completions/mean_terminated_length": 319.79644775390625, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.2598306315485388, "epoch": 0.011965811965811967, "frac_reward_zero_std": 0.0, "grad_norm": 0.05295226722955704, "kl": 0.0013264370845718076, "learning_rate": 5.681818181818182e-07, "loss": 6.704533006995916e-06, "num_tokens": 4805466.0, "reward": 1.703271508216858, "reward_std": 0.7965804934501648, "rewards/code_complexity_reward/mean": 0.6087890863418579, "rewards/code_complexity_reward/std": 0.3349955379962921, "rewards/code_execution_reward/mean": 0.19921875, "rewards/code_execution_reward/std": 0.39980348944664, "rewards/code_syntax_reward/mean": 0.3935546875, "rewards/code_syntax_reward/std": 0.204875648021698, "rewards/reasoning_present_reward_func/mean": 0.087890625, "rewards/reasoning_present_reward_func/std": 0.03265552595257759, "rewards/xmlcount_reward_func/mean": 0.413818359375, "rewards/xmlcount_reward_func/std": 0.1028018444776535, "step": 21, "step_time": 58.99625431280583 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 302.958984375, "completions/mean_terminated_length": 286.2004089355469, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2666494995355606, "epoch": 0.012535612535612535, "frac_reward_zero_std": 0.0, "grad_norm": 0.05174759402871132, "kl": 0.0014435275079449639, "learning_rate": 5.965909090909092e-07, "loss": 7.221151463454589e-06, "num_tokens": 5030405.0, "reward": 1.872802734375, "reward_std": 0.779346764087677, "rewards/code_complexity_reward/mean": 0.6943359375, "rewards/code_complexity_reward/std": 0.31320756673812866, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.421875, "rewards/code_syntax_reward/std": 0.18172365427017212, "rewards/reasoning_present_reward_func/mean": 0.0888671875, "rewards/reasoning_present_reward_func/std": 0.03148456662893295, "rewards/xmlcount_reward_func/mean": 0.423583984375, "rewards/xmlcount_reward_func/std": 0.0991731807589531, "step": 22, "step_time": 87.48433318920434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.091796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 305.427734375, "completions/mean_terminated_length": 284.54840087890625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2497819666750729, "epoch": 0.013105413105413105, "frac_reward_zero_std": 0.0, "grad_norm": 0.0511535108089447, "kl": 0.0013300326081662206, "learning_rate": 6.25e-07, "loss": 6.72325404593721e-06, "num_tokens": 5255632.0, "reward": 1.940185546875, "reward_std": 0.8169589042663574, "rewards/code_complexity_reward/mean": 0.67578125, "rewards/code_complexity_reward/std": 0.3082038462162018, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.423828125, "rewards/code_syntax_reward/std": 0.17985260486602783, "rewards/reasoning_present_reward_func/mean": 0.0927734375, "rewards/reasoning_present_reward_func/std": 0.02591804414987564, "rewards/xmlcount_reward_func/mean": 0.425537109375, "rewards/xmlcount_reward_func/std": 0.09534649550914764, "step": 23, "step_time": 60.900769595988095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.068359375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 313.265625, "completions/mean_terminated_length": 298.6834411621094, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2782149026170373, "epoch": 0.013675213675213675, "frac_reward_zero_std": 0.0, "grad_norm": 0.04802826792001724, "kl": 0.001408852143867989, "learning_rate": 6.534090909090909e-07, "loss": 7.035527232801542e-06, "num_tokens": 5486736.0, "reward": 1.9140625, "reward_std": 0.7891928553581238, "rewards/code_complexity_reward/mean": 0.6931641101837158, "rewards/code_complexity_reward/std": 0.3038599491119385, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4267578125, "rewards/code_syntax_reward/std": 0.17696848511695862, "rewards/reasoning_present_reward_func/mean": 0.08906250447034836, "rewards/reasoning_present_reward_func/std": 0.031241435557603836, "rewards/xmlcount_reward_func/mean": 0.427734375, "rewards/xmlcount_reward_func/std": 0.09591634571552277, "step": 24, "step_time": 60.012970050796866 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 300.625, "completions/mean_terminated_length": 286.5333557128906, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2625638411846012, "epoch": 0.014245014245014245, "frac_reward_zero_std": 0.0, "grad_norm": 0.052284836769104004, "kl": 0.001389131276482658, "learning_rate": 6.818181818181818e-07, "loss": 6.900751031935215e-06, "num_tokens": 5707136.0, "reward": 1.9893065690994263, "reward_std": 0.8342277407646179, "rewards/code_complexity_reward/mean": 0.69384765625, "rewards/code_complexity_reward/std": 0.31411078572273254, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4228515625, "rewards/code_syntax_reward/std": 0.18079319596290588, "rewards/reasoning_present_reward_func/mean": 0.09160156548023224, "rewards/reasoning_present_reward_func/std": 0.02776356413960457, "rewards/xmlcount_reward_func/mean": 0.427490234375, "rewards/xmlcount_reward_func/std": 0.09030769020318985, "step": 25, "step_time": 62.53582011722028 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.107421875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 317.41796875, "completions/mean_terminated_length": 294.0, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.2575743973720819, "epoch": 0.014814814814814815, "frac_reward_zero_std": 0.0, "grad_norm": 0.04940791428089142, "kl": 0.0013441207393043442, "learning_rate": 7.102272727272729e-07, "loss": 6.7132159529137425e-06, "num_tokens": 5937918.0, "reward": 1.7803223133087158, "reward_std": 0.8341856598854065, "rewards/code_complexity_reward/mean": 0.6280273199081421, "rewards/code_complexity_reward/std": 0.33635058999061584, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.3974609375, "rewards/code_syntax_reward/std": 0.2020767778158188, "rewards/reasoning_present_reward_func/mean": 0.08906249701976776, "rewards/reasoning_present_reward_func/std": 0.031241437420248985, "rewards/xmlcount_reward_func/mean": 0.415771484375, "rewards/xmlcount_reward_func/std": 0.0952211394906044, "step": 26, "step_time": 59.76681778859347 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.087890625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 314.3984375, "completions/mean_terminated_length": 295.35760498046875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2664252493996173, "epoch": 0.015384615384615385, "frac_reward_zero_std": 0.0, "grad_norm": 0.048807259649038315, "kl": 0.0013952697681816062, "learning_rate": 7.386363636363638e-07, "loss": 6.927381036803126e-06, "num_tokens": 6166338.0, "reward": 1.8597655296325684, "reward_std": 0.8092495203018188, "rewards/code_complexity_reward/mean": 0.675097644329071, "rewards/code_complexity_reward/std": 0.3240068256855011, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4130859375, "rewards/code_syntax_reward/std": 0.18966612219810486, "rewards/reasoning_present_reward_func/mean": 0.08750000596046448, "rewards/reasoning_present_reward_func/std": 0.03310423716902733, "rewards/xmlcount_reward_func/mean": 0.41455078125, "rewards/xmlcount_reward_func/std": 0.10207338631153107, "step": 27, "step_time": 79.47922466229647 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 308.208984375, "completions/mean_terminated_length": 293.71337890625, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.257618376519531, "epoch": 0.015954415954415956, "frac_reward_zero_std": 0.0, "grad_norm": 0.05753986909985542, "kl": 0.0013764404629910132, "learning_rate": 7.670454545454547e-07, "loss": 6.911170203238726e-06, "num_tokens": 6390909.0, "reward": 1.9500975608825684, "reward_std": 0.856502115726471, "rewards/code_complexity_reward/mean": 0.671679675579071, "rewards/code_complexity_reward/std": 0.3193969130516052, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4140625, "rewards/code_syntax_reward/std": 0.18882036209106445, "rewards/reasoning_present_reward_func/mean": 0.09140624850988388, "rewards/reasoning_present_reward_func/std": 0.028054583817720413, "rewards/xmlcount_reward_func/mean": 0.41748046875, "rewards/xmlcount_reward_func/std": 0.10209210216999054, "step": 28, "step_time": 56.89560460392386 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 308.666015625, "completions/mean_terminated_length": 290.4957275390625, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.2641605583485216, "epoch": 0.016524216524216526, "frac_reward_zero_std": 0.0, "grad_norm": 0.05316847935318947, "kl": 0.0013997559290146455, "learning_rate": 7.954545454545455e-07, "loss": 6.978298188187182e-06, "num_tokens": 6615914.0, "reward": 1.895751953125, "reward_std": 0.8379237651824951, "rewards/code_complexity_reward/mean": 0.6720702648162842, "rewards/code_complexity_reward/std": 0.3287752568721771, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4091796875, "rewards/code_syntax_reward/std": 0.19296257197856903, "rewards/reasoning_present_reward_func/mean": 0.09160156548023224, "rewards/reasoning_present_reward_func/std": 0.02776356413960457, "rewards/xmlcount_reward_func/mean": 0.426025390625, "rewards/xmlcount_reward_func/std": 0.09280723333358765, "step": 29, "step_time": 72.20266657881439 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.068359375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 305.203125, "completions/mean_terminated_length": 290.0293273925781, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2580829500220716, "epoch": 0.017094017094017096, "frac_reward_zero_std": 0.0, "grad_norm": 0.05144109949469566, "kl": 0.0013709424492844846, "learning_rate": 8.238636363636364e-07, "loss": 6.83543476043269e-06, "num_tokens": 6839914.0, "reward": 1.9158692359924316, "reward_std": 0.8085138201713562, "rewards/code_complexity_reward/mean": 0.68896484375, "rewards/code_complexity_reward/std": 0.31740185618400574, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.419921875, "rewards/code_syntax_reward/std": 0.1835547834634781, "rewards/reasoning_present_reward_func/mean": 0.09140625596046448, "rewards/reasoning_present_reward_func/std": 0.028054581955075264, "rewards/xmlcount_reward_func/mean": 0.428466796875, "rewards/xmlcount_reward_func/std": 0.0937318429350853, "step": 30, "step_time": 68.26877622678876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 287.193359375, "completions/mean_terminated_length": 279.0020446777344, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2594741266220808, "epoch": 0.017663817663817662, "frac_reward_zero_std": 0.0, "grad_norm": 0.06052260473370552, "kl": 0.001410011888765439, "learning_rate": 8.522727272727273e-07, "loss": 7.053662557154894e-06, "num_tokens": 7057653.0, "reward": 2.0079588890075684, "reward_std": 0.793811023235321, "rewards/code_complexity_reward/mean": 0.717578113079071, "rewards/code_complexity_reward/std": 0.2881949841976166, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4384765625, "rewards/code_syntax_reward/std": 0.16440613567829132, "rewards/reasoning_present_reward_func/mean": 0.08847656100988388, "rewards/reasoning_present_reward_func/std": 0.03196168690919876, "rewards/xmlcount_reward_func/mean": 0.419677734375, "rewards/xmlcount_reward_func/std": 0.10368571430444717, "step": 31, "step_time": 70.6590262344107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 316.16796875, "completions/mean_terminated_length": 300.4683532714844, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.26640998874790967, "epoch": 0.018233618233618232, "frac_reward_zero_std": 0.0, "grad_norm": 0.0543670579791069, "kl": 0.0013626684340124484, "learning_rate": 8.806818181818183e-07, "loss": 6.75742921885103e-06, "num_tokens": 7290579.0, "reward": 1.8778809309005737, "reward_std": 0.8092921376228333, "rewards/code_complexity_reward/mean": 0.6678711175918579, "rewards/code_complexity_reward/std": 0.3184446692466736, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4150390625, "rewards/code_syntax_reward/std": 0.1879657357931137, "rewards/reasoning_present_reward_func/mean": 0.09062500298023224, "rewards/reasoning_present_reward_func/std": 0.029176566749811172, "rewards/xmlcount_reward_func/mean": 0.427001953125, "rewards/xmlcount_reward_func/std": 0.09583879262208939, "step": 32, "step_time": 59.89162210468203 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.091796875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 314.890625, "completions/mean_terminated_length": 294.9677429199219, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.25332422833889723, "epoch": 0.018803418803418803, "frac_reward_zero_std": 0.0, "grad_norm": 0.05403255671262741, "kl": 0.001345902940556698, "learning_rate": 9.090909090909091e-07, "loss": 6.796908564865589e-06, "num_tokens": 7520467.0, "reward": 1.8452637195587158, "reward_std": 0.8669067025184631, "rewards/code_complexity_reward/mean": 0.64013671875, "rewards/code_complexity_reward/std": 0.34099453687667847, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.3984375, "rewards/code_syntax_reward/std": 0.2013591229915619, "rewards/reasoning_present_reward_func/mean": 0.09062500298023224, "rewards/reasoning_present_reward_func/std": 0.029176566749811172, "rewards/xmlcount_reward_func/mean": 0.421142578125, "rewards/xmlcount_reward_func/std": 0.09565416723489761, "step": 33, "step_time": 58.70621662028134 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 301.7109375, "completions/mean_terminated_length": 287.6916809082031, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2652526777237654, "epoch": 0.019373219373219373, "frac_reward_zero_std": 0.0, "grad_norm": 0.05474543944001198, "kl": 0.0013906538670198643, "learning_rate": 9.375000000000001e-07, "loss": 6.898795163579052e-06, "num_tokens": 7741847.0, "reward": 1.945068359375, "reward_std": 0.7918501496315002, "rewards/code_complexity_reward/mean": 0.6905273199081421, "rewards/code_complexity_reward/std": 0.30011922121047974, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4296875, "rewards/code_syntax_reward/std": 0.17398715019226074, "rewards/reasoning_present_reward_func/mean": 0.09023437649011612, "rewards/reasoning_present_reward_func/std": 0.029713962227106094, "rewards/xmlcount_reward_func/mean": 0.426025390625, "rewards/xmlcount_reward_func/std": 0.09411589056253433, "step": 34, "step_time": 69.69247165229172 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.083984375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 316.638671875, "completions/mean_terminated_length": 298.7270812988281, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.26719448529183865, "epoch": 0.019943019943019943, "frac_reward_zero_std": 0.0, "grad_norm": 0.05086144804954529, "kl": 0.0014116868514975067, "learning_rate": 9.65909090909091e-07, "loss": 7.006732630543411e-06, "num_tokens": 7975710.0, "reward": 1.8544433116912842, "reward_std": 0.8121605515480042, "rewards/code_complexity_reward/mean": 0.666015625, "rewards/code_complexity_reward/std": 0.3235948979854584, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.41015625, "rewards/code_syntax_reward/std": 0.19215121865272522, "rewards/reasoning_present_reward_func/mean": 0.08613281697034836, "rewards/reasoning_present_reward_func/std": 0.034594181925058365, "rewards/xmlcount_reward_func/mean": 0.416748046875, "rewards/xmlcount_reward_func/std": 0.09671220183372498, "step": 35, "step_time": 68.39520323742181 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.087890625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 318.625, "completions/mean_terminated_length": 299.9914245605469, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.2591923871077597, "epoch": 0.020512820512820513, "frac_reward_zero_std": 0.0, "grad_norm": 0.04890437051653862, "kl": 0.0013767741611445672, "learning_rate": 9.943181818181819e-07, "loss": 6.828631740063429e-06, "num_tokens": 8206558.0, "reward": 1.86572265625, "reward_std": 0.8153459429740906, "rewards/code_complexity_reward/mean": 0.676074206829071, "rewards/code_complexity_reward/std": 0.324110746383667, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4130859375, "rewards/code_syntax_reward/std": 0.18966612219810486, "rewards/reasoning_present_reward_func/mean": 0.09003906697034836, "rewards/reasoning_present_reward_func/std": 0.029977135360240936, "rewards/xmlcount_reward_func/mean": 0.4189453125, "rewards/xmlcount_reward_func/std": 0.09809815883636475, "step": 36, "step_time": 68.70904584322125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 300.689453125, "completions/mean_terminated_length": 287.537353515625, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.25419761799275875, "epoch": 0.021082621082621083, "frac_reward_zero_std": 0.0, "grad_norm": 0.0625494122505188, "kl": 0.0013431346278593992, "learning_rate": 1.0227272727272729e-06, "loss": 6.814370863139629e-06, "num_tokens": 8429863.0, "reward": 1.9866697788238525, "reward_std": 0.8177380561828613, "rewards/code_complexity_reward/mean": 0.693066418170929, "rewards/code_complexity_reward/std": 0.30200856924057007, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4296875, "rewards/code_syntax_reward/std": 0.17398715019226074, "rewards/reasoning_present_reward_func/mean": 0.09121093899011612, "rewards/reasoning_present_reward_func/std": 0.028341269120573997, "rewards/xmlcount_reward_func/mean": 0.427001953125, "rewards/xmlcount_reward_func/std": 0.09059029072523117, "step": 37, "step_time": 52.04918388091028 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.107421875, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 317.931640625, "completions/mean_terminated_length": 294.57550048828125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2753692422993481, "epoch": 0.021652421652421653, "frac_reward_zero_std": 0.0, "grad_norm": 0.051071327179670334, "kl": 0.0014111957480054116, "learning_rate": 1.0511363636363639e-06, "loss": 7.0348032750189304e-06, "num_tokens": 8661780.0, "reward": 1.7648437023162842, "reward_std": 0.8336398601531982, "rewards/code_complexity_reward/mean": 0.6332031488418579, "rewards/code_complexity_reward/std": 0.3505811393260956, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.390625, "rewards/code_syntax_reward/std": 0.20690147578716278, "rewards/reasoning_present_reward_func/mean": 0.09062500298023224, "rewards/reasoning_present_reward_func/std": 0.02917656861245632, "rewards/xmlcount_reward_func/mean": 0.4140625, "rewards/xmlcount_reward_func/std": 0.09672979265451431, "step": 38, "step_time": 72.19301033392549 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 312.189453125, "completions/mean_terminated_length": 291.5194091796875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2673113294877112, "epoch": 0.022222222222222223, "frac_reward_zero_std": 0.0, "grad_norm": 0.05567771941423416, "kl": 0.0014383413463292527, "learning_rate": 1.0795454545454546e-06, "loss": 7.194932550191879e-06, "num_tokens": 8892357.0, "reward": 1.862158179283142, "reward_std": 0.8464499115943909, "rewards/code_complexity_reward/mean": 0.6664062738418579, "rewards/code_complexity_reward/std": 0.3336396813392639, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.40625, "rewards/code_syntax_reward/std": 0.19534705579280853, "rewards/reasoning_present_reward_func/mean": 0.08808593451976776, "rewards/reasoning_present_reward_func/std": 0.032427072525024414, "rewards/xmlcount_reward_func/mean": 0.420166015625, "rewards/xmlcount_reward_func/std": 0.09706969559192657, "step": 39, "step_time": 52.493666449561715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 318.125, "completions/mean_terminated_length": 299.8974609375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.27600690140388906, "epoch": 0.022792022792022793, "frac_reward_zero_std": 0.0, "grad_norm": 0.04926905781030655, "kl": 0.0014277613227022812, "learning_rate": 1.1079545454545456e-06, "loss": 7.127848220989108e-06, "num_tokens": 9124765.0, "reward": 1.798828125, "reward_std": 0.7781310081481934, "rewards/code_complexity_reward/mean": 0.6653320789337158, "rewards/code_complexity_reward/std": 0.3239831328392029, "rewards/code_execution_reward/mean": 0.21875, "rewards/code_execution_reward/std": 0.41380295157432556, "rewards/code_syntax_reward/mean": 0.4111328125, "rewards/code_syntax_reward/std": 0.1913314312696457, "rewards/reasoning_present_reward_func/mean": 0.08906249701976776, "rewards/reasoning_present_reward_func/std": 0.031241439282894135, "rewards/xmlcount_reward_func/mean": 0.41455078125, "rewards/xmlcount_reward_func/std": 0.09525531530380249, "step": 40, "step_time": 68.36409596540034 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 304.583984375, "completions/mean_terminated_length": 293.4876403808594, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.2601953789126128, "epoch": 0.023361823361823363, "frac_reward_zero_std": 0.0, "grad_norm": 0.04449097067117691, "kl": 0.001366795455396641, "learning_rate": 1.1363636363636364e-06, "loss": 6.782043783459812e-06, "num_tokens": 9346456.0, "reward": 1.8865234851837158, "reward_std": 0.793371856212616, "rewards/code_complexity_reward/mean": 0.673144519329071, "rewards/code_complexity_reward/std": 0.31292521953582764, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.41796875, "rewards/code_syntax_reward/std": 0.18534722924232483, "rewards/reasoning_present_reward_func/mean": 0.0947265625, "rewards/reasoning_present_reward_func/std": 0.022372130304574966, "rewards/xmlcount_reward_func/mean": 0.43701171875, "rewards/xmlcount_reward_func/std": 0.08566395193338394, "step": 41, "step_time": 76.85227984003723 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.072265625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 306.958984375, "completions/mean_terminated_length": 290.98736572265625, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.26367646642029285, "epoch": 0.023931623931623933, "frac_reward_zero_std": 0.0, "grad_norm": 0.05667397752404213, "kl": 0.0014068803257032414, "learning_rate": 1.1647727272727274e-06, "loss": 6.941478204680607e-06, "num_tokens": 9572139.0, "reward": 1.878759741783142, "reward_std": 0.8013209700584412, "rewards/code_complexity_reward/mean": 0.6799805164337158, "rewards/code_complexity_reward/std": 0.3194671869277954, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.416015625, "rewards/code_syntax_reward/std": 0.1871020793914795, "rewards/reasoning_present_reward_func/mean": 0.08964844048023224, "rewards/reasoning_present_reward_func/std": 0.030492907389998436, "rewards/xmlcount_reward_func/mean": 0.419677734375, "rewards/xmlcount_reward_func/std": 0.08559246361255646, "step": 42, "step_time": 68.07871737051755 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.068359375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 306.017578125, "completions/mean_terminated_length": 290.903564453125, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.26270213816314936, "epoch": 0.0245014245014245, "frac_reward_zero_std": 0.0, "grad_norm": 0.052688732743263245, "kl": 0.0013867698835383635, "learning_rate": 1.1931818181818183e-06, "loss": 6.9533707574009895e-06, "num_tokens": 9797316.0, "reward": 1.941992163658142, "reward_std": 0.8073244690895081, "rewards/code_complexity_reward/mean": 0.6886719465255737, "rewards/code_complexity_reward/std": 0.3019159138202667, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.427734375, "rewards/code_syntax_reward/std": 0.17598573863506317, "rewards/reasoning_present_reward_func/mean": 0.08632812649011612, "rewards/reasoning_present_reward_func/std": 0.034388620406389236, "rewards/xmlcount_reward_func/mean": 0.4228515625, "rewards/xmlcount_reward_func/std": 0.09875830262899399, "step": 43, "step_time": 86.09883480612189 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.080078125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 304.08984375, "completions/mean_terminated_length": 285.99151611328125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.260829437058419, "epoch": 0.02507122507122507, "frac_reward_zero_std": 0.0, "grad_norm": 0.046153534203767776, "kl": 0.0013558806203946006, "learning_rate": 1.2215909090909091e-06, "loss": 6.7365472204983234e-06, "num_tokens": 10022322.0, "reward": 1.96142578125, "reward_std": 0.7819585204124451, "rewards/code_complexity_reward/mean": 0.713183581829071, "rewards/code_complexity_reward/std": 0.29847460985183716, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4345703125, "rewards/code_syntax_reward/std": 0.16878816485404968, "rewards/reasoning_present_reward_func/mean": 0.08906249701976776, "rewards/reasoning_present_reward_func/std": 0.031241437420248985, "rewards/xmlcount_reward_func/mean": 0.427734375, "rewards/xmlcount_reward_func/std": 0.0965518206357956, "step": 44, "step_time": 58.8420002926141 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 304.693359375, "completions/mean_terminated_length": 289.01470947265625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.26614307845011353, "epoch": 0.02564102564102564, "frac_reward_zero_std": 0.0, "grad_norm": 0.056763797998428345, "kl": 0.001390228773743729, "learning_rate": 1.25e-06, "loss": 6.991162081249058e-06, "num_tokens": 10246461.0, "reward": 1.87451171875, "reward_std": 0.8138747215270996, "rewards/code_complexity_reward/mean": 0.6676757335662842, "rewards/code_complexity_reward/std": 0.3211328387260437, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.412109375, "rewards/code_syntax_reward/std": 0.1905031055212021, "rewards/reasoning_present_reward_func/mean": 0.09160156548023224, "rewards/reasoning_present_reward_func/std": 0.02776356413960457, "rewards/xmlcount_reward_func/mean": 0.42578125, "rewards/xmlcount_reward_func/std": 0.09441018104553223, "step": 45, "step_time": 62.0290910275653 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.064453125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 315.78125, "completions/mean_terminated_length": 302.2630615234375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.26571162743493915, "epoch": 0.02621082621082621, "frac_reward_zero_std": 0.0, "grad_norm": 0.04046618938446045, "kl": 0.0013836860553055885, "learning_rate": 1.278409090909091e-06, "loss": 6.8985100369900465e-06, "num_tokens": 10476741.0, "reward": 1.8587403297424316, "reward_std": 0.7475959658622742, "rewards/code_complexity_reward/mean": 0.695507824420929, "rewards/code_complexity_reward/std": 0.3060053288936615, "rewards/code_execution_reward/mean": 0.2265625, "rewards/code_execution_reward/std": 0.4190165400505066, "rewards/code_syntax_reward/mean": 0.42578125, "rewards/code_syntax_reward/std": 0.17794041335582733, "rewards/reasoning_present_reward_func/mean": 0.08828125894069672, "rewards/reasoning_present_reward_func/std": 0.032195813953876495, "rewards/xmlcount_reward_func/mean": 0.422607421875, "rewards/xmlcount_reward_func/std": 0.09747491031885147, "step": 46, "step_time": 71.3558979826048 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 310.490234375, "completions/mean_terminated_length": 292.48297119140625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2664824374951422, "epoch": 0.02678062678062678, "frac_reward_zero_std": 0.0, "grad_norm": 0.05394879728555679, "kl": 0.0014261453125072876, "learning_rate": 1.3068181818181819e-06, "loss": 7.125898264348507e-06, "num_tokens": 10707120.0, "reward": 1.8512694835662842, "reward_std": 0.8110331892967224, "rewards/code_complexity_reward/mean": 0.6636718511581421, "rewards/code_complexity_reward/std": 0.3253006935119629, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.41015625, "rewards/code_syntax_reward/std": 0.19215121865272522, "rewards/reasoning_present_reward_func/mean": 0.08847656846046448, "rewards/reasoning_present_reward_func/std": 0.03196168690919876, "rewards/xmlcount_reward_func/mean": 0.41943359375, "rewards/xmlcount_reward_func/std": 0.090744748711586, "step": 47, "step_time": 52.73563615605235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 318.068359375, "completions/mean_terminated_length": 296.1456298828125, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2591571651864797, "epoch": 0.02735042735042735, "frac_reward_zero_std": 0.0, "grad_norm": 0.04637562483549118, "kl": 0.0013496231822500704, "learning_rate": 1.3352272727272728e-06, "loss": 6.8267108872532845e-06, "num_tokens": 10938931.0, "reward": 1.892822265625, "reward_std": 0.8240786194801331, "rewards/code_complexity_reward/mean": 0.67822265625, "rewards/code_complexity_reward/std": 0.323885053396225, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4150390625, "rewards/code_syntax_reward/std": 0.1879657357931137, "rewards/reasoning_present_reward_func/mean": 0.08984375, "rewards/reasoning_present_reward_func/std": 0.030236754566431046, "rewards/xmlcount_reward_func/mean": 0.420654296875, "rewards/xmlcount_reward_func/std": 0.1020672395825386, "step": 48, "step_time": 80.29911544453353 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 312.81640625, "completions/mean_terminated_length": 293.1545104980469, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2654684134759009, "epoch": 0.02792022792022792, "frac_reward_zero_std": 0.0, "grad_norm": 0.05802157148718834, "kl": 0.0014295550990937045, "learning_rate": 1.3636363636363636e-06, "loss": 7.171824108809233e-06, "num_tokens": 11168813.0, "reward": 1.797998070716858, "reward_std": 0.830233633518219, "rewards/code_complexity_reward/mean": 0.6437499523162842, "rewards/code_complexity_reward/std": 0.3399665951728821, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.3984375, "rewards/code_syntax_reward/std": 0.2013591229915619, "rewards/reasoning_present_reward_func/mean": 0.08906249701976776, "rewards/reasoning_present_reward_func/std": 0.031241437420248985, "rewards/xmlcount_reward_func/mean": 0.420654296875, "rewards/xmlcount_reward_func/std": 0.09362984448671341, "step": 49, "step_time": 57.500593579374254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 302.158203125, "completions/mean_terminated_length": 285.3354187011719, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2633062850218266, "epoch": 0.02849002849002849, "frac_reward_zero_std": 0.0, "grad_norm": 0.0461699478328228, "kl": 0.0014294228658400243, "learning_rate": 1.3920454545454546e-06, "loss": 7.150723831728101e-06, "num_tokens": 11393694.0, "reward": 1.9124023914337158, "reward_std": 0.8438594341278076, "rewards/code_complexity_reward/mean": 0.671679675579071, "rewards/code_complexity_reward/std": 0.3223244845867157, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4150390625, "rewards/code_syntax_reward/std": 0.1879657357931137, "rewards/reasoning_present_reward_func/mean": 0.0908203125, "rewards/reasoning_present_reward_func/std": 0.028902139514684677, "rewards/xmlcount_reward_func/mean": 0.42431640625, "rewards/xmlcount_reward_func/std": 0.09927339851856232, "step": 50, "step_time": 100.08429619390517 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 305.076171875, "completions/mean_terminated_length": 291.28125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.25728040863759816, "epoch": 0.02905982905982906, "frac_reward_zero_std": 0.0, "grad_norm": 0.05728168785572052, "kl": 0.0013801831310047419, "learning_rate": 1.4204545454545458e-06, "loss": 6.83922553434968e-06, "num_tokens": 11618437.0, "reward": 1.8900878429412842, "reward_std": 0.8223906755447388, "rewards/code_complexity_reward/mean": 0.6698242425918579, "rewards/code_complexity_reward/std": 0.3214792311191559, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4130859375, "rewards/code_syntax_reward/std": 0.18966612219810486, "rewards/reasoning_present_reward_func/mean": 0.08964844048023224, "rewards/reasoning_present_reward_func/std": 0.030492907389998436, "rewards/xmlcount_reward_func/mean": 0.430419921875, "rewards/xmlcount_reward_func/std": 0.0888812243938446, "step": 51, "step_time": 62.23266142513603 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 317.505859375, "completions/mean_terminated_length": 300.1255187988281, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.2630810646805912, "epoch": 0.02962962962962963, "frac_reward_zero_std": 0.0, "grad_norm": 0.050589386373758316, "kl": 0.0014329350078696734, "learning_rate": 1.4488636363636366e-06, "loss": 7.145717972889543e-06, "num_tokens": 11852408.0, "reward": 1.835351586341858, "reward_std": 0.7896996140480042, "rewards/code_complexity_reward/mean": 0.6653320789337158, "rewards/code_complexity_reward/std": 0.3101263642311096, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.41796875, "rewards/code_syntax_reward/std": 0.18534722924232483, "rewards/reasoning_present_reward_func/mean": 0.09042969346046448, "rewards/reasoning_present_reward_func/std": 0.02944713830947876, "rewards/xmlcount_reward_func/mean": 0.42333984375, "rewards/xmlcount_reward_func/std": 0.09820889681577682, "step": 52, "step_time": 60.688370710238814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 328.3203125, "completions/mean_terminated_length": 303.93804931640625, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2746072269510478, "epoch": 0.0301994301994302, "frac_reward_zero_std": 0.0, "grad_norm": 0.04282652214169502, "kl": 0.0014335680716612842, "learning_rate": 1.4772727272727275e-06, "loss": 7.182476110756397e-06, "num_tokens": 12091596.0, "reward": 1.775781273841858, "reward_std": 0.8252152800559998, "rewards/code_complexity_reward/mean": 0.6375976800918579, "rewards/code_complexity_reward/std": 0.3408827483654022, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.396484375, "rewards/code_syntax_reward/std": 0.20278719067573547, "rewards/reasoning_present_reward_func/mean": 0.08984375, "rewards/reasoning_present_reward_func/std": 0.030236754566431046, "rewards/xmlcount_reward_func/mean": 0.41943359375, "rewards/xmlcount_reward_func/std": 0.09850035607814789, "step": 53, "step_time": 50.550715452991426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 290.091796875, "completions/mean_terminated_length": 281.0711364746094, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24860157677903771, "epoch": 0.03076923076923077, "frac_reward_zero_std": 0.0, "grad_norm": 0.05378542095422745, "kl": 0.0013377694949667784, "learning_rate": 1.5056818181818183e-06, "loss": 6.799018592573702e-06, "num_tokens": 12307075.0, "reward": 1.959619164466858, "reward_std": 0.7969895601272583, "rewards/code_complexity_reward/mean": 0.7052733898162842, "rewards/code_complexity_reward/std": 0.30592888593673706, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.427734375, "rewards/code_syntax_reward/std": 0.17598573863506317, "rewards/reasoning_present_reward_func/mean": 0.09101562201976776, "rewards/reasoning_present_reward_func/std": 0.02862374298274517, "rewards/xmlcount_reward_func/mean": 0.428955078125, "rewards/xmlcount_reward_func/std": 0.08977974951267242, "step": 54, "step_time": 52.43653515074402 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 313.09765625, "completions/mean_terminated_length": 298.05462646484375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2711168983951211, "epoch": 0.03133903133903134, "frac_reward_zero_std": 0.0, "grad_norm": 0.06520966440439224, "kl": 0.001458430460843374, "learning_rate": 1.5340909090909093e-06, "loss": 7.287075277417898e-06, "num_tokens": 12534621.0, "reward": 1.8691893815994263, "reward_std": 0.8449352383613586, "rewards/code_complexity_reward/mean": 0.6636718511581421, "rewards/code_complexity_reward/std": 0.32880109548568726, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.408203125, "rewards/code_syntax_reward/std": 0.1937655806541443, "rewards/reasoning_present_reward_func/mean": 0.09003906697034836, "rewards/reasoning_present_reward_func/std": 0.029977135360240936, "rewards/xmlcount_reward_func/mean": 0.416259765625, "rewards/xmlcount_reward_func/std": 0.09723690152168274, "step": 55, "step_time": 58.24108223896474 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 312.572265625, "completions/mean_terminated_length": 293.8226623535156, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.26452045841142535, "epoch": 0.03190883190883191, "frac_reward_zero_std": 0.0, "grad_norm": 0.05252932757139206, "kl": 0.0013930439054092858, "learning_rate": 1.5625e-06, "loss": 6.941176252439618e-06, "num_tokens": 12764746.0, "reward": 1.938867211341858, "reward_std": 0.8160993456840515, "rewards/code_complexity_reward/mean": 0.686230480670929, "rewards/code_complexity_reward/std": 0.3129163980484009, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4208984375, "rewards/code_syntax_reward/std": 0.18264412879943848, "rewards/reasoning_present_reward_func/mean": 0.09003906697034836, "rewards/reasoning_present_reward_func/std": 0.029977135360240936, "rewards/xmlcount_reward_func/mean": 0.42529296875, "rewards/xmlcount_reward_func/std": 0.09238316863775253, "step": 56, "step_time": 57.799876500852406 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 320.802734375, "completions/mean_terminated_length": 299.1891174316406, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2665687669068575, "epoch": 0.03247863247863248, "frac_reward_zero_std": 0.0, "grad_norm": 0.0511140450835228, "kl": 0.0014152163166727405, "learning_rate": 1.590909090909091e-06, "loss": 7.058493793010712e-06, "num_tokens": 13002157.0, "reward": 1.8437988758087158, "reward_std": 0.8332509398460388, "rewards/code_complexity_reward/mean": 0.6590820550918579, "rewards/code_complexity_reward/std": 0.3359866440296173, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4052734375, "rewards/code_syntax_reward/std": 0.19612568616867065, "rewards/reasoning_present_reward_func/mean": 0.08925781399011612, "rewards/reasoning_present_reward_func/std": 0.030995169654488564, "rewards/xmlcount_reward_func/mean": 0.424560546875, "rewards/xmlcount_reward_func/std": 0.09553921222686768, "step": 57, "step_time": 71.32268591038883 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.099609375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 324.923828125, "completions/mean_terminated_length": 304.227783203125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.26937581109814346, "epoch": 0.03304843304843305, "frac_reward_zero_std": 0.0, "grad_norm": 0.051188986748456955, "kl": 0.001426357839591219, "learning_rate": 1.6193181818181818e-06, "loss": 7.123278919607401e-06, "num_tokens": 13238190.0, "reward": 1.865234375, "reward_std": 0.8112602233886719, "rewards/code_complexity_reward/mean": 0.666210949420929, "rewards/code_complexity_reward/std": 0.3176954984664917, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4169921875, "rewards/code_syntax_reward/std": 0.18622928857803345, "rewards/reasoning_present_reward_func/mean": 0.08867187798023224, "rewards/reasoning_present_reward_func/std": 0.03172462433576584, "rewards/xmlcount_reward_func/mean": 0.423828125, "rewards/xmlcount_reward_func/std": 0.1040218323469162, "step": 58, "step_time": 61.56194949243218 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08984375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 313.041015625, "completions/mean_terminated_length": 293.4012756347656, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.25769022572785616, "epoch": 0.03361823361823362, "frac_reward_zero_std": 0.0, "grad_norm": 0.060627784579992294, "kl": 0.0013949869135103654, "learning_rate": 1.6477272727272728e-06, "loss": 6.935035344213247e-06, "num_tokens": 13468875.0, "reward": 1.8913573026657104, "reward_std": 0.8295571804046631, "rewards/code_complexity_reward/mean": 0.67431640625, "rewards/code_complexity_reward/std": 0.32512158155441284, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.412109375, "rewards/code_syntax_reward/std": 0.1905031055212021, "rewards/reasoning_present_reward_func/mean": 0.0908203125, "rewards/reasoning_present_reward_func/std": 0.028902139514684677, "rewards/xmlcount_reward_func/mean": 0.428955078125, "rewards/xmlcount_reward_func/std": 0.09246426075696945, "step": 59, "step_time": 52.7031503431499 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.072265625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 310.61328125, "completions/mean_terminated_length": 294.9263000488281, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.2629453882109374, "epoch": 0.03418803418803419, "frac_reward_zero_std": 0.0, "grad_norm": 0.03988077864050865, "kl": 0.0014146594730846118, "learning_rate": 1.6761363636363636e-06, "loss": 7.0828828029334545e-06, "num_tokens": 13695357.0, "reward": 1.9712891578674316, "reward_std": 0.8047458529472351, "rewards/code_complexity_reward/mean": 0.69775390625, "rewards/code_complexity_reward/std": 0.30011385679244995, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.427734375, "rewards/code_syntax_reward/std": 0.17598573863506317, "rewards/reasoning_present_reward_func/mean": 0.08945313096046448, "rewards/reasoning_present_reward_func/std": 0.03074568696320057, "rewards/xmlcount_reward_func/mean": 0.42626953125, "rewards/xmlcount_reward_func/std": 0.09575556963682175, "step": 60, "step_time": 83.21608558855951 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.107421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 313.361328125, "completions/mean_terminated_length": 289.45513916015625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.25649128668010235, "epoch": 0.034757834757834755, "frac_reward_zero_std": 0.0, "grad_norm": 0.06513266265392303, "kl": 0.0014369969012477668, "learning_rate": 1.7045454545454546e-06, "loss": 7.140770321711898e-06, "num_tokens": 13924534.0, "reward": 1.8694825172424316, "reward_std": 0.8148167133331299, "rewards/code_complexity_reward/mean": 0.670117199420929, "rewards/code_complexity_reward/std": 0.32075250148773193, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4150390625, "rewards/code_syntax_reward/std": 0.1879657357931137, "rewards/reasoning_present_reward_func/mean": 0.08925781399011612, "rewards/reasoning_present_reward_func/std": 0.030995169654488564, "rewards/xmlcount_reward_func/mean": 0.427490234375, "rewards/xmlcount_reward_func/std": 0.09778565913438797, "step": 61, "step_time": 72.58686789963394 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.072265625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 318.140625, "completions/mean_terminated_length": 303.03997802734375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2580960374325514, "epoch": 0.035327635327635325, "frac_reward_zero_std": 0.0, "grad_norm": 0.05231791362166405, "kl": 0.0014476005617325427, "learning_rate": 1.7329545454545458e-06, "loss": 7.310438377317041e-06, "num_tokens": 14154670.0, "reward": 1.9468750953674316, "reward_std": 0.7834703922271729, "rewards/code_complexity_reward/mean": 0.6892578601837158, "rewards/code_complexity_reward/std": 0.2971189618110657, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.427734375, "rewards/code_syntax_reward/std": 0.17598573863506317, "rewards/reasoning_present_reward_func/mean": 0.09355469048023224, "rewards/reasoning_present_reward_func/std": 0.024579854682087898, "rewards/xmlcount_reward_func/mean": 0.43359375, "rewards/xmlcount_reward_func/std": 0.09244649857282639, "step": 62, "step_time": 60.242386911064386 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.052734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 303.908203125, "completions/mean_terminated_length": 292.32373046875, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.25173549563623965, "epoch": 0.035897435897435895, "frac_reward_zero_std": 0.0, "grad_norm": 0.0481390580534935, "kl": 0.001398667041939916, "learning_rate": 1.7613636363636365e-06, "loss": 7.055699825286865e-06, "num_tokens": 14377959.0, "reward": 1.916601538658142, "reward_std": 0.7586601972579956, "rewards/code_complexity_reward/mean": 0.6957030892372131, "rewards/code_complexity_reward/std": 0.2916836440563202, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4326171875, "rewards/code_syntax_reward/std": 0.1709035038948059, "rewards/reasoning_present_reward_func/mean": 0.08906249701976776, "rewards/reasoning_present_reward_func/std": 0.031241437420248985, "rewards/xmlcount_reward_func/mean": 0.431640625, "rewards/xmlcount_reward_func/std": 0.09874378889799118, "step": 63, "step_time": 57.89170634560287 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 295.376953125, "completions/mean_terminated_length": 286.5711364746094, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.26137595204636455, "epoch": 0.036467236467236465, "frac_reward_zero_std": 0.0, "grad_norm": 0.05898106470704079, "kl": 0.001461687675146095, "learning_rate": 1.7897727272727275e-06, "loss": 7.2631301009096205e-06, "num_tokens": 14596920.0, "reward": 1.924560546875, "reward_std": 0.778907060623169, "rewards/code_complexity_reward/mean": 0.69921875, "rewards/code_complexity_reward/std": 0.3110637068748474, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4248046875, "rewards/code_syntax_reward/std": 0.17890173196792603, "rewards/reasoning_present_reward_func/mean": 0.091796875, "rewards/reasoning_present_reward_func/std": 0.02746807038784027, "rewards/xmlcount_reward_func/mean": 0.431396484375, "rewards/xmlcount_reward_func/std": 0.09031827002763748, "step": 64, "step_time": 67.8220228953287 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.076171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 307.1484375, "completions/mean_terminated_length": 290.2579345703125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2608093989547342, "epoch": 0.037037037037037035, "frac_reward_zero_std": 0.0, "grad_norm": 0.055349696427583694, "kl": 0.0014291874385889969, "learning_rate": 1.8181818181818183e-06, "loss": 7.144717528717592e-06, "num_tokens": 14825092.0, "reward": 1.9756348133087158, "reward_std": 0.8090280294418335, "rewards/code_complexity_reward/mean": 0.69287109375, "rewards/code_complexity_reward/std": 0.2983204126358032, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4306640625, "rewards/code_syntax_reward/std": 0.17297089099884033, "rewards/reasoning_present_reward_func/mean": 0.08964844048023224, "rewards/reasoning_present_reward_func/std": 0.030492907389998436, "rewards/xmlcount_reward_func/mean": 0.430419921875, "rewards/xmlcount_reward_func/std": 0.09772945195436478, "step": 65, "step_time": 58.39125309698284 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.076171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 300.1015625, "completions/mean_terminated_length": 282.6300048828125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2633538250811398, "epoch": 0.037606837606837605, "frac_reward_zero_std": 0.0, "grad_norm": 0.05062492936849594, "kl": 0.0014147337788017467, "learning_rate": 1.8465909090909093e-06, "loss": 7.0140231400728226e-06, "num_tokens": 15045888.0, "reward": 1.9384276866912842, "reward_std": 0.8201648592948914, "rewards/code_complexity_reward/mean": 0.686718761920929, "rewards/code_complexity_reward/std": 0.31444260478019714, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4189453125, "rewards/code_syntax_reward/std": 0.1844557821750641, "rewards/reasoning_present_reward_func/mean": 0.0908203125, "rewards/reasoning_present_reward_func/std": 0.028902137652039528, "rewards/xmlcount_reward_func/mean": 0.429443359375, "rewards/xmlcount_reward_func/std": 0.09414634853601456, "step": 66, "step_time": 49.51429043896496 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.083984375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 311.58984375, "completions/mean_terminated_length": 293.2153625488281, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.25727838673628867, "epoch": 0.038176638176638175, "frac_reward_zero_std": 0.0, "grad_norm": 0.039087675511837006, "kl": 0.0014399580577446613, "learning_rate": 1.8750000000000003e-06, "loss": 7.15853093424812e-06, "num_tokens": 15276318.0, "reward": 1.8912107944488525, "reward_std": 0.7943605780601501, "rewards/code_complexity_reward/mean": 0.672070324420929, "rewards/code_complexity_reward/std": 0.30658337473869324, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.421875, "rewards/code_syntax_reward/std": 0.18172365427017212, "rewards/reasoning_present_reward_func/mean": 0.08925781399011612, "rewards/reasoning_present_reward_func/std": 0.030995169654488564, "rewards/xmlcount_reward_func/mean": 0.4287109375, "rewards/xmlcount_reward_func/std": 0.09914457052946091, "step": 67, "step_time": 62.1581353303045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.052734375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 297.8046875, "completions/mean_terminated_length": 285.88043212890625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.2619506132323295, "epoch": 0.038746438746438745, "frac_reward_zero_std": 0.0, "grad_norm": 0.06503816694021225, "kl": 0.0014414558754651807, "learning_rate": 1.903409090909091e-06, "loss": 7.264543455676176e-06, "num_tokens": 15495818.0, "reward": 1.9811036586761475, "reward_std": 0.8148695230484009, "rewards/code_complexity_reward/mean": 0.685742199420929, "rewards/code_complexity_reward/std": 0.30109360814094543, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4248046875, "rewards/code_syntax_reward/std": 0.17890173196792603, "rewards/reasoning_present_reward_func/mean": 0.09296874701976776, "rewards/reasoning_present_reward_func/std": 0.025592297315597534, "rewards/xmlcount_reward_func/mean": 0.435791015625, "rewards/xmlcount_reward_func/std": 0.08671268075704575, "step": 68, "step_time": 76.78813032526523 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 306.15625, "completions/mean_terminated_length": 286.8034362792969, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.2709082392975688, "epoch": 0.039316239316239315, "frac_reward_zero_std": 0.0, "grad_norm": 0.04973302036523819, "kl": 0.0014839644136372954, "learning_rate": 1.931818181818182e-06, "loss": 7.359456503763795e-06, "num_tokens": 15722730.0, "reward": 1.8916504383087158, "reward_std": 0.8097451329231262, "rewards/code_complexity_reward/mean": 0.6845703125, "rewards/code_complexity_reward/std": 0.3183958828449249, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4169921875, "rewards/code_syntax_reward/std": 0.18622928857803345, "rewards/reasoning_present_reward_func/mean": 0.08867187798023224, "rewards/reasoning_present_reward_func/std": 0.03172462433576584, "rewards/xmlcount_reward_func/mean": 0.426025390625, "rewards/xmlcount_reward_func/std": 0.0976242944598198, "step": 69, "step_time": 72.03397568874061 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 317.62109375, "completions/mean_terminated_length": 300.2510681152344, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.25987900886684656, "epoch": 0.039886039886039885, "frac_reward_zero_std": 0.0, "grad_norm": 0.05783833563327789, "kl": 0.0014500161196338013, "learning_rate": 1.9602272727272728e-06, "loss": 7.260648999363184e-06, "num_tokens": 15952800.0, "reward": 1.886328101158142, "reward_std": 0.8155749440193176, "rewards/code_complexity_reward/mean": 0.6696288585662842, "rewards/code_complexity_reward/std": 0.3203325867652893, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4140625, "rewards/code_syntax_reward/std": 0.18882036209106445, "rewards/reasoning_present_reward_func/mean": 0.09023437649011612, "rewards/reasoning_present_reward_func/std": 0.029713962227106094, "rewards/xmlcount_reward_func/mean": 0.43115234375, "rewards/xmlcount_reward_func/std": 0.09492369741201401, "step": 70, "step_time": 57.474913713522255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 303.248046875, "completions/mean_terminated_length": 289.3312683105469, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2614942246582359, "epoch": 0.040455840455840456, "frac_reward_zero_std": 0.0, "grad_norm": 0.057943448424339294, "kl": 0.0015223258269543294, "learning_rate": 1.9886363636363638e-06, "loss": 7.626978913322091e-06, "num_tokens": 16176103.0, "reward": 1.9597656726837158, "reward_std": 0.8162712454795837, "rewards/code_complexity_reward/mean": 0.678906261920929, "rewards/code_complexity_reward/std": 0.30365416407585144, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4228515625, "rewards/code_syntax_reward/std": 0.18079319596290588, "rewards/reasoning_present_reward_func/mean": 0.09140625596046448, "rewards/reasoning_present_reward_func/std": 0.028054581955075264, "rewards/xmlcount_reward_func/mean": 0.4345703125, "rewards/xmlcount_reward_func/std": 0.09573186188936234, "step": 71, "step_time": 60.90209249500185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 312.94921875, "completions/mean_terminated_length": 298.7908020019531, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.2587594345677644, "epoch": 0.041025641025641026, "frac_reward_zero_std": 0.0, "grad_norm": 0.0502912662923336, "kl": 0.0015326293068937957, "learning_rate": 2.0170454545454548e-06, "loss": 7.681868737563491e-06, "num_tokens": 16404045.0, "reward": 1.921533226966858, "reward_std": 0.7325757741928101, "rewards/code_complexity_reward/mean": 0.7085937261581421, "rewards/code_complexity_reward/std": 0.284351110458374, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4384765625, "rewards/code_syntax_reward/std": 0.16440613567829132, "rewards/reasoning_present_reward_func/mean": 0.08769531548023224, "rewards/reasoning_present_reward_func/std": 0.032881226390600204, "rewards/xmlcount_reward_func/mean": 0.436767578125, "rewards/xmlcount_reward_func/std": 0.09186292439699173, "step": 72, "step_time": 69.95441555883735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.083984375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 315.36328125, "completions/mean_terminated_length": 297.3347473144531, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.26148316683247685, "epoch": 0.041595441595441596, "frac_reward_zero_std": 0.0, "grad_norm": 0.0423474945127964, "kl": 0.0015179241681835265, "learning_rate": 2.0454545454545457e-06, "loss": 7.571448804810643e-06, "num_tokens": 16634183.0, "reward": 1.8990235328674316, "reward_std": 0.8100738525390625, "rewards/code_complexity_reward/mean": 0.6696288585662842, "rewards/code_complexity_reward/std": 0.30915650725364685, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.419921875, "rewards/code_syntax_reward/std": 0.1835547834634781, "rewards/reasoning_present_reward_func/mean": 0.08828125149011612, "rewards/reasoning_present_reward_func/std": 0.032195813953876495, "rewards/xmlcount_reward_func/mean": 0.43212890625, "rewards/xmlcount_reward_func/std": 0.09303250908851624, "step": 73, "step_time": 66.48579281941056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 306.611328125, "completions/mean_terminated_length": 288.2574462890625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.26409853459335864, "epoch": 0.042165242165242166, "frac_reward_zero_std": 0.0, "grad_norm": 0.044291988015174866, "kl": 0.0015010833376436494, "learning_rate": 2.0738636363636367e-06, "loss": 7.483933586627245e-06, "num_tokens": 16859784.0, "reward": 1.969970703125, "reward_std": 0.8307995200157166, "rewards/code_complexity_reward/mean": 0.6875976324081421, "rewards/code_complexity_reward/std": 0.3187357187271118, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4208984375, "rewards/code_syntax_reward/std": 0.18264412879943848, "rewards/reasoning_present_reward_func/mean": 0.09023437649011612, "rewards/reasoning_present_reward_func/std": 0.029713962227106094, "rewards/xmlcount_reward_func/mean": 0.439208984375, "rewards/xmlcount_reward_func/std": 0.08493124693632126, "step": 74, "step_time": 58.20736690983176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.068359375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 298.431640625, "completions/mean_terminated_length": 282.760986328125, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.24921432370319963, "epoch": 0.042735042735042736, "frac_reward_zero_std": 0.0, "grad_norm": 0.0629694014787674, "kl": 0.0015171710947470274, "learning_rate": 2.1022727272727277e-06, "loss": 7.532174095103983e-06, "num_tokens": 17081373.0, "reward": 1.9830565452575684, "reward_std": 0.8039528727531433, "rewards/code_complexity_reward/mean": 0.6983398795127869, "rewards/code_complexity_reward/std": 0.29879432916641235, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4306640625, "rewards/code_syntax_reward/std": 0.17297089099884033, "rewards/reasoning_present_reward_func/mean": 0.09062500298023224, "rewards/reasoning_present_reward_func/std": 0.029176566749811172, "rewards/xmlcount_reward_func/mean": 0.441162109375, "rewards/xmlcount_reward_func/std": 0.0947432816028595, "step": 75, "step_time": 66.14936178084463 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 309.896484375, "completions/mean_terminated_length": 296.4229431152344, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.25323754898272455, "epoch": 0.043304843304843306, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04706120863556862, "kl": 0.0015340185364038916, "learning_rate": 2.1306818181818183e-06, "loss": 7.69590405980125e-06, "num_tokens": 17309784.0, "reward": 2.0008788108825684, "reward_std": 0.804469645023346, "rewards/code_complexity_reward/mean": 0.694042980670929, "rewards/code_complexity_reward/std": 0.2967347800731659, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.431640625, "rewards/code_syntax_reward/std": 0.1719430834054947, "rewards/reasoning_present_reward_func/mean": 0.09199218451976776, "rewards/reasoning_present_reward_func/std": 0.02716795541346073, "rewards/xmlcount_reward_func/mean": 0.4375, "rewards/xmlcount_reward_func/std": 0.08708140254020691, "step": 76, "step_time": 77.73355190921575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.060546875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 292.248046875, "completions/mean_terminated_length": 278.0852355957031, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.24847867572680116, "epoch": 0.043874643874643876, "frac_reward_zero_std": 0.0, "grad_norm": 0.06271505355834961, "kl": 0.001568685582242324, "learning_rate": 2.1590909090909092e-06, "loss": 7.791328243911266e-06, "num_tokens": 17529039.0, "reward": 2.00146484375, "reward_std": 0.7569817304611206, "rewards/code_complexity_reward/mean": 0.7320312261581421, "rewards/code_complexity_reward/std": 0.2822529077529907, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.44140625, "rewards/code_syntax_reward/std": 0.16097907721996307, "rewards/reasoning_present_reward_func/mean": 0.09218750149011612, "rewards/reasoning_present_reward_func/std": 0.026863066479563713, "rewards/xmlcount_reward_func/mean": 0.44091796875, "rewards/xmlcount_reward_func/std": 0.09012134373188019, "step": 77, "step_time": 74.56397340539843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.056640625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 294.5078125, "completions/mean_terminated_length": 281.44927978515625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.259597634896636, "epoch": 0.044444444444444446, "frac_reward_zero_std": 0.0, "grad_norm": 0.06683129817247391, "kl": 0.001667376531258924, "learning_rate": 2.1875000000000002e-06, "loss": 8.276052540168166e-06, "num_tokens": 17747499.0, "reward": 1.99951171875, "reward_std": 0.7400264739990234, "rewards/code_complexity_reward/mean": 0.719531238079071, "rewards/code_complexity_reward/std": 0.2777096927165985, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.443359375, "rewards/code_syntax_reward/std": 0.1586231142282486, "rewards/reasoning_present_reward_func/mean": 0.09199218451976776, "rewards/reasoning_present_reward_func/std": 0.02716795541346073, "rewards/xmlcount_reward_func/mean": 0.43798828125, "rewards/xmlcount_reward_func/std": 0.08637489378452301, "step": 78, "step_time": 58.78082487359643 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.064453125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 312.306640625, "completions/mean_terminated_length": 298.549072265625, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.25891812867484987, "epoch": 0.045014245014245016, "frac_reward_zero_std": 0.0, "grad_norm": 0.03975647687911987, "kl": 0.0016401135435444303, "learning_rate": 2.2159090909090912e-06, "loss": 8.162343874573708e-06, "num_tokens": 17975224.0, "reward": 2.0057129859924316, "reward_std": 0.773293137550354, "rewards/code_complexity_reward/mean": 0.70751953125, "rewards/code_complexity_reward/std": 0.2848714590072632, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.439453125, "rewards/code_syntax_reward/std": 0.16327762603759766, "rewards/reasoning_present_reward_func/mean": 0.09140625596046448, "rewards/reasoning_present_reward_func/std": 0.028054581955075264, "rewards/xmlcount_reward_func/mean": 0.443115234375, "rewards/xmlcount_reward_func/std": 0.08619320392608643, "step": 79, "step_time": 63.83947661425918 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.072265625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 304.79296875, "completions/mean_terminated_length": 288.6526184082031, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.2537184050306678, "epoch": 0.045584045584045586, "frac_reward_zero_std": 0.0, "grad_norm": 0.05238071084022522, "kl": 0.0016340143938577967, "learning_rate": 2.2443181818181818e-06, "loss": 8.146700565703213e-06, "num_tokens": 18200438.0, "reward": 2.0213868618011475, "reward_std": 0.7892512083053589, "rewards/code_complexity_reward/mean": 0.70263671875, "rewards/code_complexity_reward/std": 0.28456351161003113, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4384765625, "rewards/code_syntax_reward/std": 0.16440613567829132, "rewards/reasoning_present_reward_func/mean": 0.09218750894069672, "rewards/reasoning_present_reward_func/std": 0.026863066479563713, "rewards/xmlcount_reward_func/mean": 0.4423828125, "rewards/xmlcount_reward_func/std": 0.08764468878507614, "step": 80, "step_time": 77.55124020483345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 307.373046875, "completions/mean_terminated_length": 290.9683532714844, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.24714159313589334, "epoch": 0.046153846153846156, "frac_reward_zero_std": 0.0, "grad_norm": 0.051232852041721344, "kl": 0.0016460742517665494, "learning_rate": 2.2727272727272728e-06, "loss": 8.168717613443732e-06, "num_tokens": 18425853.0, "reward": 2.0206053256988525, "reward_std": 0.8082380294799805, "rewards/code_complexity_reward/mean": 0.69970703125, "rewards/code_complexity_reward/std": 0.2911360263824463, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4326171875, "rewards/code_syntax_reward/std": 0.1709035038948059, "rewards/reasoning_present_reward_func/mean": 0.09238281846046448, "rewards/reasoning_present_reward_func/std": 0.026553234085440636, "rewards/xmlcount_reward_func/mean": 0.4365234375, "rewards/xmlcount_reward_func/std": 0.0925239846110344, "step": 81, "step_time": 56.77581342495978 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 294.369140625, "completions/mean_terminated_length": 278.88909912109375, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.2503754023928195, "epoch": 0.046723646723646726, "frac_reward_zero_std": 0.0, "grad_norm": 0.04064137116074562, "kl": 0.0016628617549940827, "learning_rate": 2.3011363636363637e-06, "loss": 8.301460184156895e-06, "num_tokens": 18647954.0, "reward": 2.027294874191284, "reward_std": 0.7832741737365723, "rewards/code_complexity_reward/mean": 0.7159179449081421, "rewards/code_complexity_reward/std": 0.2916407585144043, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.435546875, "rewards/code_syntax_reward/std": 0.16771192848682404, "rewards/reasoning_present_reward_func/mean": 0.09042969346046448, "rewards/reasoning_present_reward_func/std": 0.02944713830947876, "rewards/xmlcount_reward_func/mean": 0.443603515625, "rewards/xmlcount_reward_func/std": 0.08615993708372116, "step": 82, "step_time": 71.15480459108949 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.056640625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 298.818359375, "completions/mean_terminated_length": 286.0186462402344, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.260279888054356, "epoch": 0.0472934472934473, "frac_reward_zero_std": 0.0, "grad_norm": 0.0486670546233654, "kl": 0.002043503251115908, "learning_rate": 2.3295454545454547e-06, "loss": 1.0203220881521702e-05, "num_tokens": 18868837.0, "reward": 1.998046875, "reward_std": 0.7905252575874329, "rewards/code_complexity_reward/mean": 0.7058594226837158, "rewards/code_complexity_reward/std": 0.2923935353755951, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4326171875, "rewards/code_syntax_reward/std": 0.1709035038948059, "rewards/reasoning_present_reward_func/mean": 0.09199219197034836, "rewards/reasoning_present_reward_func/std": 0.02716795541346073, "rewards/xmlcount_reward_func/mean": 0.44921875, "rewards/xmlcount_reward_func/std": 0.08449733257293701, "step": 83, "step_time": 51.719677494838834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.091796875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 316.994140625, "completions/mean_terminated_length": 297.28387451171875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2654201358091086, "epoch": 0.04786324786324787, "frac_reward_zero_std": 0.015625, "grad_norm": 0.05781826749444008, "kl": 0.0018066004522552248, "learning_rate": 2.3579545454545457e-06, "loss": 9.034411050379276e-06, "num_tokens": 19099634.0, "reward": 1.9458985328674316, "reward_std": 0.8286960124969482, "rewards/code_complexity_reward/mean": 0.68115234375, "rewards/code_complexity_reward/std": 0.3146321475505829, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.419921875, "rewards/code_syntax_reward/std": 0.1835547834634781, "rewards/reasoning_present_reward_func/mean": 0.09238281846046448, "rewards/reasoning_present_reward_func/std": 0.026553234085440636, "rewards/xmlcount_reward_func/mean": 0.44189453125, "rewards/xmlcount_reward_func/std": 0.08836536854505539, "step": 84, "step_time": 66.72604755405337 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 301.31640625, "completions/mean_terminated_length": 284.4261474609375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.25057358853518963, "epoch": 0.04843304843304843, "frac_reward_zero_std": 0.0, "grad_norm": 0.0522913783788681, "kl": 0.0018201899838459212, "learning_rate": 2.3863636363636367e-06, "loss": 9.12499672267586e-06, "num_tokens": 19320724.0, "reward": 2.001415967941284, "reward_std": 0.8005436062812805, "rewards/code_complexity_reward/mean": 0.69873046875, "rewards/code_complexity_reward/std": 0.2919389009475708, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.43359375, "rewards/code_syntax_reward/std": 0.16985194385051727, "rewards/reasoning_present_reward_func/mean": 0.09296874701976776, "rewards/reasoning_present_reward_func/std": 0.025592297315597534, "rewards/xmlcount_reward_func/mean": 0.440185546875, "rewards/xmlcount_reward_func/std": 0.09314894676208496, "step": 85, "step_time": 61.83876821678132 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.052734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 303.3671875, "completions/mean_terminated_length": 291.7525939941406, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.260156482225284, "epoch": 0.049002849002849, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04092004522681236, "kl": 0.0018178255450038705, "learning_rate": 2.4147727272727277e-06, "loss": 9.090701496461406e-06, "num_tokens": 19549000.0, "reward": 1.9673340320587158, "reward_std": 0.6241146922111511, "rewards/code_complexity_reward/mean": 0.7533203363418579, "rewards/code_complexity_reward/std": 0.24116219580173492, "rewards/code_execution_reward/mean": 0.212890625, "rewards/code_execution_reward/std": 0.409751296043396, "rewards/code_syntax_reward/mean": 0.4609375, "rewards/code_syntax_reward/std": 0.13431532680988312, "rewards/reasoning_present_reward_func/mean": 0.09023438394069672, "rewards/reasoning_present_reward_func/std": 0.029713962227106094, "rewards/xmlcount_reward_func/mean": 0.449951171875, "rewards/xmlcount_reward_func/std": 0.0896625965833664, "step": 86, "step_time": 59.83685706090182 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.083984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 319.3359375, "completions/mean_terminated_length": 301.6716613769531, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.25425554905086756, "epoch": 0.04957264957264957, "frac_reward_zero_std": 0.0, "grad_norm": 0.05761490762233734, "kl": 0.0018045668166450923, "learning_rate": 2.4431818181818182e-06, "loss": 8.944189175963402e-06, "num_tokens": 19781892.0, "reward": 2.019238233566284, "reward_std": 0.7933021187782288, "rewards/code_complexity_reward/mean": 0.6859375238418579, "rewards/code_complexity_reward/std": 0.2832704782485962, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4345703125, "rewards/code_syntax_reward/std": 0.16878816485404968, "rewards/reasoning_present_reward_func/mean": 0.09160156548023224, "rewards/reasoning_present_reward_func/std": 0.02776356413960457, "rewards/xmlcount_reward_func/mean": 0.44189453125, "rewards/xmlcount_reward_func/std": 0.0910915732383728, "step": 87, "step_time": 78.58230659365654 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.060546875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 303.328125, "completions/mean_terminated_length": 289.8794250488281, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.2554986388422549, "epoch": 0.05014245014245014, "frac_reward_zero_std": 0.0, "grad_norm": 0.2389148771762848, "kl": 0.003501642031551455, "learning_rate": 2.4715909090909092e-06, "loss": 1.7440426745451987e-05, "num_tokens": 20005380.0, "reward": 2.002490282058716, "reward_std": 0.7243678569793701, "rewards/code_complexity_reward/mean": 0.7208007574081421, "rewards/code_complexity_reward/std": 0.27495232224464417, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4462890625, "rewards/code_syntax_reward/std": 0.15497584640979767, "rewards/reasoning_present_reward_func/mean": 0.09296874701976776, "rewards/reasoning_present_reward_func/std": 0.025592297315597534, "rewards/xmlcount_reward_func/mean": 0.451416015625, "rewards/xmlcount_reward_func/std": 0.08232611417770386, "step": 88, "step_time": 53.52078986167908 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 311.384765625, "completions/mean_terminated_length": 295.3016662597656, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.246652087662369, "epoch": 0.05071225071225071, "frac_reward_zero_std": 0.0, "grad_norm": 0.04147793725132942, "kl": 0.0018622805291670375, "learning_rate": 2.5e-06, "loss": 9.322702680947259e-06, "num_tokens": 20232313.0, "reward": 2.017822265625, "reward_std": 0.7772612571716309, "rewards/code_complexity_reward/mean": 0.7040039300918579, "rewards/code_complexity_reward/std": 0.2878127694129944, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4365234375, "rewards/code_syntax_reward/std": 0.16662302613258362, "rewards/reasoning_present_reward_func/mean": 0.09531249850988388, "rewards/reasoning_present_reward_func/std": 0.021157780662178993, "rewards/xmlcount_reward_func/mean": 0.453857421875, "rewards/xmlcount_reward_func/std": 0.07960281521081924, "step": 89, "step_time": 67.0124381557107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.099609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 314.58203125, "completions/mean_terminated_length": 292.74188232421875, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.2603450051974505, "epoch": 0.05128205128205128, "frac_reward_zero_std": 0.0, "grad_norm": 0.03511788323521614, "kl": 0.0019001631189894397, "learning_rate": 2.528409090909091e-06, "loss": 9.443960152566433e-06, "num_tokens": 20461395.0, "reward": 1.9472167491912842, "reward_std": 0.7997665405273438, "rewards/code_complexity_reward/mean": 0.688769519329071, "rewards/code_complexity_reward/std": 0.30062857270240784, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.431640625, "rewards/code_syntax_reward/std": 0.1719430834054947, "rewards/reasoning_present_reward_func/mean": 0.09218750149011612, "rewards/reasoning_present_reward_func/std": 0.026863066479563713, "rewards/xmlcount_reward_func/mean": 0.447509765625, "rewards/xmlcount_reward_func/std": 0.0916520431637764, "step": 90, "step_time": 69.94999926444143 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.060546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 309.353515625, "completions/mean_terminated_length": 296.29315185546875, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.2583789429627359, "epoch": 0.05185185185185185, "frac_reward_zero_std": 0.0, "grad_norm": 0.04642622917890549, "kl": 0.001994449563426315, "learning_rate": 2.556818181818182e-06, "loss": 9.937590220943093e-06, "num_tokens": 20691688.0, "reward": 1.9622070789337158, "reward_std": 0.7415587306022644, "rewards/code_complexity_reward/mean": 0.71435546875, "rewards/code_complexity_reward/std": 0.2869366705417633, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4384765625, "rewards/code_syntax_reward/std": 0.16440613567829132, "rewards/reasoning_present_reward_func/mean": 0.09550780802965164, "rewards/reasoning_present_reward_func/std": 0.020733514800667763, "rewards/xmlcount_reward_func/mean": 0.4599609375, "rewards/xmlcount_reward_func/std": 0.07407880574464798, "step": 91, "step_time": 72.62003873940557 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 296.681640625, "completions/mean_terminated_length": 284.2251892089844, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24971931567415595, "epoch": 0.05242165242165242, "frac_reward_zero_std": 0.0, "grad_norm": 0.043814633041620255, "kl": 0.0021286837545630988, "learning_rate": 2.585227272727273e-06, "loss": 1.0638890671543777e-05, "num_tokens": 20912229.0, "reward": 2.1387696266174316, "reward_std": 0.7336114645004272, "rewards/code_complexity_reward/mean": 0.7568359375, "rewards/code_complexity_reward/std": 0.24559156596660614, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4599609375, "rewards/code_syntax_reward/std": 0.1358397752046585, "rewards/reasoning_present_reward_func/mean": 0.09238281846046448, "rewards/reasoning_present_reward_func/std": 0.026553234085440636, "rewards/xmlcount_reward_func/mean": 0.45654296875, "rewards/xmlcount_reward_func/std": 0.08015495538711548, "step": 92, "step_time": 96.57543477416039 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 301.509765625, "completions/mean_terminated_length": 287.47711181640625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2554732628632337, "epoch": 0.05299145299145299, "frac_reward_zero_std": 0.0, "grad_norm": 0.04983806610107422, "kl": 0.002198128016971168, "learning_rate": 2.6136363636363637e-06, "loss": 1.097308995667845e-05, "num_tokens": 21136618.0, "reward": 2.021728515625, "reward_std": 0.8000980615615845, "rewards/code_complexity_reward/mean": 0.6875, "rewards/code_complexity_reward/std": 0.2862299680709839, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.435546875, "rewards/code_syntax_reward/std": 0.16771192848682404, "rewards/reasoning_present_reward_func/mean": 0.091796875, "rewards/reasoning_present_reward_func/std": 0.02746807038784027, "rewards/xmlcount_reward_func/mean": 0.451416015625, "rewards/xmlcount_reward_func/std": 0.08666859567165375, "step": 93, "step_time": 66.29663393180817 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 314.859375, "completions/mean_terminated_length": 297.2425537109375, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.26159683405421674, "epoch": 0.05356125356125356, "frac_reward_zero_std": 0.015625, "grad_norm": 0.039610687643289566, "kl": 0.002188298110922915, "learning_rate": 2.642045454545455e-06, "loss": 1.0854622814804316e-05, "num_tokens": 21367050.0, "reward": 1.9625976085662842, "reward_std": 0.7241719961166382, "rewards/code_complexity_reward/mean": 0.7105468511581421, "rewards/code_complexity_reward/std": 0.27572372555732727, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.44140625, "rewards/code_syntax_reward/std": 0.16097907721996307, "rewards/reasoning_present_reward_func/mean": 0.09238281100988388, "rewards/reasoning_present_reward_func/std": 0.026553234085440636, "rewards/xmlcount_reward_func/mean": 0.45654296875, "rewards/xmlcount_reward_func/std": 0.07704271376132965, "step": 94, "step_time": 60.559049104340374 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.091796875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 325.126953125, "completions/mean_terminated_length": 306.23870849609375, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.24917435040697455, "epoch": 0.05413105413105413, "frac_reward_zero_std": 0.0, "grad_norm": 0.0366891510784626, "kl": 0.002176914456867962, "learning_rate": 2.6704545454545457e-06, "loss": 1.0805786587297916e-05, "num_tokens": 21603947.0, "reward": 1.9341309070587158, "reward_std": 0.724240243434906, "rewards/code_complexity_reward/mean": 0.700488269329071, "rewards/code_complexity_reward/std": 0.2871425449848175, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.4375, "rewards/code_syntax_reward/std": 0.16552117466926575, "rewards/reasoning_present_reward_func/mean": 0.0947265625, "rewards/reasoning_present_reward_func/std": 0.022372128441929817, "rewards/xmlcount_reward_func/mean": 0.461181640625, "rewards/xmlcount_reward_func/std": 0.071592316031456, "step": 95, "step_time": 79.13377552013844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 307.98828125, "completions/mean_terminated_length": 293.47698974609375, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.26275120535865426, "epoch": 0.0547008547008547, "frac_reward_zero_std": 0.0, "grad_norm": 0.04448787122964859, "kl": 0.0023023289704724448, "learning_rate": 2.6988636363636367e-06, "loss": 1.1444935807958245e-05, "num_tokens": 21831509.0, "reward": 1.9855468273162842, "reward_std": 0.7373301386833191, "rewards/code_complexity_reward/mean": 0.7097656726837158, "rewards/code_complexity_reward/std": 0.27840110659599304, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4404296875, "rewards/code_syntax_reward/std": 0.16213536262512207, "rewards/reasoning_present_reward_func/mean": 0.09414062649011612, "rewards/reasoning_present_reward_func/std": 0.023509245365858078, "rewards/xmlcount_reward_func/mean": 0.4580078125, "rewards/xmlcount_reward_func/std": 0.07626517117023468, "step": 96, "step_time": 85.74596656300128 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 299.220703125, "completions/mean_terminated_length": 286.9111328125, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.2561193623114377, "epoch": 0.05527065527065527, "frac_reward_zero_std": 0.0, "grad_norm": 0.04756823554635048, "kl": 0.00249755613731395, "learning_rate": 2.7272727272727272e-06, "loss": 1.2427057299646549e-05, "num_tokens": 22053350.0, "reward": 2.0777831077575684, "reward_std": 0.7703681588172913, "rewards/code_complexity_reward/mean": 0.72412109375, "rewards/code_complexity_reward/std": 0.2719596326351166, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4453125, "rewards/code_syntax_reward/std": 0.15620718896389008, "rewards/reasoning_present_reward_func/mean": 0.09121093899011612, "rewards/reasoning_present_reward_func/std": 0.02834126725792885, "rewards/xmlcount_reward_func/mean": 0.455810546875, "rewards/xmlcount_reward_func/std": 0.08477076888084412, "step": 97, "step_time": 59.490785494446754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.060546875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 301.845703125, "completions/mean_terminated_length": 288.30145263671875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.24513352708891034, "epoch": 0.05584045584045584, "frac_reward_zero_std": 0.0, "grad_norm": 0.04687436297535896, "kl": 0.0025402872033737367, "learning_rate": 2.7556818181818186e-06, "loss": 1.261302713828627e-05, "num_tokens": 22276703.0, "reward": 2.0796875953674316, "reward_std": 0.7122769951820374, "rewards/code_complexity_reward/mean": 0.7430664300918579, "rewards/code_complexity_reward/std": 0.24945296347141266, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.455078125, "rewards/code_syntax_reward/std": 0.1431187242269516, "rewards/reasoning_present_reward_func/mean": 0.09199219197034836, "rewards/reasoning_present_reward_func/std": 0.02716795541346073, "rewards/xmlcount_reward_func/mean": 0.46533203125, "rewards/xmlcount_reward_func/std": 0.07390285283327103, "step": 98, "step_time": 70.71302082855254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 313.083984375, "completions/mean_terminated_length": 298.0399169921875, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.2513106169644743, "epoch": 0.05641025641025641, "frac_reward_zero_std": 0.0, "grad_norm": 0.04315786436200142, "kl": 0.0024436699368379777, "learning_rate": 2.784090909090909e-06, "loss": 1.2094911653548479e-05, "num_tokens": 22504418.0, "reward": 1.9408202171325684, "reward_std": 0.6885427832603455, "rewards/code_complexity_reward/mean": 0.7164062261581421, "rewards/code_complexity_reward/std": 0.26480501890182495, "rewards/code_execution_reward/mean": 0.2265625, "rewards/code_execution_reward/std": 0.4190165400505066, "rewards/code_syntax_reward/mean": 0.44921875, "rewards/code_syntax_reward/std": 0.15118376910686493, "rewards/reasoning_present_reward_func/mean": 0.09160156548023224, "rewards/reasoning_present_reward_func/std": 0.02776356413960457, "rewards/xmlcount_reward_func/mean": 0.45703125, "rewards/xmlcount_reward_func/std": 0.08734435588121414, "step": 99, "step_time": 61.04436453990638 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 310.53515625, "completions/mean_terminated_length": 295.29833984375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2492448748089373, "epoch": 0.05698005698005698, "frac_reward_zero_std": 0.0, "grad_norm": 0.04412516951560974, "kl": 0.002447464950819267, "learning_rate": 2.8125e-06, "loss": 1.2238248018547893e-05, "num_tokens": 22730052.0, "reward": 2.09912109375, "reward_std": 0.7229527235031128, "rewards/code_complexity_reward/mean": 0.73193359375, "rewards/code_complexity_reward/std": 0.2532774806022644, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.455078125, "rewards/code_syntax_reward/std": 0.1431187242269516, "rewards/reasoning_present_reward_func/mean": 0.0927734375, "rewards/reasoning_present_reward_func/std": 0.02591804414987564, "rewards/xmlcount_reward_func/mean": 0.4599609375, "rewards/xmlcount_reward_func/std": 0.07651533931493759, "step": 100, "step_time": 61.89334254898131 }, { "epoch": 0.05698005698005698, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.06625, "eval_completions/max_length": 416.67, "eval_completions/max_terminated_length": 398.66, "eval_completions/mean_length": 304.61, "eval_completions/mean_terminated_length": 295.4446697998047, "eval_completions/min_length": 209.39, "eval_completions/min_terminated_length": 209.39, "eval_entropy": 0.25123013541102407, "eval_frac_reward_zero_std": 0.0, "eval_kl": 0.002579880077391863, "eval_loss": 0.0024312541354447603, "eval_num_tokens": 22730052.0, "eval_reward": 2.0293125104904175, "eval_reward_std": 0.4547412573173642, "eval_rewards/code_complexity_reward/mean": 0.7415625047683716, "eval_rewards/code_complexity_reward/std": 0.15827444912865757, "eval_rewards/code_execution_reward/mean": 0.27875, "eval_rewards/code_execution_reward/std": 0.23188325613737107, "eval_rewards/code_syntax_reward/mean": 0.45625, "eval_rewards/code_syntax_reward/std": 0.08109838545322418, "eval_rewards/reasoning_present_reward_func/mean": 0.0915000031515956, "eval_rewards/reasoning_present_reward_func/std": 0.018981204964220524, "eval_rewards/xmlcount_reward_func/mean": 0.46125, "eval_rewards/xmlcount_reward_func/std": 0.06234732661396265, "eval_runtime": 1933.1273, "eval_samples_per_second": 0.052, "eval_steps_per_second": 0.007, "step": 100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.048828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 299.017578125, "completions/mean_terminated_length": 288.0841979980469, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.25202495930716395, "epoch": 0.05754985754985755, "frac_reward_zero_std": 0.0, "grad_norm": 0.04956540837883949, "kl": 0.0029298962217580993, "learning_rate": 2.8409090909090916e-06, "loss": 1.4647259376943111e-05, "num_tokens": 22952341.0, "reward": 2.153027296066284, "reward_std": 0.6989705562591553, "rewards/code_complexity_reward/mean": 0.758984386920929, "rewards/code_complexity_reward/std": 0.2276342660188675, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4658203125, "rewards/code_syntax_reward/std": 0.12630419433116913, "rewards/reasoning_present_reward_func/mean": 0.0927734375, "rewards/reasoning_present_reward_func/std": 0.02591804414987564, "rewards/xmlcount_reward_func/mean": 0.46044921875, "rewards/xmlcount_reward_func/std": 0.08103231340646744, "step": 101, "step_time": 106.3747112005949 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.064453125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 299.45703125, "completions/mean_terminated_length": 284.814208984375, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.2541806404478848, "epoch": 0.05811965811965812, "frac_reward_zero_std": 0.0, "grad_norm": 0.04316093027591705, "kl": 0.002743189310422167, "learning_rate": 2.869318181818182e-06, "loss": 1.3781071174889803e-05, "num_tokens": 23175375.0, "reward": 2.0807619094848633, "reward_std": 0.7318655848503113, "rewards/code_complexity_reward/mean": 0.737597644329071, "rewards/code_complexity_reward/std": 0.2563241124153137, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.453125, "rewards/code_syntax_reward/std": 0.14588283002376556, "rewards/reasoning_present_reward_func/mean": 0.09414062649011612, "rewards/reasoning_present_reward_func/std": 0.023509247228503227, "rewards/xmlcount_reward_func/mean": 0.4677734375, "rewards/xmlcount_reward_func/std": 0.07293486595153809, "step": 102, "step_time": 72.08230900764465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 291.92578125, "completions/mean_terminated_length": 275.2815246582031, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.25170717225410044, "epoch": 0.05868945868945869, "frac_reward_zero_std": 0.0, "grad_norm": 0.036982517689466476, "kl": 0.0027622776287898887, "learning_rate": 2.897727272727273e-06, "loss": 1.3673503417521715e-05, "num_tokens": 23392937.0, "reward": 2.1651368141174316, "reward_std": 0.7465693950653076, "rewards/code_complexity_reward/mean": 0.7587890625, "rewards/code_complexity_reward/std": 0.249404177069664, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.45703125, "rewards/code_syntax_reward/std": 0.14027291536331177, "rewards/reasoning_present_reward_func/mean": 0.09140624850988388, "rewards/reasoning_present_reward_func/std": 0.028054583817720413, "rewards/xmlcount_reward_func/mean": 0.46337890625, "rewards/xmlcount_reward_func/std": 0.07623226940631866, "step": 103, "step_time": 68.03710007574409 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.033203125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 280.876953125, "completions/mean_terminated_length": 272.93939208984375, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.2465027787256986, "epoch": 0.05925925925925926, "frac_reward_zero_std": 0.0, "grad_norm": 0.04209044575691223, "kl": 0.003607352224207716, "learning_rate": 2.9261363636363637e-06, "loss": 1.789143425412476e-05, "num_tokens": 23607658.0, "reward": 2.1295409202575684, "reward_std": 0.7034477591514587, "rewards/code_complexity_reward/mean": 0.7624022960662842, "rewards/code_complexity_reward/std": 0.2396538108587265, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4609375, "rewards/code_syntax_reward/std": 0.13431532680988312, "rewards/reasoning_present_reward_func/mean": 0.09492187201976776, "rewards/reasoning_present_reward_func/std": 0.021976543590426445, "rewards/xmlcount_reward_func/mean": 0.467529296875, "rewards/xmlcount_reward_func/std": 0.0691654160618782, "step": 104, "step_time": 61.91604941431433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 305.6953125, "completions/mean_terminated_length": 292.85479736328125, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.246559567283839, "epoch": 0.05982905982905983, "frac_reward_zero_std": 0.015625, "grad_norm": 0.027451619505882263, "kl": 0.003422526542635751, "learning_rate": 2.954545454545455e-06, "loss": 1.702731242403388e-05, "num_tokens": 23834278.0, "reward": 2.080322265625, "reward_std": 0.6888667345046997, "rewards/code_complexity_reward/mean": 0.732421875, "rewards/code_complexity_reward/std": 0.23283246159553528, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4619140625, "rewards/code_syntax_reward/std": 0.13276617228984833, "rewards/reasoning_present_reward_func/mean": 0.095703125, "rewards/reasoning_present_reward_func/std": 0.02029850147664547, "rewards/xmlcount_reward_func/mean": 0.475830078125, "rewards/xmlcount_reward_func/std": 0.060538552701473236, "step": 105, "step_time": 71.17014014348388 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 281.880859375, "completions/mean_terminated_length": 266.53961181640625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.24209923716261983, "epoch": 0.0603988603988604, "frac_reward_zero_std": 0.0, "grad_norm": 0.04580177739262581, "kl": 0.003626885129051516, "learning_rate": 2.9829545454545457e-06, "loss": 1.811998663470149e-05, "num_tokens": 24047953.0, "reward": 2.166748046875, "reward_std": 0.7433110475540161, "rewards/code_complexity_reward/mean": 0.7621093988418579, "rewards/code_complexity_reward/std": 0.25491228699684143, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.45703125, "rewards/code_syntax_reward/std": 0.14027291536331177, "rewards/reasoning_present_reward_func/mean": 0.09335937350988388, "rewards/reasoning_present_reward_func/std": 0.02492343820631504, "rewards/xmlcount_reward_func/mean": 0.469482421875, "rewards/xmlcount_reward_func/std": 0.06917232275009155, "step": 106, "step_time": 59.104884828440845 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.037109375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 288.490234375, "completions/mean_terminated_length": 279.8762512207031, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24842161335982382, "epoch": 0.06096866096866097, "frac_reward_zero_std": 0.0, "grad_norm": 0.03638351336121559, "kl": 0.003033861315998365, "learning_rate": 3.0113636363636366e-06, "loss": 1.5015946701169014e-05, "num_tokens": 24262460.0, "reward": 2.1773927211761475, "reward_std": 0.6764564514160156, "rewards/code_complexity_reward/mean": 0.7626953125, "rewards/code_complexity_reward/std": 0.21830374002456665, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.470703125, "rewards/code_syntax_reward/std": 0.11754623055458069, "rewards/reasoning_present_reward_func/mean": 0.09511718899011612, "rewards/reasoning_present_reward_func/std": 0.02157193422317505, "rewards/xmlcount_reward_func/mean": 0.473876953125, "rewards/xmlcount_reward_func/std": 0.06741638481616974, "step": 107, "step_time": 60.70329389348626 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.041015625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 301.314453125, "completions/mean_terminated_length": 292.303466796875, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.2446919628418982, "epoch": 0.06153846153846154, "frac_reward_zero_std": 0.0, "grad_norm": 0.03414026275277138, "kl": 0.0029751545316685224, "learning_rate": 3.039772727272727e-06, "loss": 1.5061144949868321e-05, "num_tokens": 24486493.0, "reward": 2.17431640625, "reward_std": 0.6592091917991638, "rewards/code_complexity_reward/mean": 0.7646484375, "rewards/code_complexity_reward/std": 0.2125934213399887, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4697265625, "rewards/code_syntax_reward/std": 0.11936526000499725, "rewards/reasoning_present_reward_func/mean": 0.0966796875, "rewards/reasoning_present_reward_func/std": 0.01793418452143669, "rewards/xmlcount_reward_func/mean": 0.47998046875, "rewards/xmlcount_reward_func/std": 0.06079302728176117, "step": 108, "step_time": 67.78925467282534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.080078125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 307.662109375, "completions/mean_terminated_length": 289.874755859375, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 0.24227836751379073, "epoch": 0.062108262108262105, "frac_reward_zero_std": 0.0, "grad_norm": 0.04105231538414955, "kl": 0.003928337464458309, "learning_rate": 3.0681818181818186e-06, "loss": 1.9657734810607508e-05, "num_tokens": 24714904.0, "reward": 2.0399413108825684, "reward_std": 0.7120633125305176, "rewards/code_complexity_reward/mean": 0.7279297113418579, "rewards/code_complexity_reward/std": 0.25981929898262024, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4521484375, "rewards/code_syntax_reward/std": 0.1472356915473938, "rewards/reasoning_present_reward_func/mean": 0.095703125, "rewards/reasoning_present_reward_func/std": 0.02029850147664547, "rewards/xmlcount_reward_func/mean": 0.46923828125, "rewards/xmlcount_reward_func/std": 0.07059646397829056, "step": 109, "step_time": 69.0622095791623 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 281.0859375, "completions/mean_terminated_length": 270.7183532714844, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.23977561062201858, "epoch": 0.06267806267806268, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03480077534914017, "kl": 0.003260235205743811, "learning_rate": 3.096590909090909e-06, "loss": 1.6484467778354883e-05, "num_tokens": 24924196.0, "reward": 2.2211427688598633, "reward_std": 0.7039175629615784, "rewards/code_complexity_reward/mean": 0.77197265625, "rewards/code_complexity_reward/std": 0.23043949902057648, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.466796875, "rewards/code_syntax_reward/std": 0.12461719661951065, "rewards/reasoning_present_reward_func/mean": 0.09687500447034836, "rewards/reasoning_present_reward_func/std": 0.01741628162562847, "rewards/xmlcount_reward_func/mean": 0.481201171875, "rewards/xmlcount_reward_func/std": 0.05401558429002762, "step": 110, "step_time": 65.97956762555987 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 279.736328125, "completions/mean_terminated_length": 271.2732849121094, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2497386180330068, "epoch": 0.06324786324786325, "frac_reward_zero_std": 0.0, "grad_norm": 0.0379418209195137, "kl": 0.004551016603727476, "learning_rate": 3.125e-06, "loss": 2.2692896891385317e-05, "num_tokens": 25132573.0, "reward": 2.1980957984924316, "reward_std": 0.6568918228149414, "rewards/code_complexity_reward/mean": 0.7813476324081421, "rewards/code_complexity_reward/std": 0.21099601686000824, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4716796875, "rewards/code_syntax_reward/std": 0.11569035053253174, "rewards/reasoning_present_reward_func/mean": 0.0966796875, "rewards/reasoning_present_reward_func/std": 0.01793418452143669, "rewards/xmlcount_reward_func/mean": 0.479248046875, "rewards/xmlcount_reward_func/std": 0.054429713636636734, "step": 111, "step_time": 78.20506523642689 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 303.734375, "completions/mean_terminated_length": 287.0379638671875, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.2468906466383487, "epoch": 0.06381766381766382, "frac_reward_zero_std": 0.0, "grad_norm": 0.04042204096913338, "kl": 0.003322987467981875, "learning_rate": 3.1534090909090916e-06, "loss": 1.655897358432412e-05, "num_tokens": 25356901.0, "reward": 2.0778322219848633, "reward_std": 0.7160972356796265, "rewards/code_complexity_reward/mean": 0.7244141101837158, "rewards/code_complexity_reward/std": 0.24464848637580872, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4580078125, "rewards/code_syntax_reward/std": 0.13881781697273254, "rewards/reasoning_present_reward_func/mean": 0.09511718899011612, "rewards/reasoning_present_reward_func/std": 0.0215719323605299, "rewards/xmlcount_reward_func/mean": 0.47412109375, "rewards/xmlcount_reward_func/std": 0.06497183442115784, "step": 112, "step_time": 74.16754655633122 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.029296875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 285.328125, "completions/mean_terminated_length": 278.4869079589844, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.24321692273952067, "epoch": 0.06438746438746439, "frac_reward_zero_std": 0.0, "grad_norm": 0.04391995817422867, "kl": 0.004023191309897811, "learning_rate": 3.181818181818182e-06, "loss": 2.0122271962463856e-05, "num_tokens": 25571061.0, "reward": 2.1237306594848633, "reward_std": 0.6678791046142578, "rewards/code_complexity_reward/mean": 0.764941394329071, "rewards/code_complexity_reward/std": 0.2222381979227066, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4677734375, "rewards/code_syntax_reward/std": 0.12289927154779434, "rewards/reasoning_present_reward_func/mean": 0.09511718899011612, "rewards/reasoning_present_reward_func/std": 0.02157193422317505, "rewards/xmlcount_reward_func/mean": 0.4775390625, "rewards/xmlcount_reward_func/std": 0.060944125056266785, "step": 113, "step_time": 56.89906317740679 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 304.298828125, "completions/mean_terminated_length": 290.45208740234375, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.2428118409588933, "epoch": 0.06495726495726496, "frac_reward_zero_std": 0.0, "grad_norm": 0.03003343753516674, "kl": 0.003665830023237504, "learning_rate": 3.210227272727273e-06, "loss": 1.8429767806082964e-05, "num_tokens": 25796926.0, "reward": 2.085986375808716, "reward_std": 0.6860988140106201, "rewards/code_complexity_reward/mean": 0.743945300579071, "rewards/code_complexity_reward/std": 0.23602311313152313, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4619140625, "rewards/code_syntax_reward/std": 0.13276617228984833, "rewards/reasoning_present_reward_func/mean": 0.09375, "rewards/reasoning_present_reward_func/std": 0.02422981895506382, "rewards/xmlcount_reward_func/mean": 0.471923828125, "rewards/xmlcount_reward_func/std": 0.07524623721837997, "step": 114, "step_time": 67.82089888770133 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 290.296875, "completions/mean_terminated_length": 281.2845458984375, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.24211826757527888, "epoch": 0.06552706552706553, "frac_reward_zero_std": 0.0, "grad_norm": 0.03248855471611023, "kl": 0.004022811317554442, "learning_rate": 3.2386363636363637e-06, "loss": 1.989430165849626e-05, "num_tokens": 26012846.0, "reward": 2.138964891433716, "reward_std": 0.6772894263267517, "rewards/code_complexity_reward/mean": 0.76123046875, "rewards/code_complexity_reward/std": 0.22464551031589508, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4677734375, "rewards/code_syntax_reward/std": 0.12289927154779434, "rewards/reasoning_present_reward_func/mean": 0.09648437798023224, "rewards/reasoning_present_reward_func/std": 0.01843547262251377, "rewards/xmlcount_reward_func/mean": 0.4794921875, "rewards/xmlcount_reward_func/std": 0.061630137264728546, "step": 115, "step_time": 66.74210408236831 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 305.2421875, "completions/mean_terminated_length": 292.37347412109375, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.2525792580563575, "epoch": 0.0660968660968661, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03583407774567604, "kl": 0.003826652753559756, "learning_rate": 3.267045454545455e-06, "loss": 1.9147235434502363e-05, "num_tokens": 26237618.0, "reward": 2.051318407058716, "reward_std": 0.6713321208953857, "rewards/code_complexity_reward/mean": 0.7362304925918579, "rewards/code_complexity_reward/std": 0.23728084564208984, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4619140625, "rewards/code_syntax_reward/std": 0.13276617228984833, "rewards/reasoning_present_reward_func/mean": 0.09511718899011612, "rewards/reasoning_present_reward_func/std": 0.0215719323605299, "rewards/xmlcount_reward_func/mean": 0.476806640625, "rewards/xmlcount_reward_func/std": 0.05887819081544876, "step": 116, "step_time": 68.03880220931023 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.033203125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 289.767578125, "completions/mean_terminated_length": 282.1353759765625, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.23700966988690197, "epoch": 0.06666666666666667, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03202664107084274, "kl": 0.0038824027415103046, "learning_rate": 3.2954545454545456e-06, "loss": 1.9421058823354542e-05, "num_tokens": 26453947.0, "reward": 2.213574171066284, "reward_std": 0.6327967643737793, "rewards/code_complexity_reward/mean": 0.77392578125, "rewards/code_complexity_reward/std": 0.188408762216568, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09589843451976776, "rewards/reasoning_present_reward_func/std": 0.019852032884955406, "rewards/xmlcount_reward_func/mean": 0.482421875, "rewards/xmlcount_reward_func/std": 0.05124280974268913, "step": 117, "step_time": 57.67655144352466 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 286.28125, "completions/mean_terminated_length": 273.2231140136719, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "entropy": 0.24281635810621083, "epoch": 0.06723646723646724, "frac_reward_zero_std": 0.0, "grad_norm": 0.032295964658260345, "kl": 0.0034388325384497875, "learning_rate": 3.3238636363636366e-06, "loss": 1.7342448700219393e-05, "num_tokens": 26670187.0, "reward": 2.1422364711761475, "reward_std": 0.65696781873703, "rewards/code_complexity_reward/mean": 0.7544921636581421, "rewards/code_complexity_reward/std": 0.21526694297790527, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4697265625, "rewards/code_syntax_reward/std": 0.11936526000499725, "rewards/reasoning_present_reward_func/mean": 0.09746094048023224, "rewards/reasoning_present_reward_func/std": 0.015746228396892548, "rewards/xmlcount_reward_func/mean": 0.484619140625, "rewards/xmlcount_reward_func/std": 0.05045295134186745, "step": 118, "step_time": 59.918970185332 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025390625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 270.66796875, "completions/mean_terminated_length": 264.3807678222656, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.2327547932509333, "epoch": 0.0678062678062678, "frac_reward_zero_std": 0.0, "grad_norm": 0.02757546491920948, "kl": 0.004132227957597934, "learning_rate": 3.352272727272727e-06, "loss": 2.0722945919260383e-05, "num_tokens": 26876841.0, "reward": 2.3121094703674316, "reward_std": 0.6338732242584229, "rewards/code_complexity_reward/mean": 0.7933593988418579, "rewards/code_complexity_reward/std": 0.1780928671360016, "rewards/code_execution_reward/mean": 0.451171875, "rewards/code_execution_reward/std": 0.498096764087677, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.09785155951976776, "rewards/reasoning_present_reward_func/std": 0.014513419941067696, "rewards/xmlcount_reward_func/mean": 0.4892578125, "rewards/xmlcount_reward_func/std": 0.04567214846611023, "step": 119, "step_time": 53.0842995159328 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 296.951171875, "completions/mean_terminated_length": 288.2093505859375, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.23919856594875455, "epoch": 0.06837606837606838, "frac_reward_zero_std": 0.015625, "grad_norm": 0.030881032347679138, "kl": 0.004811102897292585, "learning_rate": 3.3806818181818186e-06, "loss": 2.403767211944796e-05, "num_tokens": 27098488.0, "reward": 2.133007764816284, "reward_std": 0.6732096076011658, "rewards/code_complexity_reward/mean": 0.7544922232627869, "rewards/code_complexity_reward/std": 0.22714370489120483, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.46875, "rewards/code_syntax_reward/std": 0.12114909291267395, "rewards/reasoning_present_reward_func/mean": 0.09726562350988388, "rewards/reasoning_present_reward_func/std": 0.016324250027537346, "rewards/xmlcount_reward_func/mean": 0.482421875, "rewards/xmlcount_reward_func/std": 0.0584879070520401, "step": 120, "step_time": 60.12772897724062 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 291.087890625, "completions/mean_terminated_length": 283.961669921875, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.23854593792930245, "epoch": 0.06894586894586895, "frac_reward_zero_std": 0.0, "grad_norm": 0.029666872695088387, "kl": 0.004786501785929431, "learning_rate": 3.409090909090909e-06, "loss": 2.4025794118642807e-05, "num_tokens": 27317901.0, "reward": 2.0907716751098633, "reward_std": 0.5910921692848206, "rewards/code_complexity_reward/mean": 0.77392578125, "rewards/code_complexity_reward/std": 0.19938461482524872, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4765625, "rewards/code_syntax_reward/std": 0.10578890144824982, "rewards/reasoning_present_reward_func/mean": 0.09785155951976776, "rewards/reasoning_present_reward_func/std": 0.014513419941067696, "rewards/xmlcount_reward_func/mean": 0.484619140625, "rewards/xmlcount_reward_func/std": 0.05045295134186745, "step": 121, "step_time": 69.21082510333508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 298.619140625, "completions/mean_terminated_length": 287.2037048339844, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.241809832630679, "epoch": 0.06951566951566951, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03455536812543869, "kl": 0.005605282891337993, "learning_rate": 3.4375e-06, "loss": 2.8126887627877295e-05, "num_tokens": 27538426.0, "reward": 2.042529344558716, "reward_std": 0.6013614535331726, "rewards/code_complexity_reward/mean": 0.74609375, "rewards/code_complexity_reward/std": 0.21017272770404816, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.47265625, "rewards/code_syntax_reward/std": 0.11379580944776535, "rewards/reasoning_present_reward_func/mean": 0.09746094048023224, "rewards/reasoning_present_reward_func/std": 0.015746228396892548, "rewards/xmlcount_reward_func/mean": 0.482177734375, "rewards/xmlcount_reward_func/std": 0.055460125207901, "step": 122, "step_time": 66.44363083317876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.044921875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 284.044921875, "completions/mean_terminated_length": 273.3230895996094, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.24177940795198083, "epoch": 0.07008547008547009, "frac_reward_zero_std": 0.0, "grad_norm": 0.028510145843029022, "kl": 0.004439199146872852, "learning_rate": 3.4659090909090915e-06, "loss": 2.226616197731346e-05, "num_tokens": 27755601.0, "reward": 2.122997999191284, "reward_std": 0.6421623826026917, "rewards/code_complexity_reward/mean": 0.77099609375, "rewards/code_complexity_reward/std": 0.2136569768190384, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4716796875, "rewards/code_syntax_reward/std": 0.11569035053253174, "rewards/reasoning_present_reward_func/mean": 0.09687499701976776, "rewards/reasoning_present_reward_func/std": 0.01741628162562847, "rewards/xmlcount_reward_func/mean": 0.484619140625, "rewards/xmlcount_reward_func/std": 0.04732576012611389, "step": 123, "step_time": 67.7660019909963 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 286.39453125, "completions/mean_terminated_length": 275.2991638183594, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.25102593121118844, "epoch": 0.07065527065527065, "frac_reward_zero_std": 0.0, "grad_norm": 0.034988753497600555, "kl": 0.005088720878120512, "learning_rate": 3.494318181818182e-06, "loss": 2.5497440219623968e-05, "num_tokens": 27971011.0, "reward": 2.075000047683716, "reward_std": 0.6157544851303101, "rewards/code_complexity_reward/mean": 0.7619141340255737, "rewards/code_complexity_reward/std": 0.20757344365119934, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.470703125, "rewards/code_syntax_reward/std": 0.11754623055458069, "rewards/reasoning_present_reward_func/mean": 0.09726562350988388, "rewards/reasoning_present_reward_func/std": 0.016324250027537346, "rewards/xmlcount_reward_func/mean": 0.4833984375, "rewards/xmlcount_reward_func/std": 0.05825053155422211, "step": 124, "step_time": 68.49849692266434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 282.67578125, "completions/mean_terminated_length": 274.3198547363281, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.23939642868936062, "epoch": 0.07122507122507123, "frac_reward_zero_std": 0.015625, "grad_norm": 0.02970810793340206, "kl": 0.00484416483232053, "learning_rate": 3.522727272727273e-06, "loss": 2.4444074369966984e-05, "num_tokens": 28183789.0, "reward": 2.2084474563598633, "reward_std": 0.6492643356323242, "rewards/code_complexity_reward/mean": 0.7748047113418579, "rewards/code_complexity_reward/std": 0.2001466304063797, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4755859375, "rewards/code_syntax_reward/std": 0.10785966366529465, "rewards/reasoning_present_reward_func/mean": 0.09843750298023224, "rewards/reasoning_present_reward_func/std": 0.012414088472723961, "rewards/xmlcount_reward_func/mean": 0.488525390625, "rewards/xmlcount_reward_func/std": 0.044473692774772644, "step": 125, "step_time": 67.39528692048043 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.029296875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 284.146484375, "completions/mean_terminated_length": 277.2696228027344, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.23133662319742143, "epoch": 0.07179487179487179, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03216296806931496, "kl": 0.004378714726044564, "learning_rate": 3.5511363636363636e-06, "loss": 2.1665386157110333e-05, "num_tokens": 28399968.0, "reward": 2.163525342941284, "reward_std": 0.6065900325775146, "rewards/code_complexity_reward/mean": 0.7779296636581421, "rewards/code_complexity_reward/std": 0.19537511467933655, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.09726563096046448, "rewards/reasoning_present_reward_func/std": 0.016324250027537346, "rewards/xmlcount_reward_func/mean": 0.489501953125, "rewards/xmlcount_reward_func/std": 0.04115380719304085, "step": 126, "step_time": 61.65642487164587 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.064453125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 310.5546875, "completions/mean_terminated_length": 296.6764221191406, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.24653290049172938, "epoch": 0.07236467236467237, "frac_reward_zero_std": 0.0, "grad_norm": 0.033658117055892944, "kl": 0.005033038660258171, "learning_rate": 3.579545454545455e-06, "loss": 2.5122310034930706e-05, "num_tokens": 28632708.0, "reward": 2.0440917015075684, "reward_std": 0.6660027503967285, "rewards/code_complexity_reward/mean": 0.716015636920929, "rewards/code_complexity_reward/std": 0.2415648251771927, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.462890625, "rewards/code_syntax_reward/std": 0.1311914473772049, "rewards/reasoning_present_reward_func/mean": 0.09882812947034836, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.487060546875, "rewards/xmlcount_reward_func/std": 0.04475747048854828, "step": 127, "step_time": 73.03615812491626 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.056640625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 292.654296875, "completions/mean_terminated_length": 279.4844665527344, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.24563070316798985, "epoch": 0.07293447293447293, "frac_reward_zero_std": 0.046875, "grad_norm": 0.03206995874643326, "kl": 0.0049622617625573184, "learning_rate": 3.6079545454545456e-06, "loss": 2.4781325919320807e-05, "num_tokens": 28850731.0, "reward": 2.0475587844848633, "reward_std": 0.6151633858680725, "rewards/code_complexity_reward/mean": 0.746874988079071, "rewards/code_complexity_reward/std": 0.21880747377872467, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.4697265625, "rewards/code_syntax_reward/std": 0.11936526000499725, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843363426625729, "rewards/xmlcount_reward_func/mean": 0.48583984375, "rewards/xmlcount_reward_func/std": 0.05170458182692528, "step": 128, "step_time": 78.1614407831803 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025390625, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 284.1953125, "completions/mean_terminated_length": 278.2605285644531, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.238505243556574, "epoch": 0.0735042735042735, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03238175809383392, "kl": 0.004730640379420947, "learning_rate": 3.6363636363636366e-06, "loss": 2.3542219423688948e-05, "num_tokens": 29068111.0, "reward": 2.2044923305511475, "reward_std": 0.6031792163848877, "rewards/code_complexity_reward/mean": 0.79150390625, "rewards/code_complexity_reward/std": 0.1736876517534256, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09707031399011612, "rewards/reasoning_present_reward_func/std": 0.016880230978131294, "rewards/xmlcount_reward_func/mean": 0.48779296875, "rewards/xmlcount_reward_func/std": 0.046632301062345505, "step": 129, "step_time": 73.85559160634875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.033203125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 270.115234375, "completions/mean_terminated_length": 261.80810546875, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.24288135673850775, "epoch": 0.07407407407407407, "frac_reward_zero_std": 0.0, "grad_norm": 0.030127517879009247, "kl": 0.006149267454020446, "learning_rate": 3.6647727272727276e-06, "loss": 3.092165570706129e-05, "num_tokens": 29275282.0, "reward": 2.2410645484924316, "reward_std": 0.6243681311607361, "rewards/code_complexity_reward/mean": 0.7940429449081421, "rewards/code_complexity_reward/std": 0.18734069168567657, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.09765625, "rewards/reasoning_present_reward_func/std": 0.015143636614084244, "rewards/xmlcount_reward_func/mean": 0.489013671875, "rewards/xmlcount_reward_func/std": 0.041025906801223755, "step": 130, "step_time": 59.52848753053695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 290.921875, "completions/mean_terminated_length": 284.7068176269531, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.24114991794340312, "epoch": 0.07464387464387465, "frac_reward_zero_std": 0.015625, "grad_norm": 0.026450002565979958, "kl": 0.005201411619054852, "learning_rate": 3.6931818181818186e-06, "loss": 2.6118079404113814e-05, "num_tokens": 29493930.0, "reward": 2.098876953125, "reward_std": 0.5775098204612732, "rewards/code_complexity_reward/mean": 0.76904296875, "rewards/code_complexity_reward/std": 0.19051291048526764, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.09765625, "rewards/reasoning_present_reward_func/std": 0.015143636614084244, "rewards/xmlcount_reward_func/mean": 0.489013671875, "rewards/xmlcount_reward_func/std": 0.04853684827685356, "step": 131, "step_time": 70.33074223902076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 287.685546875, "completions/mean_terminated_length": 280.4495849609375, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.23015796090476215, "epoch": 0.07521367521367521, "frac_reward_zero_std": 0.0, "grad_norm": 0.029116477817296982, "kl": 0.0046033238104428165, "learning_rate": 3.721590909090909e-06, "loss": 2.301274798810482e-05, "num_tokens": 29710961.0, "reward": 2.21533203125, "reward_std": 0.6052196621894836, "rewards/code_complexity_reward/mean": 0.788378894329071, "rewards/code_complexity_reward/std": 0.1753852516412735, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09785155951976776, "rewards/reasoning_present_reward_func/std": 0.014513419941067696, "rewards/xmlcount_reward_func/mean": 0.490234375, "rewards/xmlcount_reward_func/std": 0.0431438647210598, "step": 132, "step_time": 54.92604566644877 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.033203125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 281.580078125, "completions/mean_terminated_length": 273.66668701171875, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.2396164764650166, "epoch": 0.07578347578347579, "frac_reward_zero_std": 0.0, "grad_norm": 0.026791486889123917, "kl": 0.007885394796176115, "learning_rate": 3.7500000000000005e-06, "loss": 3.9409613236784935e-05, "num_tokens": 29924354.0, "reward": 2.1812009811401367, "reward_std": 0.6197040677070618, "rewards/code_complexity_reward/mean": 0.7829101085662842, "rewards/code_complexity_reward/std": 0.1921551674604416, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09824219346046448, "rewards/reasoning_present_reward_func/std": 0.013154060579836369, "rewards/xmlcount_reward_func/mean": 0.489501953125, "rewards/xmlcount_reward_func/std": 0.04402561113238335, "step": 133, "step_time": 67.95832489710301 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.052734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 282.388671875, "completions/mean_terminated_length": 269.606201171875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.24680513376370072, "epoch": 0.07635327635327635, "frac_reward_zero_std": 0.015625, "grad_norm": 0.031729746609926224, "kl": 0.00592074272208265, "learning_rate": 3.7784090909090915e-06, "loss": 2.9551465559052303e-05, "num_tokens": 30137809.0, "reward": 2.163281202316284, "reward_std": 0.6705846786499023, "rewards/code_complexity_reward/mean": 0.7701172232627869, "rewards/code_complexity_reward/std": 0.2238376885652542, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.46875, "rewards/code_syntax_reward/std": 0.12114909291267395, "rewards/reasoning_present_reward_func/mean": 0.09824219346046448, "rewards/reasoning_present_reward_func/std": 0.013154060579836369, "rewards/xmlcount_reward_func/mean": 0.486328125, "rewards/xmlcount_reward_func/std": 0.046880096197128296, "step": 134, "step_time": 67.26280712801963 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 282.533203125, "completions/mean_terminated_length": 273.20526123046875, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.22744566132314503, "epoch": 0.07692307692307693, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03210160508751869, "kl": 0.005374823722377187, "learning_rate": 3.806818181818182e-06, "loss": 2.6740366593003273e-05, "num_tokens": 30351738.0, "reward": 2.186572313308716, "reward_std": 0.6299377083778381, "rewards/code_complexity_reward/mean": 0.774121105670929, "rewards/code_complexity_reward/std": 0.19039872288703918, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.09824219346046448, "rewards/reasoning_present_reward_func/std": 0.013154060579836369, "rewards/xmlcount_reward_func/mean": 0.489013671875, "rewards/xmlcount_reward_func/std": 0.04594787582755089, "step": 135, "step_time": 137.23508570436388 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.037109375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 282.16796875, "completions/mean_terminated_length": 273.3103332519531, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.24181347619742155, "epoch": 0.07749287749287749, "frac_reward_zero_std": 0.0, "grad_norm": 0.02988155372440815, "kl": 0.004454839108802844, "learning_rate": 3.8352272727272735e-06, "loss": 2.217304427176714e-05, "num_tokens": 30567944.0, "reward": 2.1541991233825684, "reward_std": 0.6373175382614136, "rewards/code_complexity_reward/mean": 0.765917956829071, "rewards/code_complexity_reward/std": 0.20959268510341644, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4736328125, "rewards/code_syntax_reward/std": 0.11186064779758453, "rewards/reasoning_present_reward_func/mean": 0.09824219346046448, "rewards/reasoning_present_reward_func/std": 0.013154060579836369, "rewards/xmlcount_reward_func/mean": 0.490234375, "rewards/xmlcount_reward_func/std": 0.03785909339785576, "step": 136, "step_time": 58.24350380618125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 284.154296875, "completions/mean_terminated_length": 277.7489929199219, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.24022808158770204, "epoch": 0.07806267806267807, "frac_reward_zero_std": 0.0, "grad_norm": 0.03318030759692192, "kl": 0.0054608193604508415, "learning_rate": 3.863636363636364e-06, "loss": 2.7130008675158024e-05, "num_tokens": 30781655.0, "reward": 2.209228515625, "reward_std": 0.6120560169219971, "rewards/code_complexity_reward/mean": 0.77685546875, "rewards/code_complexity_reward/std": 0.1807423084974289, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.0986328125, "rewards/reasoning_present_reward_func/std": 0.011623830534517765, "rewards/xmlcount_reward_func/mean": 0.490966796875, "rewards/xmlcount_reward_func/std": 0.040757179260253906, "step": 137, "step_time": 58.45725578535348 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 288.26171875, "completions/mean_terminated_length": 277.2581787109375, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.2250970690511167, "epoch": 0.07863247863247863, "frac_reward_zero_std": 0.046875, "grad_norm": 0.028804007917642593, "kl": 0.004722987887362251, "learning_rate": 3.8920454545454554e-06, "loss": 2.3610882635693997e-05, "num_tokens": 30997357.0, "reward": 2.1917481422424316, "reward_std": 0.6377124190330505, "rewards/code_complexity_reward/mean": 0.7744140625, "rewards/code_complexity_reward/std": 0.208455428481102, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.474609375, "rewards/code_syntax_reward/std": 0.10988271236419678, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.489990234375, "rewards/xmlcount_reward_func/std": 0.04127552732825279, "step": 138, "step_time": 106.83859555982053 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 273.33203125, "completions/mean_terminated_length": 266.6224670410156, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.23532405635342002, "epoch": 0.07920227920227921, "frac_reward_zero_std": 0.015625, "grad_norm": 0.029524076730012894, "kl": 0.0048986958572641015, "learning_rate": 3.9204545454545456e-06, "loss": 2.447789302095771e-05, "num_tokens": 31207663.0, "reward": 2.193603515625, "reward_std": 0.5958921313285828, "rewards/code_complexity_reward/mean": 0.783398449420929, "rewards/code_complexity_reward/std": 0.18539580702781677, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09843750298023224, "rewards/reasoning_present_reward_func/std": 0.012414088472723961, "rewards/xmlcount_reward_func/mean": 0.491455078125, "rewards/xmlcount_reward_func/std": 0.0408625453710556, "step": 139, "step_time": 69.32139460928738 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 286.94140625, "completions/mean_terminated_length": 279.68145751953125, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.2356984217185527, "epoch": 0.07977207977207977, "frac_reward_zero_std": 0.03125, "grad_norm": 0.028352031484246254, "kl": 0.00481121346456348, "learning_rate": 3.9488636363636366e-06, "loss": 2.4027482140809298e-05, "num_tokens": 31424505.0, "reward": 2.0978517532348633, "reward_std": 0.5721566677093506, "rewards/code_complexity_reward/mean": 0.7556641101837158, "rewards/code_complexity_reward/std": 0.18233413994312286, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09804687649011612, "rewards/reasoning_present_reward_func/std": 0.013851807452738285, "rewards/xmlcount_reward_func/mean": 0.4931640625, "rewards/xmlcount_reward_func/std": 0.03052297607064247, "step": 140, "step_time": 70.47402698732913 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 280.7109375, "completions/mean_terminated_length": 273.25, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.23436772916465998, "epoch": 0.08034188034188035, "frac_reward_zero_std": 0.0, "grad_norm": 0.03486361354589462, "kl": 0.006055128469597548, "learning_rate": 3.9772727272727275e-06, "loss": 3.008649218827486e-05, "num_tokens": 31637509.0, "reward": 2.2002930641174316, "reward_std": 0.611660897731781, "rewards/code_complexity_reward/mean": 0.7789062261581421, "rewards/code_complexity_reward/std": 0.18310323357582092, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.0986328125, "rewards/reasoning_present_reward_func/std": 0.011623830534517765, "rewards/xmlcount_reward_func/mean": 0.48974609375, "rewards/xmlcount_reward_func/std": 0.043030209839344025, "step": 141, "step_time": 70.86699433904141 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 265.8671875, "completions/mean_terminated_length": 260.46307373046875, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.22122781770303845, "epoch": 0.08091168091168091, "frac_reward_zero_std": 0.03125, "grad_norm": 0.02999163791537285, "kl": 0.005093448588013416, "learning_rate": 4.0056818181818185e-06, "loss": 2.5291605197708122e-05, "num_tokens": 31842937.0, "reward": 2.314257860183716, "reward_std": 0.6136627793312073, "rewards/code_complexity_reward/mean": 0.787109375, "rewards/code_complexity_reward/std": 0.16328886151313782, "rewards/code_execution_reward/mean": 0.451171875, "rewards/code_execution_reward/std": 0.498096764087677, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09843750298023224, "rewards/reasoning_present_reward_func/std": 0.012414088472723961, "rewards/xmlcount_reward_func/mean": 0.4912109375, "rewards/xmlcount_reward_func/std": 0.04043426737189293, "step": 142, "step_time": 67.32497257646173 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 283.513671875, "completions/mean_terminated_length": 278.49700927734375, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.22628093883395195, "epoch": 0.08148148148148149, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03357382491230965, "kl": 0.005384043179219589, "learning_rate": 4.0340909090909095e-06, "loss": 2.698953539947979e-05, "num_tokens": 32054920.0, "reward": 2.1967287063598633, "reward_std": 0.5997738838195801, "rewards/code_complexity_reward/mean": 0.769824206829071, "rewards/code_complexity_reward/std": 0.17340993881225586, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09902344644069672, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.490966796875, "rewards/xmlcount_reward_func/std": 0.04702700302004814, "step": 143, "step_time": 58.147213322110474 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 287.890625, "completions/mean_terminated_length": 276.86883544921875, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.24195219622924924, "epoch": 0.08205128205128205, "frac_reward_zero_std": 0.03125, "grad_norm": 0.030953196808695793, "kl": 0.006499028757389169, "learning_rate": 4.0625000000000005e-06, "loss": 3.2643351005390286e-05, "num_tokens": 32270696.0, "reward": 2.194531202316284, "reward_std": 0.684166669845581, "rewards/code_complexity_reward/mean": 0.7499023675918579, "rewards/code_complexity_reward/std": 0.2189527302980423, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.4697265625, "rewards/code_syntax_reward/std": 0.11936526000499725, "rewards/reasoning_present_reward_func/mean": 0.09843750298023224, "rewards/reasoning_present_reward_func/std": 0.012414088472723961, "rewards/xmlcount_reward_func/mean": 0.48779296875, "rewards/xmlcount_reward_func/std": 0.04980305954813957, "step": 144, "step_time": 70.28946940042078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 273.537109375, "completions/mean_terminated_length": 263.8434753417969, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.23376435972750187, "epoch": 0.08262108262108261, "frac_reward_zero_std": 0.015625, "grad_norm": 0.02829640731215477, "kl": 0.005202555934374686, "learning_rate": 4.0909090909090915e-06, "loss": 2.6086519937962294e-05, "num_tokens": 32480747.0, "reward": 2.1236815452575684, "reward_std": 0.5990386605262756, "rewards/code_complexity_reward/mean": 0.782910168170929, "rewards/code_complexity_reward/std": 0.19735512137413025, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.09882812201976776, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.490966796875, "rewards/xmlcount_reward_func/std": 0.040757179260253906, "step": 145, "step_time": 88.33725954312831 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 256.38671875, "completions/mean_terminated_length": 253.35574340820312, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.23522128351032734, "epoch": 0.08319088319088319, "frac_reward_zero_std": 0.0, "grad_norm": 0.028721405193209648, "kl": 0.008855567855789559, "learning_rate": 4.1193181818181825e-06, "loss": 4.4391112169250846e-05, "num_tokens": 32677089.0, "reward": 2.2530274391174316, "reward_std": 0.5688619017601013, "rewards/code_complexity_reward/mean": 0.808398425579071, "rewards/code_complexity_reward/std": 0.1463775634765625, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09843750298023224, "rewards/reasoning_present_reward_func/std": 0.012414088472723961, "rewards/xmlcount_reward_func/mean": 0.49365234375, "rewards/xmlcount_reward_func/std": 0.03161102160811424, "step": 146, "step_time": 67.71780115645379 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 272.62890625, "completions/mean_terminated_length": 266.8840026855469, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.23838872998021543, "epoch": 0.08376068376068375, "frac_reward_zero_std": 0.0, "grad_norm": 0.037867844104766846, "kl": 0.007141836282244185, "learning_rate": 4.1477272727272734e-06, "loss": 3.565091174095869e-05, "num_tokens": 32885891.0, "reward": 2.190136671066284, "reward_std": 0.5904514789581299, "rewards/code_complexity_reward/mean": 0.7829101085662842, "rewards/code_complexity_reward/std": 0.17205995321273804, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.494140625, "rewards/xmlcount_reward_func/std": 0.03357883170247078, "step": 147, "step_time": 66.8256057947874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.037109375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 276.787109375, "completions/mean_terminated_length": 267.72210693359375, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.23031375091522932, "epoch": 0.08433048433048433, "frac_reward_zero_std": 0.046875, "grad_norm": 0.03024953417479992, "kl": 0.008820864193694433, "learning_rate": 4.176136363636364e-06, "loss": 4.379175879876129e-05, "num_tokens": 33093534.0, "reward": 2.2403321266174316, "reward_std": 0.6411822438240051, "rewards/code_complexity_reward/mean": 0.771484375, "rewards/code_complexity_reward/std": 0.18723277747631073, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812851272523403, "rewards/xmlcount_reward_func/mean": 0.49169921875, "rewards/xmlcount_reward_func/std": 0.03397137299180031, "step": 148, "step_time": 68.3468192750588 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.037109375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 279.494140625, "completions/mean_terminated_length": 270.533447265625, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.23501223442144692, "epoch": 0.0849002849002849, "frac_reward_zero_std": 0.0, "grad_norm": 0.026934975758194923, "kl": 0.006401706166798249, "learning_rate": 4.204545454545455e-06, "loss": 3.193790325894952e-05, "num_tokens": 33307019.0, "reward": 2.1083006858825684, "reward_std": 0.5664153099060059, "rewards/code_complexity_reward/mean": 0.770703136920929, "rewards/code_complexity_reward/std": 0.1799953579902649, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09882812201976776, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.49169921875, "rewards/xmlcount_reward_func/std": 0.0389997661113739, "step": 149, "step_time": 92.1337155662477 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 291.4296875, "completions/mean_terminated_length": 285.2289123535156, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.22761371149681509, "epoch": 0.08547008547008547, "frac_reward_zero_std": 0.0, "grad_norm": 0.03148827701807022, "kl": 0.006653449276200263, "learning_rate": 4.2329545454545455e-06, "loss": 3.35160584654659e-05, "num_tokens": 33526191.0, "reward": 2.132568359375, "reward_std": 0.5795267820358276, "rewards/code_complexity_reward/mean": 0.7623047232627869, "rewards/code_complexity_reward/std": 0.18525636196136475, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843363426625729, "rewards/xmlcount_reward_func/mean": 0.490966796875, "rewards/xmlcount_reward_func/std": 0.0422309935092926, "step": 150, "step_time": 79.67152531817555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.037109375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 276.759765625, "completions/mean_terminated_length": 267.6936950683594, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.23692869115620852, "epoch": 0.08603988603988603, "frac_reward_zero_std": 0.015625, "grad_norm": 0.02877177856862545, "kl": 0.005587522278801771, "learning_rate": 4.2613636363636365e-06, "loss": 2.7910631615668535e-05, "num_tokens": 33736692.0, "reward": 2.123486280441284, "reward_std": 0.5861653685569763, "rewards/code_complexity_reward/mean": 0.7646484375, "rewards/code_complexity_reward/std": 0.18854233622550964, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.491455078125, "rewards/xmlcount_reward_func/std": 0.04304894059896469, "step": 151, "step_time": 69.57354555372149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025390625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 259.859375, "completions/mean_terminated_length": 253.29058837890625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.23230596538633108, "epoch": 0.08660968660968661, "frac_reward_zero_std": 0.0, "grad_norm": 0.03399709239602089, "kl": 0.0070000358173274435, "learning_rate": 4.2897727272727275e-06, "loss": 3.489819937385619e-05, "num_tokens": 33935924.0, "reward": 2.2806153297424316, "reward_std": 0.6160892844200134, "rewards/code_complexity_reward/mean": 0.80224609375, "rewards/code_complexity_reward/std": 0.1782987415790558, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.494384765625, "rewards/xmlcount_reward_func/std": 0.03316274285316467, "step": 152, "step_time": 75.46144266519696 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 262.517578125, "completions/mean_terminated_length": 261.0471496582031, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 0.22809100640006363, "epoch": 0.08717948717948718, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03516139090061188, "kl": 0.005782895990705583, "learning_rate": 4.3181818181818185e-06, "loss": 2.894428325816989e-05, "num_tokens": 34138101.0, "reward": 2.2538084983825684, "reward_std": 0.5409684777259827, "rewards/code_complexity_reward/mean": 0.809765636920929, "rewards/code_complexity_reward/std": 0.12359718978404999, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09882812947034836, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.03096972592175007, "step": 153, "step_time": 49.960932582616806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.041015625, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 263.802734375, "completions/mean_terminated_length": 253.1873779296875, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.21943280496634543, "epoch": 0.08774928774928775, "frac_reward_zero_std": 0.015625, "grad_norm": 0.028053345158696175, "kl": 0.006032327015418559, "learning_rate": 4.3465909090909095e-06, "loss": 3.0195398721843958e-05, "num_tokens": 34340880.0, "reward": 2.2365236282348633, "reward_std": 0.6372960805892944, "rewards/code_complexity_reward/mean": 0.7865234017372131, "rewards/code_complexity_reward/std": 0.19367195665836334, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.4912109375, "rewards/xmlcount_reward_func/std": 0.04118354618549347, "step": 154, "step_time": 61.63625087309629 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 266.203125, "completions/mean_terminated_length": 260.30401611328125, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.22768216743133962, "epoch": 0.08831908831908832, "frac_reward_zero_std": 0.015625, "grad_norm": 0.029344459995627403, "kl": 0.007504681823775172, "learning_rate": 4.3750000000000005e-06, "loss": 3.751103940885514e-05, "num_tokens": 34547752.0, "reward": 2.231689453125, "reward_std": 0.5819029211997986, "rewards/code_complexity_reward/mean": 0.7989258170127869, "rewards/code_complexity_reward/std": 0.16688647866249084, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09902344644069672, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.028075991198420525, "step": 155, "step_time": 68.59308481495827 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 275.708984375, "completions/mean_terminated_length": 267.09918212890625, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.23875601473264396, "epoch": 0.08888888888888889, "frac_reward_zero_std": 0.0, "grad_norm": 0.03037039376795292, "kl": 0.009676208290329669, "learning_rate": 4.4034090909090914e-06, "loss": 4.8516143579036e-05, "num_tokens": 34755411.0, "reward": 2.0950684547424316, "reward_std": 0.5638630986213684, "rewards/code_complexity_reward/mean": 0.775097668170929, "rewards/code_complexity_reward/std": 0.18532244861125946, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.09902344644069672, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.492431640625, "rewards/xmlcount_reward_func/std": 0.03953738510608673, "step": 156, "step_time": 82.27442093938589 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 267.390625, "completions/mean_terminated_length": 262.5179443359375, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.22207813733257353, "epoch": 0.08945868945868946, "frac_reward_zero_std": 0.0, "grad_norm": 0.02983471192419529, "kl": 0.0059995314477419015, "learning_rate": 4.4318181818181824e-06, "loss": 2.9789225663989782e-05, "num_tokens": 34963963.0, "reward": 2.23486328125, "reward_std": 0.5770822167396545, "rewards/code_complexity_reward/mean": 0.795605480670929, "rewards/code_complexity_reward/std": 0.15710574388504028, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09843750298023224, "rewards/reasoning_present_reward_func/std": 0.012414088472723961, "rewards/xmlcount_reward_func/mean": 0.4931640625, "rewards/xmlcount_reward_func/std": 0.03769467771053314, "step": 157, "step_time": 87.92740410100669 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.033203125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 274.556640625, "completions/mean_terminated_length": 266.40203857421875, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.2276715962216258, "epoch": 0.09002849002849003, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03791023790836334, "kl": 0.005838448829308618, "learning_rate": 4.460227272727273e-06, "loss": 2.9249647923279554e-05, "num_tokens": 35174248.0, "reward": 2.1624512672424316, "reward_std": 0.6219780445098877, "rewards/code_complexity_reward/mean": 0.76318359375, "rewards/code_complexity_reward/std": 0.1992574781179428, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.494384765625, "rewards/xmlcount_reward_func/std": 0.027071015909314156, "step": 158, "step_time": 86.2628696365282 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 254.4911651611328, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.22701350320130587, "epoch": 0.0905982905982906, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03278573602437973, "kl": 0.008847331151628168, "learning_rate": 4.4886363636363636e-06, "loss": 4.427321255207062e-05, "num_tokens": 35373568.0, "reward": 2.2826170921325684, "reward_std": 0.5872612595558167, "rewards/code_complexity_reward/mean": 0.8023437261581421, "rewards/code_complexity_reward/std": 0.15424323081970215, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.02327641472220421, "step": 159, "step_time": 56.27227452490479 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 267.970703125, "completions/mean_terminated_length": 264.588134765625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.2142344773747027, "epoch": 0.09116809116809117, "frac_reward_zero_std": 0.03125, "grad_norm": 0.028848862275481224, "kl": 0.006597093746677274, "learning_rate": 4.517045454545455e-06, "loss": 3.323773853480816e-05, "num_tokens": 35580233.0, "reward": 2.2391114234924316, "reward_std": 0.5575006604194641, "rewards/code_complexity_reward/mean": 0.7952147722244263, "rewards/code_complexity_reward/std": 0.1567513644695282, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660499989986, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.02948698401451111, "step": 160, "step_time": 68.68677063565701 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 268.494140625, "completions/mean_terminated_length": 263.6434326171875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2237681734841317, "epoch": 0.09173789173789174, "frac_reward_zero_std": 0.0, "grad_norm": 0.03271971642971039, "kl": 0.006673404663160909, "learning_rate": 4.5454545454545455e-06, "loss": 3.3158285077661276e-05, "num_tokens": 35786702.0, "reward": 2.204345703125, "reward_std": 0.5956704020500183, "rewards/code_complexity_reward/mean": 0.7699218988418579, "rewards/code_complexity_reward/std": 0.17668282985687256, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.493408203125, "rewards/xmlcount_reward_func/std": 0.038934629410505295, "step": 161, "step_time": 66.67008492536843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.033203125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 272.453125, "completions/mean_terminated_length": 264.2262878417969, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.22564757196232677, "epoch": 0.09230769230769231, "frac_reward_zero_std": 0.015625, "grad_norm": 0.030583854764699936, "kl": 0.0066975741283386014, "learning_rate": 4.5738636363636365e-06, "loss": 3.344437573105097e-05, "num_tokens": 35997102.0, "reward": 2.2142577171325684, "reward_std": 0.6003173589706421, "rewards/code_complexity_reward/mean": 0.800976574420929, "rewards/code_complexity_reward/std": 0.18457922339439392, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09882812947034836, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.4931640625, "rewards/xmlcount_reward_func/std": 0.03339334949851036, "step": 162, "step_time": 49.59168167691678 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 253.025390625, "completions/mean_terminated_length": 249.43565368652344, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.22492899373173714, "epoch": 0.09287749287749288, "frac_reward_zero_std": 0.0, "grad_norm": 0.030454207211732864, "kl": 0.014640436798799783, "learning_rate": 4.6022727272727275e-06, "loss": 7.321208249777555e-05, "num_tokens": 36193779.0, "reward": 2.2972166538238525, "reward_std": 0.6059843301773071, "rewards/code_complexity_reward/mean": 0.8017578125, "rewards/code_complexity_reward/std": 0.17291343212127686, "rewards/code_execution_reward/mean": 0.4140625, "rewards/code_execution_reward/std": 0.49304109811782837, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812851272523403, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02117939107120037, "step": 163, "step_time": 61.721249118447304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 260.828125, "completions/mean_terminated_length": 256.333984375, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.21715011377818882, "epoch": 0.09344729344729345, "frac_reward_zero_std": 0.015625, "grad_norm": 0.029680252075195312, "kl": 0.007391958009975497, "learning_rate": 4.6306818181818185e-06, "loss": 3.69538611266762e-05, "num_tokens": 36396995.0, "reward": 2.264404296875, "reward_std": 0.5777798295021057, "rewards/code_complexity_reward/mean": 0.796582043170929, "rewards/code_complexity_reward/std": 0.1544608622789383, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.02250281721353531, "step": 164, "step_time": 68.41828662902117 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 267.59765625, "completions/mean_terminated_length": 261.7320251464844, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22585192881524563, "epoch": 0.09401709401709402, "frac_reward_zero_std": 0.0, "grad_norm": 0.031483910977840424, "kl": 0.007720043944573263, "learning_rate": 4.6590909090909095e-06, "loss": 3.849793574772775e-05, "num_tokens": 36602373.0, "reward": 2.1920900344848633, "reward_std": 0.6040598750114441, "rewards/code_complexity_reward/mean": 0.78466796875, "rewards/code_complexity_reward/std": 0.17304763197898865, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09882812947034836, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.4921875, "rewards/xmlcount_reward_func/std": 0.04756805673241615, "step": 165, "step_time": 80.42201786395162 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 263.76171875, "completions/mean_terminated_length": 259.8214416503906, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.22946506226435304, "epoch": 0.0945868945868946, "frac_reward_zero_std": 0.0, "grad_norm": 0.027241099625825882, "kl": 0.006073450967960525, "learning_rate": 4.6875000000000004e-06, "loss": 3.0339753720909357e-05, "num_tokens": 36805731.0, "reward": 2.2533202171325684, "reward_std": 0.6045701503753662, "rewards/code_complexity_reward/mean": 0.7819335460662842, "rewards/code_complexity_reward/std": 0.16430716216564178, "rewards/code_execution_reward/mean": 0.390625, "rewards/code_execution_reward/std": 0.48836761713027954, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09882812947034836, "rewards/reasoning_present_reward_func/std": 0.010772225446999073, "rewards/xmlcount_reward_func/mean": 0.49462890625, "rewards/xmlcount_reward_func/std": 0.03711671754717827, "step": 166, "step_time": 86.41079159080982 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 264.611328125, "completions/mean_terminated_length": 259.17962646484375, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.22738393512554467, "epoch": 0.09515669515669516, "frac_reward_zero_std": 0.0, "grad_norm": 0.029343795031309128, "kl": 0.006483265347924316, "learning_rate": 4.715909090909091e-06, "loss": 3.2169336918741465e-05, "num_tokens": 37009732.0, "reward": 2.3077149391174316, "reward_std": 0.6115842461585999, "rewards/code_complexity_reward/mean": 0.80078125, "rewards/code_complexity_reward/std": 0.16340726613998413, "rewards/code_execution_reward/mean": 0.423828125, "rewards/code_execution_reward/std": 0.4946470856666565, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49462890625, "rewards/xmlcount_reward_func/std": 0.0327395424246788, "step": 167, "step_time": 77.29272319562733 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 261.716796875, "completions/mean_terminated_length": 258.7490234375, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.22753759566694498, "epoch": 0.09572649572649573, "frac_reward_zero_std": 0.0, "grad_norm": 0.03418998047709465, "kl": 0.0077688361052423716, "learning_rate": 4.744318181818182e-06, "loss": 3.873657260555774e-05, "num_tokens": 37211035.0, "reward": 2.187206983566284, "reward_std": 0.5399272441864014, "rewards/code_complexity_reward/mean": 0.79833984375, "rewards/code_complexity_reward/std": 0.1532178521156311, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03200530633330345, "step": 168, "step_time": 97.25261761527508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 267.873046875, "completions/mean_terminated_length": 263.50494384765625, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.23191841458901763, "epoch": 0.0962962962962963, "frac_reward_zero_std": 0.0, "grad_norm": 0.028814589604735374, "kl": 0.011909290085895918, "learning_rate": 4.772727272727273e-06, "loss": 5.965906893834472e-05, "num_tokens": 37418594.0, "reward": 2.2107911109924316, "reward_std": 0.5610324144363403, "rewards/code_complexity_reward/mean": 0.7894531488418579, "rewards/code_complexity_reward/std": 0.16381607949733734, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09882812947034836, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.494384765625, "rewards/xmlcount_reward_func/std": 0.02924293279647827, "step": 169, "step_time": 57.50079554878175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 247.056640625, "completions/mean_terminated_length": 244.44378662109375, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.22290028631687164, "epoch": 0.09686609686609686, "frac_reward_zero_std": 0.0, "grad_norm": 0.03025882877409458, "kl": 0.0075426404328027274, "learning_rate": 4.8011363636363635e-06, "loss": 3.7769204936921597e-05, "num_tokens": 37613591.0, "reward": 2.2734375, "reward_std": 0.5525603294372559, "rewards/code_complexity_reward/mean": 0.8135741949081421, "rewards/code_complexity_reward/std": 0.1466064751148224, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 170, "step_time": 56.10628432221711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 257.20703125, "completions/mean_terminated_length": 252.64810180664062, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.2277202382683754, "epoch": 0.09743589743589744, "frac_reward_zero_std": 0.015625, "grad_norm": 0.028972722589969635, "kl": 0.007617121427756501, "learning_rate": 4.829545454545455e-06, "loss": 3.800990816671401e-05, "num_tokens": 37813241.0, "reward": 2.328369140625, "reward_std": 0.5920333862304688, "rewards/code_complexity_reward/mean": 0.806347668170929, "rewards/code_complexity_reward/std": 0.1523583084344864, "rewards/code_execution_reward/mean": 0.435546875, "rewards/code_execution_reward/std": 0.49631330370903015, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.023742562159895897, "step": 171, "step_time": 61.41264686640352 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 265.171875, "completions/mean_terminated_length": 264.2039489746094, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.23089456208981574, "epoch": 0.098005698005698, "frac_reward_zero_std": 0.0, "grad_norm": 0.03593392297625542, "kl": 0.007174423415563069, "learning_rate": 4.8579545454545455e-06, "loss": 3.582268254831433e-05, "num_tokens": 38020105.0, "reward": 2.308642864227295, "reward_std": 0.567533016204834, "rewards/code_complexity_reward/mean": 0.8017578125, "rewards/code_complexity_reward/std": 0.13460397720336914, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.027517380192875862, "step": 172, "step_time": 57.61320408992469 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 258.927734375, "completions/mean_terminated_length": 254.39959716796875, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.21826883382163942, "epoch": 0.09857549857549858, "frac_reward_zero_std": 0.046875, "grad_norm": 0.03323492780327797, "kl": 0.009217640912538627, "learning_rate": 4.8863636363636365e-06, "loss": 4.6101456973701715e-05, "num_tokens": 38223580.0, "reward": 2.2564940452575684, "reward_std": 0.5822392106056213, "rewards/code_complexity_reward/mean": 0.7992187738418579, "rewards/code_complexity_reward/std": 0.1521836370229721, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.028498241677880287, "step": 173, "step_time": 57.92216788046062 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.029296875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 270.138671875, "completions/mean_terminated_length": 262.8390197753906, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.23073746566660702, "epoch": 0.09914529914529914, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03186408057808876, "kl": 0.012371542874461738, "learning_rate": 4.9147727272727275e-06, "loss": 6.186071550473571e-05, "num_tokens": 38430155.0, "reward": 2.230517864227295, "reward_std": 0.6064861416816711, "rewards/code_complexity_reward/mean": 0.79345703125, "rewards/code_complexity_reward/std": 0.18204954266548157, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.029144739732146263, "step": 174, "step_time": 49.588353796862066 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 259.578125, "completions/mean_terminated_length": 254.5498046875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.22275311825796962, "epoch": 0.09971509971509972, "frac_reward_zero_std": 0.015625, "grad_norm": 0.030160419642925262, "kl": 0.008789054918452166, "learning_rate": 4.9431818181818184e-06, "loss": 4.408753011375666e-05, "num_tokens": 38634155.0, "reward": 2.224609375, "reward_std": 0.5651596188545227, "rewards/code_complexity_reward/mean": 0.7861328125, "rewards/code_complexity_reward/std": 0.15190598368644714, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.023132286965847015, "step": 175, "step_time": 72.09028520714492 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 259.708984375, "completions/mean_terminated_length": 252.616455078125, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.23107722704298794, "epoch": 0.10028490028490028, "frac_reward_zero_std": 0.0, "grad_norm": 0.03222771733999252, "kl": 0.008470974178635515, "learning_rate": 4.9715909090909094e-06, "loss": 4.2402680264785886e-05, "num_tokens": 38835614.0, "reward": 2.1974120140075684, "reward_std": 0.6020060777664185, "rewards/code_complexity_reward/mean": 0.7876952886581421, "rewards/code_complexity_reward/std": 0.18179082870483398, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.491943359375, "rewards/xmlcount_reward_func/std": 0.04242851212620735, "step": 176, "step_time": 61.11085470486432 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 257.064453125, "completions/mean_terminated_length": 254.55029296875, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.22047252906486392, "epoch": 0.10085470085470086, "frac_reward_zero_std": 0.03125, "grad_norm": 0.028593501076102257, "kl": 0.006939666582184145, "learning_rate": 5e-06, "loss": 3.473396645858884e-05, "num_tokens": 39035815.0, "reward": 2.2545900344848633, "reward_std": 0.5420217514038086, "rewards/code_complexity_reward/mean": 0.8116210699081421, "rewards/code_complexity_reward/std": 0.13427601754665375, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.02048126794397831, "step": 177, "step_time": 60.83623108826578 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 273.12890625, "completions/mean_terminated_length": 265.4233703613281, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.22929529612883925, "epoch": 0.10142450142450142, "frac_reward_zero_std": 0.0, "grad_norm": 0.03140818327665329, "kl": 0.007122178663848899, "learning_rate": 4.9999950518215325e-06, "loss": 3.576950985006988e-05, "num_tokens": 39246241.0, "reward": 2.1649413108825684, "reward_std": 0.6207232475280762, "rewards/code_complexity_reward/mean": 0.7715820074081421, "rewards/code_complexity_reward/std": 0.1966152936220169, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.4931640625, "rewards/xmlcount_reward_func/std": 0.036874573677778244, "step": 178, "step_time": 80.22801981680095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 251.94921875, "completions/mean_terminated_length": 247.82144165039062, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.23114630905911326, "epoch": 0.101994301994302, "frac_reward_zero_std": 0.015625, "grad_norm": 0.051851097494363785, "kl": 0.007063601242407458, "learning_rate": 4.999980207305716e-06, "loss": 3.5310134990140796e-05, "num_tokens": 39444895.0, "reward": 2.2195801734924316, "reward_std": 0.5586240291595459, "rewards/code_complexity_reward/mean": 0.8135742545127869, "rewards/code_complexity_reward/std": 0.1501672863960266, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.0986328125, "rewards/reasoning_present_reward_func/std": 0.011623830534517765, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.035742104053497314, "step": 179, "step_time": 60.24565010238439 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 256.20703125, "completions/mean_terminated_length": 251.11155700683594, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.22522255359217525, "epoch": 0.10256410256410256, "frac_reward_zero_std": 0.046875, "grad_norm": 0.0590946339070797, "kl": 0.00781680030922871, "learning_rate": 4.9999554665113136e-06, "loss": 3.896702401107177e-05, "num_tokens": 39646217.0, "reward": 2.2045412063598633, "reward_std": 0.5924344658851624, "rewards/code_complexity_reward/mean": 0.78369140625, "rewards/code_complexity_reward/std": 0.181191548705101, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.019682783633470535, "step": 180, "step_time": 55.09398630075157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 249.8515625, "completions/mean_terminated_length": 246.21783447265625, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.22329535381868482, "epoch": 0.10313390313390314, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03612012043595314, "kl": 0.01010030059114797, "learning_rate": 4.999920829536264e-06, "loss": 5.0661954446695745e-05, "num_tokens": 39842269.0, "reward": 2.254443645477295, "reward_std": 0.5672839879989624, "rewards/code_complexity_reward/mean": 0.80078125, "rewards/code_complexity_reward/std": 0.1503400355577469, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.03503330051898956, "step": 181, "step_time": 67.5313371103257 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025390625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 268.41015625, "completions/mean_terminated_length": 262.0641174316406, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.21910062991082668, "epoch": 0.1037037037037037, "frac_reward_zero_std": 0.0, "grad_norm": 1.0953675508499146, "kl": 0.19100136246561306, "learning_rate": 4.9998762965176775e-06, "loss": 0.000954199640545994, "num_tokens": 40050231.0, "reward": 2.218994140625, "reward_std": 0.6230109930038452, "rewards/code_complexity_reward/mean": 0.76611328125, "rewards/code_complexity_reward/std": 0.1868188977241516, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.03487611562013626, "step": 182, "step_time": 50.94344433117658 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 242.978515625, "completions/mean_terminated_length": 240.325439453125, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.2161163641139865, "epoch": 0.10427350427350428, "frac_reward_zero_std": 0.03125, "grad_norm": 0.032990459352731705, "kl": 0.007100522016116884, "learning_rate": 4.999821867631841e-06, "loss": 3.546290099620819e-05, "num_tokens": 40240516.0, "reward": 2.3604001998901367, "reward_std": 0.5580582618713379, "rewards/code_complexity_reward/mean": 0.833789050579071, "rewards/code_complexity_reward/std": 0.1254381686449051, "rewards/code_execution_reward/mean": 0.435546875, "rewards/code_execution_reward/std": 0.49631330370903015, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.019755469635128975, "step": 183, "step_time": 59.4509237408638 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 250.8359375, "completions/mean_terminated_length": 248.26036071777344, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.22078685299493372, "epoch": 0.10484330484330484, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03657340258359909, "kl": 0.008316805487993406, "learning_rate": 4.999757543094213e-06, "loss": 4.16029361076653e-05, "num_tokens": 40437936.0, "reward": 2.29541015625, "reward_std": 0.5569068193435669, "rewards/code_complexity_reward/mean": 0.80712890625, "rewards/code_complexity_reward/std": 0.12909802794456482, "rewards/code_execution_reward/mean": 0.396484375, "rewards/code_execution_reward/std": 0.4896455705165863, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.015517610125243664, "step": 184, "step_time": 97.5724693601951 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 240.978515625, "completions/mean_terminated_length": 239.9156951904297, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2244284264743328, "epoch": 0.10541310541310542, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03246992826461792, "kl": 0.011796285027230624, "learning_rate": 4.999683323159425e-06, "loss": 5.892875196877867e-05, "num_tokens": 40629557.0, "reward": 2.3482909202575684, "reward_std": 0.547899603843689, "rewards/code_complexity_reward/mean": 0.8233398199081421, "rewards/code_complexity_reward/std": 0.12306099385023117, "rewards/code_execution_reward/mean": 0.43359375, "rewards/code_execution_reward/std": 0.4960552453994751, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02746524289250374, "step": 185, "step_time": 69.69704043958336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 254.876953125, "completions/mean_terminated_length": 250.27633666992188, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.22199866734445095, "epoch": 0.10598290598290598, "frac_reward_zero_std": 0.015625, "grad_norm": 0.0414302758872509, "kl": 0.010655488382326439, "learning_rate": 4.999599208121278e-06, "loss": 5.315156886354089e-05, "num_tokens": 40826638.0, "reward": 2.206005811691284, "reward_std": 0.578415036201477, "rewards/code_complexity_reward/mean": 0.7874999642372131, "rewards/code_complexity_reward/std": 0.16743138432502747, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.0382225327193737, "step": 186, "step_time": 80.5548697207123 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 241.0859375, "completions/mean_terminated_length": 240.02354431152344, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.2191325705498457, "epoch": 0.10655270655270656, "frac_reward_zero_std": 0.015625, "grad_norm": 0.030925938859581947, "kl": 0.008358314051292837, "learning_rate": 4.99950519831275e-06, "loss": 4.17516530433204e-05, "num_tokens": 41019170.0, "reward": 2.3258299827575684, "reward_std": 0.5433308482170105, "rewards/code_complexity_reward/mean": 0.8236328363418579, "rewards/code_complexity_reward/std": 0.11045221239328384, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.025197163224220276, "step": 187, "step_time": 60.23031500168145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 249.73828125, "completions/mean_terminated_length": 245.04571533203125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2283793850801885, "epoch": 0.10712250712250712, "frac_reward_zero_std": 0.03125, "grad_norm": 0.035678327083587646, "kl": 0.009606618856196292, "learning_rate": 4.999401294105978e-06, "loss": 4.813373743672855e-05, "num_tokens": 41216892.0, "reward": 2.284619092941284, "reward_std": 0.56431645154953, "rewards/code_complexity_reward/mean": 0.813671886920929, "rewards/code_complexity_reward/std": 0.13663052022457123, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.494384765625, "rewards/xmlcount_reward_func/std": 0.036665868014097214, "step": 188, "step_time": 65.31899401079863 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 256.48828125, "completions/mean_terminated_length": 252.94654846191406, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.23101114691235125, "epoch": 0.1076923076923077, "frac_reward_zero_std": 0.0, "grad_norm": 0.036713674664497375, "kl": 0.009531591174891219, "learning_rate": 4.999287495912273e-06, "loss": 4.723983147414401e-05, "num_tokens": 41418326.0, "reward": 2.203662157058716, "reward_std": 0.5604984164237976, "rewards/code_complexity_reward/mean": 0.7937500476837158, "rewards/code_complexity_reward/std": 0.16355876624584198, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.023893006145954132, "step": 189, "step_time": 66.14895130135119 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 243.607421875, "completions/mean_terminated_length": 239.8871307373047, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.2300442266277969, "epoch": 0.10826210826210826, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03727519512176514, "kl": 0.029726511784247123, "learning_rate": 4.99916380418211e-06, "loss": 0.00014863675460219383, "num_tokens": 41613549.0, "reward": 2.2262697219848633, "reward_std": 0.5375033020973206, "rewards/code_complexity_reward/mean": 0.8240234851837158, "rewards/code_complexity_reward/std": 0.1401320993900299, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.017314758151769638, "step": 190, "step_time": 57.57137386407703 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.033203125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 255.849609375, "completions/mean_terminated_length": 247.0525360107422, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.22313991701230407, "epoch": 0.10883190883190884, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03254648670554161, "kl": 0.00941066701489035, "learning_rate": 4.999030219405129e-06, "loss": 4.696586984209716e-05, "num_tokens": 41813424.0, "reward": 2.2369630336761475, "reward_std": 0.6173927783966064, "rewards/code_complexity_reward/mean": 0.7802734375, "rewards/code_complexity_reward/std": 0.1869298815727234, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.023651836439967155, "step": 191, "step_time": 60.99516074731946 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 258.099609375, "completions/mean_terminated_length": 253.04183959960938, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.2186401216313243, "epoch": 0.1094017094017094, "frac_reward_zero_std": 0.0, "grad_norm": 0.048880986869335175, "kl": 0.008909617710742168, "learning_rate": 4.998886742110129e-06, "loss": 4.460039781406522e-05, "num_tokens": 42018011.0, "reward": 2.227099657058716, "reward_std": 0.5684821605682373, "rewards/code_complexity_reward/mean": 0.8028320074081421, "rewards/code_complexity_reward/std": 0.1570483148097992, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.03043578378856182, "step": 192, "step_time": 75.16699422895908 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 254.46484375, "completions/mean_terminated_length": 250.3769989013672, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.2232261870522052, "epoch": 0.10997150997150996, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03342365473508835, "kl": 0.008618419233243912, "learning_rate": 4.998733372865072e-06, "loss": 4.3373991502448916e-05, "num_tokens": 42215657.0, "reward": 2.1732423305511475, "reward_std": 0.5663158297538757, "rewards/code_complexity_reward/mean": 0.7897460460662842, "rewards/code_complexity_reward/std": 0.17183633148670197, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.03194179758429527, "step": 193, "step_time": 58.18244500271976 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 239.544921875, "completions/mean_terminated_length": 236.8579864501953, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.22969416179694235, "epoch": 0.11054131054131054, "frac_reward_zero_std": 0.015625, "grad_norm": 0.032569289207458496, "kl": 0.0092943570780335, "learning_rate": 4.998570112277077e-06, "loss": 4.658516263589263e-05, "num_tokens": 42404520.0, "reward": 2.2312989234924316, "reward_std": 0.5496713519096375, "rewards/code_complexity_reward/mean": 0.8116210699081421, "rewards/code_complexity_reward/std": 0.14102889597415924, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03050634078681469, "step": 194, "step_time": 70.43878047727048 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 251.66796875, "completions/mean_terminated_length": 249.10060119628906, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.22760605602525175, "epoch": 0.1111111111111111, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03183756396174431, "kl": 0.009347907907795161, "learning_rate": 4.998396960992419e-06, "loss": 4.653599171433598e-05, "num_tokens": 42602758.0, "reward": 2.2394533157348633, "reward_std": 0.5361547470092773, "rewards/code_complexity_reward/mean": 0.814160168170929, "rewards/code_complexity_reward/std": 0.13326282799243927, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.018998827785253525, "step": 195, "step_time": 68.17053152620792 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 252.7265625, "completions/mean_terminated_length": 249.13267517089844, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.22301403596065938, "epoch": 0.11168091168091168, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03325219079852104, "kl": 0.01062687306693988, "learning_rate": 4.998213919696522e-06, "loss": 5.32890553586185e-05, "num_tokens": 42798146.0, "reward": 2.219189405441284, "reward_std": 0.5558809041976929, "rewards/code_complexity_reward/mean": 0.798632800579071, "rewards/code_complexity_reward/std": 0.14691239595413208, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.018141774460673332, "step": 196, "step_time": 87.90138953179121 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 239.708984375, "completions/mean_terminated_length": 238.10414123535156, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.23400262161158025, "epoch": 0.11225071225071225, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03745296970009804, "kl": 0.010279818707203958, "learning_rate": 4.998020989113965e-06, "loss": 5.144948954693973e-05, "num_tokens": 42987269.0, "reward": 2.28564453125, "reward_std": 0.546928346157074, "rewards/code_complexity_reward/mean": 0.824999988079071, "rewards/code_complexity_reward/std": 0.12601836025714874, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02333279326558113, "step": 197, "step_time": 108.03134009148926 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 257.490234375, "completions/mean_terminated_length": 252.4203338623047, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.2291192498523742, "epoch": 0.11282051282051282, "frac_reward_zero_std": 0.078125, "grad_norm": 0.031891729682683945, "kl": 0.010313205886632204, "learning_rate": 4.997818170008471e-06, "loss": 5.143114321981557e-05, "num_tokens": 43190584.0, "reward": 2.1617674827575684, "reward_std": 0.5722265243530273, "rewards/code_complexity_reward/mean": 0.7876952886581421, "rewards/code_complexity_reward/std": 0.18130576610565186, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.493408203125, "rewards/xmlcount_reward_func/std": 0.037330903112888336, "step": 198, "step_time": 88.93806058261544 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 230.306640625, "completions/mean_terminated_length": 227.52859497070312, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.22543807746842504, "epoch": 0.11339031339031339, "frac_reward_zero_std": 0.015625, "grad_norm": 0.031524937599897385, "kl": 0.010344042377255391, "learning_rate": 4.9976054631829085e-06, "loss": 5.171092925593257e-05, "num_tokens": 43375133.0, "reward": 2.34326171875, "reward_std": 0.5729178190231323, "rewards/code_complexity_reward/mean": 0.8267577886581421, "rewards/code_complexity_reward/std": 0.1421695053577423, "rewards/code_execution_reward/mean": 0.42578125, "rewards/code_execution_reward/std": 0.4949444830417633, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 199, "step_time": 58.232906410470605 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 252.7109375, "completions/mean_terminated_length": 250.1538543701172, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.22604031371884048, "epoch": 0.11396011396011396, "frac_reward_zero_std": 0.03125, "grad_norm": 0.036410123109817505, "kl": 0.01414657874556724, "learning_rate": 4.997382869479286e-06, "loss": 7.068936974974349e-05, "num_tokens": 43574505.0, "reward": 2.150390625, "reward_std": 0.553347647190094, "rewards/code_complexity_reward/mean": 0.7862304449081421, "rewards/code_complexity_reward/std": 0.16833724081516266, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.021983280777931213, "step": 200, "step_time": 56.5386199709028 }, { "epoch": 0.11396011396011396, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.01875, "eval_completions/max_length": 337.49, "eval_completions/max_terminated_length": 327.3, "eval_completions/mean_length": 247.76375, "eval_completions/mean_terminated_length": 244.83153671264648, "eval_completions/min_length": 176.11, "eval_completions/min_terminated_length": 176.11, "eval_entropy": 0.22615335285663604, "eval_frac_reward_zero_std": 0.01, "eval_kl": 0.01041782246902585, "eval_loss": -0.0014950978802517056, "eval_num_tokens": 43574505.0, "eval_reward": 2.178437511920929, "eval_reward_std": 0.3009494418092072, "eval_rewards/code_complexity_reward/mean": 0.7948125046491623, "eval_rewards/code_complexity_reward/std": 0.09499903708696365, "eval_rewards/code_execution_reward/mean": 0.30125, "eval_rewards/code_execution_reward/std": 0.20394095689058303, "eval_rewards/code_syntax_reward/mean": 0.48625, "eval_rewards/code_syntax_reward/std": 0.03345976293087006, "eval_rewards/reasoning_present_reward_func/mean": 0.0998750015348196, "eval_rewards/reasoning_present_reward_func/std": 0.000353553406894207, "eval_rewards/xmlcount_reward_func/mean": 0.49625, "eval_rewards/xmlcount_reward_func/std": 0.008638332411646844, "eval_runtime": 1558.6461, "eval_samples_per_second": 0.064, "eval_steps_per_second": 0.008, "step": 200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 246.943359375, "completions/mean_terminated_length": 244.85629272460938, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.22361276322044432, "epoch": 0.11452991452991453, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03531897813081741, "kl": 0.01345664422842674, "learning_rate": 4.997150389778752e-06, "loss": 6.715253402944654e-05, "num_tokens": 43773052.0, "reward": 2.254199266433716, "reward_std": 0.559268057346344, "rewards/code_complexity_reward/mean": 0.787109375, "rewards/code_complexity_reward/std": 0.14365185797214508, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.021983280777931213, "step": 201, "step_time": 58.662919039838016 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 258.46875, "completions/mean_terminated_length": 253.93238830566406, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2303747448604554, "epoch": 0.1150997150997151, "frac_reward_zero_std": 0.0625, "grad_norm": 0.03256956487894058, "kl": 0.014936520990886493, "learning_rate": 4.996908025001584e-06, "loss": 7.483505760319531e-05, "num_tokens": 43975052.0, "reward": 2.1961426734924316, "reward_std": 0.5536558032035828, "rewards/code_complexity_reward/mean": 0.795117199420929, "rewards/code_complexity_reward/std": 0.1523321121931076, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.018141774460673332, "step": 202, "step_time": 57.63747015129775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 246.759765625, "completions/mean_terminated_length": 245.19647216796875, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.22250059596262872, "epoch": 0.11566951566951567, "frac_reward_zero_std": 0.015625, "grad_norm": 0.031841784715652466, "kl": 0.009594220675353426, "learning_rate": 4.9966557761071976e-06, "loss": 4.785860073752701e-05, "num_tokens": 44172153.0, "reward": 2.276074171066284, "reward_std": 0.5696466565132141, "rewards/code_complexity_reward/mean": 0.808886706829071, "rewards/code_complexity_reward/std": 0.14195355772972107, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.025770151987671852, "step": 203, "step_time": 60.944370582699776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 257.4296875, "completions/mean_terminated_length": 254.91912841796875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.22625490417703986, "epoch": 0.11623931623931624, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03309253603219986, "kl": 0.00973151048674481, "learning_rate": 4.996393644094128e-06, "loss": 4.884973168373108e-05, "num_tokens": 44372501.0, "reward": 2.1736817359924316, "reward_std": 0.5532168745994568, "rewards/code_complexity_reward/mean": 0.7791992425918579, "rewards/code_complexity_reward/std": 0.16121463477611542, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.01981583796441555, "step": 204, "step_time": 78.88559730816633 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 254.115234375, "completions/mean_terminated_length": 251.57199096679688, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.2192483462858945, "epoch": 0.1168091168091168, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03274528309702873, "kl": 0.010346571209083777, "learning_rate": 4.996121630000035e-06, "loss": 5.1835639169439673e-05, "num_tokens": 44573728.0, "reward": 2.2074220180511475, "reward_std": 0.5093414187431335, "rewards/code_complexity_reward/mean": 0.8034179210662842, "rewards/code_complexity_reward/std": 0.12312403321266174, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.017314758151769638, "step": 205, "step_time": 59.56771903298795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 222.955078125, "completions/mean_terminated_length": 221.8215789794922, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.2179740930441767, "epoch": 0.11737891737891738, "frac_reward_zero_std": 0.015625, "grad_norm": 0.036799170076847076, "kl": 0.01111530917842174, "learning_rate": 4.995839734901701e-06, "loss": 5.5736396461725235e-05, "num_tokens": 44755865.0, "reward": 2.349414110183716, "reward_std": 0.5487751960754395, "rewards/code_complexity_reward/mean": 0.8306640386581421, "rewards/code_complexity_reward/std": 0.1238139346241951, "rewards/code_execution_reward/mean": 0.423828125, "rewards/code_execution_reward/std": 0.4946470856666565, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 206, "step_time": 79.14148010872304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 252.021484375, "completions/mean_terminated_length": 249.9744110107422, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.21861835452727973, "epoch": 0.11794871794871795, "frac_reward_zero_std": 0.015625, "grad_norm": 0.036016304045915604, "kl": 0.011288708505162504, "learning_rate": 4.995547959915018e-06, "loss": 5.62734785489738e-05, "num_tokens": 44953468.0, "reward": 2.241162061691284, "reward_std": 0.5551870465278625, "rewards/code_complexity_reward/mean": 0.8029296398162842, "rewards/code_complexity_reward/std": 0.14087095856666565, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.030623575672507286, "step": 207, "step_time": 66.36230265256017 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 225.64453125, "completions/mean_terminated_length": 223.3897705078125, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.22810390614904463, "epoch": 0.11851851851851852, "frac_reward_zero_std": 0.015625, "grad_norm": 0.037395209074020386, "kl": 0.01184777374874102, "learning_rate": 4.9952463061949905e-06, "loss": 5.9418962337076664e-05, "num_tokens": 45137942.0, "reward": 2.3270020484924316, "reward_std": 0.5656158924102783, "rewards/code_complexity_reward/mean": 0.815625011920929, "rewards/code_complexity_reward/std": 0.14456899464130402, "rewards/code_execution_reward/mean": 0.419921875, "rewards/code_execution_reward/std": 0.4940285086631775, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 208, "step_time": 67.80448626726866 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 237.458984375, "completions/mean_terminated_length": 236.92172241210938, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.2345377302262932, "epoch": 0.11908831908831909, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03549792617559433, "kl": 0.012629249846213497, "learning_rate": 4.994934774935728e-06, "loss": 6.319255044218153e-05, "num_tokens": 45325857.0, "reward": 2.2818846702575684, "reward_std": 0.5721438527107239, "rewards/code_complexity_reward/mean": 0.80615234375, "rewards/code_complexity_reward/std": 0.1500692516565323, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 209, "step_time": 64.4466890739277 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 250.40234375, "completions/mean_terminated_length": 248.34251403808594, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22112784977070987, "epoch": 0.11965811965811966, "frac_reward_zero_std": 0.03125, "grad_norm": 0.036498334258794785, "kl": 0.010202210236457177, "learning_rate": 4.994613367370438e-06, "loss": 5.0947710406035185e-05, "num_tokens": 45525527.0, "reward": 2.19287109375, "reward_std": 0.5097200274467468, "rewards/code_complexity_reward/mean": 0.8082031011581421, "rewards/code_complexity_reward/std": 0.12794895470142365, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.03380218520760536, "step": 210, "step_time": 119.90158825740218 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 242.724609375, "completions/mean_terminated_length": 240.06903076171875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.21953152609057724, "epoch": 0.12022792022792023, "frac_reward_zero_std": 0.046875, "grad_norm": 0.03251488506793976, "kl": 0.009934089132002555, "learning_rate": 4.994282084771429e-06, "loss": 4.979909863322973e-05, "num_tokens": 45717642.0, "reward": 2.3020505905151367, "reward_std": 0.5441182851791382, "rewards/code_complexity_reward/mean": 0.8101562261581421, "rewards/code_complexity_reward/std": 0.12568549811840057, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 211, "step_time": 69.99835882615298 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 242.998046875, "completions/mean_terminated_length": 240.3451690673828, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.2230339793022722, "epoch": 0.1207977207977208, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03227604553103447, "kl": 0.017901916478876956, "learning_rate": 4.9939409284500955e-06, "loss": 8.965832239482552e-05, "num_tokens": 45912809.0, "reward": 2.2486329078674316, "reward_std": 0.547955334186554, "rewards/code_complexity_reward/mean": 0.8047851324081421, "rewards/code_complexity_reward/std": 0.13810129463672638, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.017314758151769638, "step": 212, "step_time": 107.69361136015505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 480.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 230.00390625, "completions/mean_terminated_length": 230.00390625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.20922037702985108, "epoch": 0.12136752136752137, "frac_reward_zero_std": 0.046875, "grad_norm": 0.032484445720911026, "kl": 0.010990127004333772, "learning_rate": 4.9935898997569195e-06, "loss": 5.496549420058727e-05, "num_tokens": 46096875.0, "reward": 2.338916063308716, "reward_std": 0.5478636622428894, "rewards/code_complexity_reward/mean": 0.8236328363418579, "rewards/code_complexity_reward/std": 0.11865226924419403, "rewards/code_execution_reward/mean": 0.41796875, "rewards/code_execution_reward/std": 0.4937073290348053, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 213, "step_time": 61.48407945409417 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 243.951171875, "completions/mean_terminated_length": 241.3076934814453, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.21876182965934277, "epoch": 0.12193732193732194, "frac_reward_zero_std": 0.015625, "grad_norm": 0.038339756429195404, "kl": 0.012270642328076065, "learning_rate": 4.993229000081465e-06, "loss": 6.111207039793953e-05, "num_tokens": 46291282.0, "reward": 2.1668944358825684, "reward_std": 0.5066198110580444, "rewards/code_complexity_reward/mean": 0.79345703125, "rewards/code_complexity_reward/std": 0.13678286969661713, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 214, "step_time": 50.150592640042305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 228.033203125, "completions/mean_terminated_length": 224.666015625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22810626775026321, "epoch": 0.1225071225071225, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03448980301618576, "kl": 0.011798999061284121, "learning_rate": 4.992858230852366e-06, "loss": 5.897751543670893e-05, "num_tokens": 46474627.0, "reward": 2.3203125, "reward_std": 0.5716325044631958, "rewards/code_complexity_reward/mean": 0.825390636920929, "rewards/code_complexity_reward/std": 0.132379412651062, "rewards/code_execution_reward/mean": 0.40625, "rewards/code_execution_reward/std": 0.49161264300346375, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.032946839928627014, "step": 215, "step_time": 80.73108415305614 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 227.958984375, "completions/mean_terminated_length": 226.84510803222656, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.23014864465221763, "epoch": 0.12307692307692308, "frac_reward_zero_std": 0.046875, "grad_norm": 0.038807302713394165, "kl": 0.013145415439794306, "learning_rate": 4.99247759353733e-06, "loss": 6.575696170330048e-05, "num_tokens": 46662070.0, "reward": 2.2679686546325684, "reward_std": 0.5611412525177002, "rewards/code_complexity_reward/mean": 0.8119140267372131, "rewards/code_complexity_reward/std": 0.15045477449893951, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 216, "step_time": 89.1671647252515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 508.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 223.77734375, "completions/mean_terminated_length": 223.77734375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2220143375452608, "epoch": 0.12364672364672365, "frac_reward_zero_std": 0.046875, "grad_norm": 0.03797873109579086, "kl": 0.013833445860655047, "learning_rate": 4.992087089643128e-06, "loss": 6.929173832759261e-05, "num_tokens": 46845092.0, "reward": 2.3732423782348633, "reward_std": 0.5845611095428467, "rewards/code_complexity_reward/mean": 0.8212890625, "rewards/code_complexity_reward/std": 0.13758434355258942, "rewards/code_execution_reward/mean": 0.4609375, "rewards/code_execution_reward/std": 0.4989593029022217, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.028128057718276978, "step": 217, "step_time": 58.31205150298774 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 224.357421875, "completions/mean_terminated_length": 222.66209411621094, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.21589902602136135, "epoch": 0.12421652421652421, "frac_reward_zero_std": 0.046875, "grad_norm": 0.03269258141517639, "kl": 0.01790780468581943, "learning_rate": 4.991686720715581e-06, "loss": 8.957673708209768e-05, "num_tokens": 47027347.0, "reward": 2.341552734375, "reward_std": 0.5552327632904053, "rewards/code_complexity_reward/mean": 0.8262695074081421, "rewards/code_complexity_reward/std": 0.12720981240272522, "rewards/code_execution_reward/mean": 0.423828125, "rewards/code_execution_reward/std": 0.4946470856666565, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.014529787935316563, "step": 218, "step_time": 69.09449215326458 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 237.921875, "completions/mean_terminated_length": 235.21893310546875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.22107094037346542, "epoch": 0.12478632478632479, "frac_reward_zero_std": 0.0, "grad_norm": 0.04637696221470833, "kl": 0.014032901075552218, "learning_rate": 4.9912764883395715e-06, "loss": 7.013033609837294e-05, "num_tokens": 47218555.0, "reward": 2.189697265625, "reward_std": 0.5303761959075928, "rewards/code_complexity_reward/mean": 0.7959960699081421, "rewards/code_complexity_reward/std": 0.14496974647045135, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 219, "step_time": 57.60388054139912 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 236.654296875, "completions/mean_terminated_length": 234.48622131347656, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.21863399422727525, "epoch": 0.12535612535612536, "frac_reward_zero_std": 0.0, "grad_norm": 0.04483490064740181, "kl": 0.012307901837630197, "learning_rate": 4.99085639413902e-06, "loss": 6.137541640782729e-05, "num_tokens": 47406242.0, "reward": 2.265429735183716, "reward_std": 0.5547837018966675, "rewards/code_complexity_reward/mean": 0.8021484613418579, "rewards/code_complexity_reward/std": 0.13899722695350647, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.032946839928627014, "step": 220, "step_time": 58.18419759999961 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 237.40234375, "completions/mean_terminated_length": 235.78390502929688, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.2229446256533265, "epoch": 0.1259259259259259, "frac_reward_zero_std": 0.015625, "grad_norm": 0.036203671246767044, "kl": 0.013815741025609896, "learning_rate": 4.990426439776886e-06, "loss": 6.899406434968114e-05, "num_tokens": 47599304.0, "reward": 2.167773485183716, "reward_std": 0.5119860172271729, "rewards/code_complexity_reward/mean": 0.7989258170127869, "rewards/code_complexity_reward/std": 0.1397293359041214, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03480498120188713, "step": 221, "step_time": 68.18977180402726 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 235.955078125, "completions/mean_terminated_length": 231.57342529296875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.21527987229637802, "epoch": 0.1264957264957265, "frac_reward_zero_std": 0.03125, "grad_norm": 0.033039648085832596, "kl": 0.014827143771981355, "learning_rate": 4.989986626955165e-06, "loss": 7.415096479235217e-05, "num_tokens": 47790329.0, "reward": 2.261474609375, "reward_std": 0.5762601494789124, "rewards/code_complexity_reward/mean": 0.804492175579071, "rewards/code_complexity_reward/std": 0.16117636859416962, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.016442574560642242, "step": 222, "step_time": 53.04466658271849 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 227.53125, "completions/mean_terminated_length": 225.85462951660156, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.22391229961067438, "epoch": 0.12706552706552707, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04215112328529358, "kl": 0.013388700950599741, "learning_rate": 4.989536957414874e-06, "loss": 6.681165541522205e-05, "num_tokens": 47973937.0, "reward": 2.3087401390075684, "reward_std": 0.5296127796173096, "rewards/code_complexity_reward/mean": 0.8232421875, "rewards/code_complexity_reward/std": 0.12222190946340561, "rewards/code_execution_reward/mean": 0.390625, "rewards/code_execution_reward/std": 0.48836761713027954, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 223, "step_time": 74.74048539530486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 219.0859375, "completions/mean_terminated_length": 216.19723510742188, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.21355437603779137, "epoch": 0.12763532763532764, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04449039325118065, "kl": 0.01694991151452996, "learning_rate": 4.989077432936048e-06, "loss": 8.495530346408486e-05, "num_tokens": 48153973.0, "reward": 2.3680663108825684, "reward_std": 0.5686469674110413, "rewards/code_complexity_reward/mean": 0.8369140625, "rewards/code_complexity_reward/std": 0.13412266969680786, "rewards/code_execution_reward/mean": 0.439453125, "rewards/code_execution_reward/std": 0.49680593609809875, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 224, "step_time": 49.85627874918282 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 223.462890625, "completions/mean_terminated_length": 220.6173553466797, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.22821127902716398, "epoch": 0.1282051282051282, "frac_reward_zero_std": 0.0625, "grad_norm": 0.03957195580005646, "kl": 0.016121354085044004, "learning_rate": 4.988608055337735e-06, "loss": 8.041126420721412e-05, "num_tokens": 48335450.0, "reward": 2.2999022006988525, "reward_std": 0.5787263512611389, "rewards/code_complexity_reward/mean": 0.8099609613418579, "rewards/code_complexity_reward/std": 0.14922630786895752, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 225, "step_time": 62.62576558906585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 239.1328125, "completions/mean_terminated_length": 233.6972198486328, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.22805376583710313, "epoch": 0.12877492877492877, "frac_reward_zero_std": 0.03125, "grad_norm": 0.04009047523140907, "kl": 0.015848876952077262, "learning_rate": 4.9881288264779865e-06, "loss": 7.927583646960557e-05, "num_tokens": 48528198.0, "reward": 2.15576171875, "reward_std": 0.5534569621086121, "rewards/code_complexity_reward/mean": 0.7940429449081421, "rewards/code_complexity_reward/std": 0.1780209243297577, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.02048126794397831, "step": 226, "step_time": 57.02042163815349 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 230.759765625, "completions/mean_terminated_length": 228.54527282714844, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.22085609612986445, "epoch": 0.12934472934472935, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03421340882778168, "kl": 0.016283621269394644, "learning_rate": 4.98763974825385e-06, "loss": 8.125983003992587e-05, "num_tokens": 48715003.0, "reward": 2.201904296875, "reward_std": 0.530733585357666, "rewards/code_complexity_reward/mean": 0.7999023199081421, "rewards/code_complexity_reward/std": 0.14776545763015747, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.014529787935316563, "step": 227, "step_time": 57.79005901608616 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 227.55078125, "completions/mean_terminated_length": 224.17787170410156, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22435043775476515, "epoch": 0.12991452991452992, "frac_reward_zero_std": 0.015625, "grad_norm": 0.03924553096294403, "kl": 0.017392973604728468, "learning_rate": 4.987140822601363e-06, "loss": 8.708945824764669e-05, "num_tokens": 48901813.0, "reward": 2.2813477516174316, "reward_std": 0.5535875558853149, "rewards/code_complexity_reward/mean": 0.818066418170929, "rewards/code_complexity_reward/std": 0.14795000851154327, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 228, "step_time": 103.14013423305005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 232.439453125, "completions/mean_terminated_length": 229.6824493408203, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.23099598730914295, "epoch": 0.13048433048433047, "frac_reward_zero_std": 0.0, "grad_norm": 0.03868521377444267, "kl": 0.01591534333419986, "learning_rate": 4.986632051495544e-06, "loss": 7.983222167240456e-05, "num_tokens": 49090934.0, "reward": 2.2428712844848633, "reward_std": 0.5527112483978271, "rewards/code_complexity_reward/mean": 0.81982421875, "rewards/code_complexity_reward/std": 0.1417067050933838, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 229, "step_time": 67.69246964622289 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 236.765625, "completions/mean_terminated_length": 232.39683532714844, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.22699373424984515, "epoch": 0.13105413105413105, "frac_reward_zero_std": 0.03125, "grad_norm": 0.038089267909526825, "kl": 0.03206017654156312, "learning_rate": 4.986113436950385e-06, "loss": 0.00016046559903770685, "num_tokens": 49280270.0, "reward": 2.1252927780151367, "reward_std": 0.5357201099395752, "rewards/code_complexity_reward/mean": 0.7884765863418579, "rewards/code_complexity_reward/std": 0.16559216380119324, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.02449164353311062, "step": 230, "step_time": 79.56046002265066 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 228.208984375, "completions/mean_terminated_length": 223.13121032714844, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.22765685222111642, "epoch": 0.13162393162393163, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03820062428712845, "kl": 0.019607209891546518, "learning_rate": 4.985584981018844e-06, "loss": 9.796026279218495e-05, "num_tokens": 49465617.0, "reward": 2.2608399391174316, "reward_std": 0.5770052671432495, "rewards/code_complexity_reward/mean": 0.798535168170929, "rewards/code_complexity_reward/std": 0.16317784786224365, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.01892954669892788, "step": 231, "step_time": 78.99627125822008 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 226.794921875, "completions/mean_terminated_length": 224.54920959472656, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.2206795436795801, "epoch": 0.1321937321937322, "frac_reward_zero_std": 0.015625, "grad_norm": 0.037624239921569824, "kl": 0.018486932385712862, "learning_rate": 4.985046685792836e-06, "loss": 9.241003863280639e-05, "num_tokens": 49651264.0, "reward": 2.2162108421325684, "reward_std": 0.5148084759712219, "rewards/code_complexity_reward/mean": 0.8166015148162842, "rewards/code_complexity_reward/std": 0.12628118693828583, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 232, "step_time": 135.97053617518395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 229.302734375, "completions/mean_terminated_length": 227.6365509033203, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.22152003133669496, "epoch": 0.13276353276353275, "frac_reward_zero_std": 0.0, "grad_norm": 0.036677926778793335, "kl": 0.020134654274443164, "learning_rate": 4.984498553403225e-06, "loss": 0.00010104046668857336, "num_tokens": 49837315.0, "reward": 2.191357374191284, "reward_std": 0.5165649056434631, "rewards/code_complexity_reward/mean": 0.810546875, "rewards/code_complexity_reward/std": 0.13151270151138306, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 233, "step_time": 69.76579935662448 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 214.052734375, "completions/mean_terminated_length": 213.46966552734375, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.2315221552271396, "epoch": 0.13333333333333333, "frac_reward_zero_std": 0.09375, "grad_norm": 0.040634091943502426, "kl": 0.01983856820152141, "learning_rate": 4.983940586019819e-06, "loss": 9.900690929498523e-05, "num_tokens": 50013942.0, "reward": 2.297119140625, "reward_std": 0.5255485773086548, "rewards/code_complexity_reward/mean": 0.8311523199081421, "rewards/code_complexity_reward/std": 0.10921907424926758, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 234, "step_time": 65.9157869713381 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 208.314453125, "completions/mean_terminated_length": 207.12353515625, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.22595178917981684, "epoch": 0.1339031339031339, "frac_reward_zero_std": 0.0625, "grad_norm": 0.03688649833202362, "kl": 0.022013053108821623, "learning_rate": 4.983372785851354e-06, "loss": 0.00011018158693332225, "num_tokens": 50189327.0, "reward": 2.315380811691284, "reward_std": 0.5729033350944519, "rewards/code_complexity_reward/mean": 0.8271484375, "rewards/code_complexity_reward/std": 0.14418022334575653, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.022693097591400146, "step": 235, "step_time": 56.70779230631888 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 229.78515625, "completions/mean_terminated_length": 226.43875122070312, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2219766592606902, "epoch": 0.13447293447293449, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03743145242333412, "kl": 0.02003083903400693, "learning_rate": 4.982795155145491e-06, "loss": 9.999758913181722e-05, "num_tokens": 50375609.0, "reward": 2.2469725608825684, "reward_std": 0.5720714926719666, "rewards/code_complexity_reward/mean": 0.8112304210662842, "rewards/code_complexity_reward/std": 0.15258967876434326, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.015517610125243664, "step": 236, "step_time": 55.346920476295054 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 437.0, "completions/max_terminated_length": 437.0, "completions/mean_length": 206.556640625, "completions/mean_terminated_length": 206.556640625, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.21693312865681946, "epoch": 0.13504273504273503, "frac_reward_zero_std": 0.046875, "grad_norm": 0.04374729469418526, "kl": 0.019589014613302425, "learning_rate": 4.9822076961888065e-06, "loss": 9.818043326959014e-05, "num_tokens": 50550286.0, "reward": 2.3560056686401367, "reward_std": 0.5492672324180603, "rewards/code_complexity_reward/mean": 0.8285156488418579, "rewards/code_complexity_reward/std": 0.12389370799064636, "rewards/code_execution_reward/mean": 0.431640625, "rewards/code_execution_reward/std": 0.4957893490791321, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.021303100511431694, "step": 237, "step_time": 94.01764661632478 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 219.3046875, "completions/mean_terminated_length": 218.73190307617188, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.21575153153389692, "epoch": 0.1356125356125356, "frac_reward_zero_std": 0.046875, "grad_norm": 0.038149479776620865, "kl": 0.02007386011246126, "learning_rate": 4.981610411306782e-06, "loss": 0.00010031455894932151, "num_tokens": 50728266.0, "reward": 2.3709959983825684, "reward_std": 0.5716847777366638, "rewards/code_complexity_reward/mean": 0.8122069835662842, "rewards/code_complexity_reward/std": 0.14531971514225006, "rewards/code_execution_reward/mean": 0.466796875, "rewards/code_execution_reward/std": 0.4993842542171478, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02465205453336239, "step": 238, "step_time": 62.378716139122844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 217.63671875, "completions/mean_terminated_length": 216.48236083984375, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.2201450273860246, "epoch": 0.1361823361823362, "frac_reward_zero_std": 0.03125, "grad_norm": 0.0390055850148201, "kl": 0.020925016069668345, "learning_rate": 4.981003302863795e-06, "loss": 0.00010465719969943166, "num_tokens": 50911136.0, "reward": 2.334423780441284, "reward_std": 0.5495731234550476, "rewards/code_complexity_reward/mean": 0.827441394329071, "rewards/code_complexity_reward/std": 0.1241956353187561, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 239, "step_time": 58.234191049821675 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 201.123046875, "completions/mean_terminated_length": 200.51467895507812, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2244277549907565, "epoch": 0.13675213675213677, "frac_reward_zero_std": 0.03125, "grad_norm": 0.052069343626499176, "kl": 0.02479355161631247, "learning_rate": 4.98038637326311e-06, "loss": 0.00012409884948283434, "num_tokens": 51082463.0, "reward": 2.317578077316284, "reward_std": 0.5360155701637268, "rewards/code_complexity_reward/mean": 0.8369140625, "rewards/code_complexity_reward/std": 0.11843173205852509, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.022032126784324646, "step": 240, "step_time": 76.53657666314393 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 228.83203125, "completions/mean_terminated_length": 228.2778778076172, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.22188056004233658, "epoch": 0.13732193732193732, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03801631182432175, "kl": 0.02077814361837227, "learning_rate": 4.9797596249468696e-06, "loss": 0.00010389648377895355, "num_tokens": 51267585.0, "reward": 2.2507810592651367, "reward_std": 0.5358907580375671, "rewards/code_complexity_reward/mean": 0.8145507574081421, "rewards/code_complexity_reward/std": 0.13787665963172913, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.025821086019277573, "step": 241, "step_time": 60.29297201521695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 220.064453125, "completions/mean_terminated_length": 217.7657470703125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22666385350748897, "epoch": 0.1378917378917379, "frac_reward_zero_std": 0.03125, "grad_norm": 0.04176280274987221, "kl": 0.022858659242046997, "learning_rate": 4.979123060396084e-06, "loss": 0.00011411159357521683, "num_tokens": 51450618.0, "reward": 2.249072313308716, "reward_std": 0.5449423789978027, "rewards/code_complexity_reward/mean": 0.8114258050918579, "rewards/code_complexity_reward/std": 0.13097302615642548, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 242, "step_time": 56.930800720117986 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 510.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 203.8359375, "completions/mean_terminated_length": 203.8359375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22885773633606732, "epoch": 0.13846153846153847, "frac_reward_zero_std": 0.0625, "grad_norm": 0.040655914694070816, "kl": 0.026743376583908685, "learning_rate": 4.978476682130621e-06, "loss": 0.00013383616169448942, "num_tokens": 51622718.0, "reward": 2.330810546875, "reward_std": 0.529329776763916, "rewards/code_complexity_reward/mean": 0.8428710699081421, "rewards/code_complexity_reward/std": 0.11210038512945175, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 243, "step_time": 67.93896865099669 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 507.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 190.654296875, "completions/mean_terminated_length": 190.654296875, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.21688415575772524, "epoch": 0.13903133903133902, "frac_reward_zero_std": 0.078125, "grad_norm": 0.06436245143413544, "kl": 0.048477117699803784, "learning_rate": 4.9778204927091955e-06, "loss": 0.000242740468820557, "num_tokens": 51785445.0, "reward": 2.3739256858825684, "reward_std": 0.5357487797737122, "rewards/code_complexity_reward/mean": 0.845703125, "rewards/code_complexity_reward/std": 0.1132645457983017, "rewards/code_execution_reward/mean": 0.43359375, "rewards/code_execution_reward/std": 0.4960552453994751, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 244, "step_time": 56.118102193810046 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 222.513671875, "completions/mean_terminated_length": 220.2342529296875, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.22408338566310704, "epoch": 0.1396011396011396, "frac_reward_zero_std": 0.0, "grad_norm": 0.03642168268561363, "kl": 0.026879356664721854, "learning_rate": 4.977154494729363e-06, "loss": 0.00013451492122840136, "num_tokens": 51971292.0, "reward": 2.2296876907348633, "reward_std": 0.5450206398963928, "rewards/code_complexity_reward/mean": 0.805859386920929, "rewards/code_complexity_reward/std": 0.1533859223127365, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.015517610125243664, "step": 245, "step_time": 57.67960701417178 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 218.359375, "completions/mean_terminated_length": 214.2891082763672, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22031981102190912, "epoch": 0.14017094017094017, "frac_reward_zero_std": 0.03125, "grad_norm": 0.039769336581230164, "kl": 0.04950402403483167, "learning_rate": 4.976478690827504e-06, "loss": 0.0002478991518728435, "num_tokens": 52150956.0, "reward": 2.3590331077575684, "reward_std": 0.5948245525360107, "rewards/code_complexity_reward/mean": 0.8092772960662842, "rewards/code_complexity_reward/std": 0.16180507838726044, "rewards/code_execution_reward/mean": 0.462890625, "rewards/code_execution_reward/std": 0.4991086423397064, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.023893006145954132, "step": 246, "step_time": 58.514386370778084 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 201.111328125, "completions/mean_terminated_length": 200.5029296875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.23070854414254427, "epoch": 0.14074074074074075, "frac_reward_zero_std": 0.046875, "grad_norm": 0.03996627405285835, "kl": 0.026581618934869766, "learning_rate": 4.975793083678818e-06, "loss": 0.00013272935757413507, "num_tokens": 52325301.0, "reward": 2.283447265625, "reward_std": 0.5356225967407227, "rewards/code_complexity_reward/mean": 0.8388671875, "rewards/code_complexity_reward/std": 0.12132295966148376, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.026264816522598267, "step": 247, "step_time": 58.00085554737598 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 210.7265625, "completions/mean_terminated_length": 209.54510498046875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.22541227657347918, "epoch": 0.1413105413105413, "frac_reward_zero_std": 0.09375, "grad_norm": 0.04154182970523834, "kl": 0.0261933818255784, "learning_rate": 4.97509767599731e-06, "loss": 0.00013114407192915678, "num_tokens": 52499049.0, "reward": 2.284912109375, "reward_std": 0.5485710501670837, "rewards/code_complexity_reward/mean": 0.826953113079071, "rewards/code_complexity_reward/std": 0.1397372931241989, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 248, "step_time": 67.07241114228964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 221.888671875, "completions/mean_terminated_length": 220.1787872314453, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22655118582770228, "epoch": 0.14188034188034188, "frac_reward_zero_std": 0.03125, "grad_norm": 0.04708736017346382, "kl": 0.04480671198689379, "learning_rate": 4.974392470535781e-06, "loss": 0.00022455811267718673, "num_tokens": 52684464.0, "reward": 2.255859375, "reward_std": 0.5413592457771301, "rewards/code_complexity_reward/mean": 0.8130859136581421, "rewards/code_complexity_reward/std": 0.1425057053565979, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.030144967138767242, "step": 249, "step_time": 66.9971422124654 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 213.53515625, "completions/mean_terminated_length": 212.95108032226562, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.22117138211615384, "epoch": 0.14245014245014245, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04109414294362068, "kl": 0.025543141033267602, "learning_rate": 4.973677470085816e-06, "loss": 0.00012777617666870356, "num_tokens": 52861410.0, "reward": 2.3080077171325684, "reward_std": 0.5392353534698486, "rewards/code_complexity_reward/mean": 0.818554699420929, "rewards/code_complexity_reward/std": 0.1269754320383072, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.028043000027537346, "step": 250, "step_time": 65.05382530298084 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 214.78515625, "completions/mean_terminated_length": 213.61961364746094, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.23387988633476198, "epoch": 0.14301994301994303, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04893677309155464, "kl": 0.025837604072876275, "learning_rate": 4.9729526774777755e-06, "loss": 0.00012912026431877166, "num_tokens": 53042820.0, "reward": 2.239941358566284, "reward_std": 0.5123894214630127, "rewards/code_complexity_reward/mean": 0.810351550579071, "rewards/code_complexity_reward/std": 0.12025850266218185, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 251, "step_time": 79.9410479767248 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 202.23046875, "completions/mean_terminated_length": 202.23046875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.22125875018537045, "epoch": 0.14358974358974358, "frac_reward_zero_std": 0.046875, "grad_norm": 0.04214031994342804, "kl": 0.029122982232365757, "learning_rate": 4.972218095580783e-06, "loss": 0.0001454517914680764, "num_tokens": 53213666.0, "reward": 2.345654249191284, "reward_std": 0.5492428541183472, "rewards/code_complexity_reward/mean": 0.834277331829071, "rewards/code_complexity_reward/std": 0.12248193472623825, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 252, "step_time": 62.54309350438416 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 481.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 203.9375, "completions/mean_terminated_length": 203.9375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2341863801702857, "epoch": 0.14415954415954416, "frac_reward_zero_std": 0.046875, "grad_norm": 0.0541110634803772, "kl": 0.07339693891117349, "learning_rate": 4.971473727302712e-06, "loss": 0.00036720119533129036, "num_tokens": 53384770.0, "reward": 2.2833008766174316, "reward_std": 0.5507351160049438, "rewards/code_complexity_reward/mean": 0.8294922113418579, "rewards/code_complexity_reward/std": 0.13999347388744354, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 253, "step_time": 74.10365837439895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 208.154296875, "completions/mean_terminated_length": 206.96275329589844, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.2190987546928227, "epoch": 0.14472934472934473, "frac_reward_zero_std": 0.03125, "grad_norm": 0.03821369260549545, "kl": 0.02798245451413095, "learning_rate": 4.970719575590174e-06, "loss": 0.00013984407996758819, "num_tokens": 53558201.0, "reward": 2.3061037063598633, "reward_std": 0.5412311553955078, "rewards/code_complexity_reward/mean": 0.8281249403953552, "rewards/code_complexity_reward/std": 0.11325982958078384, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.024042511358857155, "step": 254, "step_time": 66.53963920008391 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 207.81640625, "completions/mean_terminated_length": 205.4212646484375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.229809848126024, "epoch": 0.1452991452991453, "frac_reward_zero_std": 0.046875, "grad_norm": 0.04256806895136833, "kl": 0.026662305783247575, "learning_rate": 4.969955643428513e-06, "loss": 0.00013328979548532516, "num_tokens": 53737419.0, "reward": 2.279589891433716, "reward_std": 0.5631312727928162, "rewards/code_complexity_reward/mean": 0.818652331829071, "rewards/code_complexity_reward/std": 0.1416589617729187, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.024554960429668427, "step": 255, "step_time": 62.58200112544 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 201.205078125, "completions/mean_terminated_length": 199.9862823486328, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.22295247786678374, "epoch": 0.14586894586894586, "frac_reward_zero_std": 0.015625, "grad_norm": 0.047399114817380905, "kl": 0.028520565989310853, "learning_rate": 4.969181933841788e-06, "loss": 0.00014257809380069375, "num_tokens": 53908900.0, "reward": 2.234423875808716, "reward_std": 0.5241830945014954, "rewards/code_complexity_reward/mean": 0.8197265863418579, "rewards/code_complexity_reward/std": 0.13282833993434906, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.029656626284122467, "step": 256, "step_time": 64.89122360572219 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 203.13671875, "completions/mean_terminated_length": 201.92550659179688, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22395734209567308, "epoch": 0.14643874643874644, "frac_reward_zero_std": 0.03125, "grad_norm": 0.04485850781202316, "kl": 0.029390872456133366, "learning_rate": 4.968398449892759e-06, "loss": 0.0001471338327974081, "num_tokens": 54080850.0, "reward": 2.339160203933716, "reward_std": 0.5373133420944214, "rewards/code_complexity_reward/mean": 0.8316406011581421, "rewards/code_complexity_reward/std": 0.11529271304607391, "rewards/code_execution_reward/mean": 0.4140625, "rewards/code_execution_reward/std": 0.49304109811782837, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 257, "step_time": 50.71517966967076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 206.8671875, "completions/mean_terminated_length": 205.67059326171875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.236181674990803, "epoch": 0.147008547008547, "frac_reward_zero_std": 0.078125, "grad_norm": 0.041504427790641785, "kl": 0.02731829414551612, "learning_rate": 4.967605194682883e-06, "loss": 0.00013655559450853616, "num_tokens": 54259486.0, "reward": 2.23828125, "reward_std": 0.5111639499664307, "rewards/code_complexity_reward/mean": 0.8328125476837158, "rewards/code_complexity_reward/std": 0.12158243358135223, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 258, "step_time": 81.21202265098691 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 203.734375, "completions/mean_terminated_length": 202.52549743652344, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2315621201414615, "epoch": 0.1475783475783476, "frac_reward_zero_std": 0.046875, "grad_norm": 0.05680086836218834, "kl": 0.03063290503632743, "learning_rate": 4.966802171352292e-06, "loss": 0.00015333850751630962, "num_tokens": 54432998.0, "reward": 2.246875047683716, "reward_std": 0.5335789322853088, "rewards/code_complexity_reward/mean": 0.8274414539337158, "rewards/code_complexity_reward/std": 0.11933399736881256, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02697930857539177, "step": 259, "step_time": 58.85241849627346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 212.794921875, "completions/mean_terminated_length": 212.2093963623047, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2193978875875473, "epoch": 0.14814814814814814, "frac_reward_zero_std": 0.03125, "grad_norm": 0.04645537585020065, "kl": 0.0292634086945327, "learning_rate": 4.965989383079791e-06, "loss": 0.0001464096421841532, "num_tokens": 54611741.0, "reward": 2.239551067352295, "reward_std": 0.5090287923812866, "rewards/code_complexity_reward/mean": 0.8177734613418579, "rewards/code_complexity_reward/std": 0.1234949454665184, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 260, "step_time": 66.41367921419442 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 504.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 222.68359375, "completions/mean_terminated_length": 222.68359375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.22106097033247352, "epoch": 0.14871794871794872, "frac_reward_zero_std": 0.0625, "grad_norm": 0.03851547837257385, "kl": 0.02738363826938439, "learning_rate": 4.965166833082835e-06, "loss": 0.00013673497596755624, "num_tokens": 54794019.0, "reward": 2.162304639816284, "reward_std": 0.4949376583099365, "rewards/code_complexity_reward/mean": 0.8010742664337158, "rewards/code_complexity_reward/std": 0.1357872635126114, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 261, "step_time": 65.46212504338473 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 204.462890625, "completions/mean_terminated_length": 203.86105346679688, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22527120471931994, "epoch": 0.1492877492877493, "frac_reward_zero_std": 0.109375, "grad_norm": 0.04440036788582802, "kl": 0.032227126212092116, "learning_rate": 4.9643345246175255e-06, "loss": 0.00016111208242364228, "num_tokens": 54969144.0, "reward": 2.2696290016174316, "reward_std": 0.5277643799781799, "rewards/code_complexity_reward/mean": 0.8332030773162842, "rewards/code_complexity_reward/std": 0.12235896289348602, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 262, "step_time": 76.18814270943403 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 194.716796875, "completions/mean_terminated_length": 193.47256469726562, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2287544496357441, "epoch": 0.14985754985754987, "frac_reward_zero_std": 0.0625, "grad_norm": 0.04921532794833183, "kl": 0.03245594957843423, "learning_rate": 4.963492460978589e-06, "loss": 0.00016253258218057454, "num_tokens": 55136807.0, "reward": 2.3749513626098633, "reward_std": 0.5453716516494751, "rewards/code_complexity_reward/mean": 0.828906238079071, "rewards/code_complexity_reward/std": 0.12076231092214584, "rewards/code_execution_reward/mean": 0.451171875, "rewards/code_execution_reward/std": 0.498096764087677, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.01981583796441555, "step": 263, "step_time": 75.05802320782095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 188.298828125, "completions/mean_terminated_length": 187.6653594970703, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.22174898558296263, "epoch": 0.15042735042735042, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04558636620640755, "kl": 0.03371718979906291, "learning_rate": 4.962640645499372e-06, "loss": 0.00016871359548531473, "num_tokens": 55303912.0, "reward": 2.351123094558716, "reward_std": 0.5277244448661804, "rewards/code_complexity_reward/mean": 0.8516601324081421, "rewards/code_complexity_reward/std": 0.09823379665613174, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 264, "step_time": 61.28625417407602 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 440.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 200.7265625, "completions/mean_terminated_length": 200.7265625, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2282600982580334, "epoch": 0.150997150997151, "frac_reward_zero_std": 0.03125, "grad_norm": 0.041258469223976135, "kl": 0.0317201663274318, "learning_rate": 4.961779081551821e-06, "loss": 0.0001584445999469608, "num_tokens": 55476068.0, "reward": 2.2563962936401367, "reward_std": 0.5074824690818787, "rewards/code_complexity_reward/mean": 0.8282226324081421, "rewards/code_complexity_reward/std": 0.10930023342370987, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 265, "step_time": 43.47778430022299 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 209.59765625, "completions/mean_terminated_length": 207.81533813476562, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22193042538128793, "epoch": 0.15156695156695157, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04180821403861046, "kl": 0.03095546006807126, "learning_rate": 4.960907772546475e-06, "loss": 0.00015481706941500306, "num_tokens": 55653222.0, "reward": 2.1768555641174316, "reward_std": 0.48451709747314453, "rewards/code_complexity_reward/mean": 0.8209960460662842, "rewards/code_complexity_reward/std": 0.11883972585201263, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 266, "step_time": 76.25449394155294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 426.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 200.318359375, "completions/mean_terminated_length": 200.318359375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.22035250812768936, "epoch": 0.15213675213675212, "frac_reward_zero_std": 0.015625, "grad_norm": 0.039160359650850296, "kl": 0.03085860432474874, "learning_rate": 4.960026721932448e-06, "loss": 0.00015475385589525104, "num_tokens": 55823065.0, "reward": 2.28369140625, "reward_std": 0.5027056336402893, "rewards/code_complexity_reward/mean": 0.837207019329071, "rewards/code_complexity_reward/std": 0.093147873878479, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 267, "step_time": 58.422633840702474 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 196.029296875, "completions/mean_terminated_length": 194.7902069091797, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2320973586756736, "epoch": 0.1527065527065527, "frac_reward_zero_std": 0.0625, "grad_norm": 0.042360641062259674, "kl": 0.03359586957958527, "learning_rate": 4.959135933197417e-06, "loss": 0.00016811798559501767, "num_tokens": 55991368.0, "reward": 2.276611328125, "reward_std": 0.5262703895568848, "rewards/code_complexity_reward/mean": 0.8418945670127869, "rewards/code_complexity_reward/std": 0.1246432214975357, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 268, "step_time": 106.31858994998038 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 190.419921875, "completions/mean_terminated_length": 189.7906036376953, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.21778240473940969, "epoch": 0.15327635327635328, "frac_reward_zero_std": 0.09375, "grad_norm": 0.04298964887857437, "kl": 0.031498841068241745, "learning_rate": 4.958235409867604e-06, "loss": 0.00015705838450230658, "num_tokens": 56156775.0, "reward": 2.3622069358825684, "reward_std": 0.5539397597312927, "rewards/code_complexity_reward/mean": 0.8429687023162842, "rewards/code_complexity_reward/std": 0.12547121942043304, "rewards/code_execution_reward/mean": 0.42578125, "rewards/code_execution_reward/std": 0.4949444830417633, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 269, "step_time": 88.0942621352151 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 208.671875, "completions/mean_terminated_length": 206.28346252441406, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22606214927509427, "epoch": 0.15384615384615385, "frac_reward_zero_std": 0.03125, "grad_norm": 0.06657952815294266, "kl": 0.03492280573118478, "learning_rate": 4.957325155507774e-06, "loss": 0.00017450316227041185, "num_tokens": 56332135.0, "reward": 2.2762207984924316, "reward_std": 0.5353042483329773, "rewards/code_complexity_reward/mean": 0.825390636920929, "rewards/code_complexity_reward/std": 0.12390358000993729, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.023952921852469444, "step": 270, "step_time": 59.75664753187448 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 477.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 202.1015625, "completions/mean_terminated_length": 202.1015625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22357050608843565, "epoch": 0.1544159544159544, "frac_reward_zero_std": 0.109375, "grad_norm": 0.042017001658678055, "kl": 0.03257827440393157, "learning_rate": 4.956405173721203e-06, "loss": 0.00016292263171635568, "num_tokens": 56504283.0, "reward": 2.2747559547424316, "reward_std": 0.5036084651947021, "rewards/code_complexity_reward/mean": 0.837597668170929, "rewards/code_complexity_reward/std": 0.1006217896938324, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 271, "step_time": 52.43077183701098 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 200.630859375, "completions/mean_terminated_length": 198.79568481445312, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22590368846431375, "epoch": 0.15498575498575498, "frac_reward_zero_std": 0.046875, "grad_norm": 0.04031599685549736, "kl": 0.03822638790006749, "learning_rate": 4.955475468149683e-06, "loss": 0.0001913538872031495, "num_tokens": 56676726.0, "reward": 2.2227540016174316, "reward_std": 0.5225070118904114, "rewards/code_complexity_reward/mean": 0.8273437023162842, "rewards/code_complexity_reward/std": 0.129632830619812, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 272, "step_time": 105.98312493879348 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 457.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 196.001953125, "completions/mean_terminated_length": 196.001953125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22941024089232087, "epoch": 0.15555555555555556, "frac_reward_zero_std": 0.0625, "grad_norm": 0.046180929988622665, "kl": 0.037546047431533225, "learning_rate": 4.9545360424734896e-06, "loss": 0.00018755260680336505, "num_tokens": 56845103.0, "reward": 2.2942872047424316, "reward_std": 0.5154110193252563, "rewards/code_complexity_reward/mean": 0.8402343392372131, "rewards/code_complexity_reward/std": 0.11604347825050354, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 273, "step_time": 61.9121949961409 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 497.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 189.873046875, "completions/mean_terminated_length": 189.873046875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.22742412239313126, "epoch": 0.15612535612535614, "frac_reward_zero_std": 0.03125, "grad_norm": 0.04915747791528702, "kl": 0.034991172680747695, "learning_rate": 4.953586900411381e-06, "loss": 0.00017496122745797038, "num_tokens": 57009806.0, "reward": 2.3441405296325684, "reward_std": 0.5222680568695068, "rewards/code_complexity_reward/mean": 0.84619140625, "rewards/code_complexity_reward/std": 0.0959051325917244, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.019099153578281403, "step": 274, "step_time": 65.02024428546429 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 480.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 189.27734375, "completions/mean_terminated_length": 189.27734375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.21579604013822973, "epoch": 0.15669515669515668, "frac_reward_zero_std": 0.09375, "grad_norm": 0.040354400873184204, "kl": 0.03935672639636323, "learning_rate": 4.952628045720577e-06, "loss": 0.00019662617705762386, "num_tokens": 57177148.0, "reward": 2.3875975608825684, "reward_std": 0.5420096516609192, "rewards/code_complexity_reward/mean": 0.83984375, "rewards/code_complexity_reward/std": 0.12684787809848785, "rewards/code_execution_reward/mean": 0.451171875, "rewards/code_execution_reward/std": 0.498096764087677, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 275, "step_time": 136.04866564180702 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 489.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 198.8125, "completions/mean_terminated_length": 198.8125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2219151477329433, "epoch": 0.15726495726495726, "frac_reward_zero_std": 0.0625, "grad_norm": 0.04655470326542854, "kl": 0.03588120869244449, "learning_rate": 4.9516594821967435e-06, "loss": 0.00017965008737519383, "num_tokens": 57349828.0, "reward": 2.3036131858825684, "reward_std": 0.5528804063796997, "rewards/code_complexity_reward/mean": 0.82666015625, "rewards/code_complexity_reward/std": 0.14227387309074402, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 276, "step_time": 47.710278392769396 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 422.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 197.158203125, "completions/mean_terminated_length": 197.158203125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.22839635610580444, "epoch": 0.15783475783475784, "frac_reward_zero_std": 0.046875, "grad_norm": 0.042586278170347214, "kl": 0.04282307904213667, "learning_rate": 4.950681213673983e-06, "loss": 0.00021413987269625068, "num_tokens": 57522789.0, "reward": 2.2201173305511475, "reward_std": 0.5008559823036194, "rewards/code_complexity_reward/mean": 0.822265625, "rewards/code_complexity_reward/std": 0.11705093830823898, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 277, "step_time": 64.51526249293238 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 197.8125, "completions/mean_terminated_length": 197.19764709472656, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22991410014219582, "epoch": 0.15840455840455842, "frac_reward_zero_std": 0.0625, "grad_norm": 0.041882239282131195, "kl": 0.03837390206172131, "learning_rate": 4.94969324402481e-06, "loss": 0.00019191244791727513, "num_tokens": 57693381.0, "reward": 2.293017625808716, "reward_std": 0.5291098952293396, "rewards/code_complexity_reward/mean": 0.8358398675918579, "rewards/code_complexity_reward/std": 0.11338794976472855, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 278, "step_time": 67.70926204975694 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 193.1953125, "completions/mean_terminated_length": 192.57142639160156, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2221901419106871, "epoch": 0.15897435897435896, "frac_reward_zero_std": 0.046875, "grad_norm": 0.04795067384839058, "kl": 0.03975115469074808, "learning_rate": 4.948695577160148e-06, "loss": 0.00019888789393007755, "num_tokens": 57859793.0, "reward": 2.2978515625, "reward_std": 0.5653634071350098, "rewards/code_complexity_reward/mean": 0.828906238079071, "rewards/code_complexity_reward/std": 0.14060387015342712, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 279, "step_time": 56.893591868691146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 188.787109375, "completions/mean_terminated_length": 188.15460205078125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23917598649859428, "epoch": 0.15954415954415954, "frac_reward_zero_std": 0.015625, "grad_norm": 0.04743498936295509, "kl": 0.044819605827797204, "learning_rate": 4.9476882170293006e-06, "loss": 0.00022430537501350045, "num_tokens": 58022772.0, "reward": 2.2733888626098633, "reward_std": 0.5554621815681458, "rewards/code_complexity_reward/mean": 0.820117175579071, "rewards/code_complexity_reward/std": 0.15450060367584229, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 280, "step_time": 65.504090372473 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 483.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 196.787109375, "completions/mean_terminated_length": 196.787109375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22539272555150092, "epoch": 0.16011396011396012, "frac_reward_zero_std": 0.09375, "grad_norm": 0.04116227105259895, "kl": 0.039170692703919485, "learning_rate": 4.946671167619948e-06, "loss": 0.0001958406064659357, "num_tokens": 58190295.0, "reward": 2.2845702171325684, "reward_std": 0.5176400542259216, "rewards/code_complexity_reward/mean": 0.839062511920929, "rewards/code_complexity_reward/std": 0.10689304769039154, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 281, "step_time": 55.101550688035786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 187.583984375, "completions/mean_terminated_length": 186.94911193847656, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22940012183971703, "epoch": 0.1606837606837607, "frac_reward_zero_std": 0.078125, "grad_norm": 0.04403425753116608, "kl": 0.042964210209902376, "learning_rate": 4.9456444329581255e-06, "loss": 0.00021472906519193202, "num_tokens": 58357066.0, "reward": 2.27294921875, "reward_std": 0.49777457118034363, "rewards/code_complexity_reward/mean": 0.8411133289337158, "rewards/code_complexity_reward/std": 0.10305570811033249, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.026930565014481544, "step": 282, "step_time": 57.43356974981725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 204.44140625, "completions/mean_terminated_length": 203.8395233154297, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.22838484891690314, "epoch": 0.16125356125356125, "frac_reward_zero_std": 0.0625, "grad_norm": 0.044702786952257156, "kl": 0.035792033100733534, "learning_rate": 4.944608017108203e-06, "loss": 0.00017916558135766536, "num_tokens": 58533796.0, "reward": 2.189453125, "reward_std": 0.5046541094779968, "rewards/code_complexity_reward/mean": 0.8326171636581421, "rewards/code_complexity_reward/std": 0.13074596226215363, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 283, "step_time": 69.3836506800726 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 201.724609375, "completions/mean_terminated_length": 201.11741638183594, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2338359453715384, "epoch": 0.16182336182336182, "frac_reward_zero_std": 0.09375, "grad_norm": 0.05406853184103966, "kl": 0.036032194358995184, "learning_rate": 4.9435619241728785e-06, "loss": 0.00018014395027421415, "num_tokens": 58704927.0, "reward": 2.1964354515075684, "reward_std": 0.4668179750442505, "rewards/code_complexity_reward/mean": 0.8359375596046448, "rewards/code_complexity_reward/std": 0.10866338014602661, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 284, "step_time": 66.8891871292144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 470.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 186.71875, "completions/mean_terminated_length": 186.71875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22838804661296308, "epoch": 0.1623931623931624, "frac_reward_zero_std": 0.046875, "grad_norm": 0.057478804141283035, "kl": 0.042129234614549205, "learning_rate": 4.942506158293155e-06, "loss": 0.0002106749452650547, "num_tokens": 58870199.0, "reward": 2.3113279342651367, "reward_std": 0.5223372578620911, "rewards/code_complexity_reward/mean": 0.8470702767372131, "rewards/code_complexity_reward/std": 0.11330824345350266, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 285, "step_time": 74.90042733959854 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 189.359375, "completions/mean_terminated_length": 188.7279815673828, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2179607474245131, "epoch": 0.16296296296296298, "frac_reward_zero_std": 0.09375, "grad_norm": 0.048611268401145935, "kl": 0.03953049183473922, "learning_rate": 4.941440723648328e-06, "loss": 0.00019772982341237366, "num_tokens": 59034759.0, "reward": 2.254199266433716, "reward_std": 0.5176951289176941, "rewards/code_complexity_reward/mean": 0.8365234136581421, "rewards/code_complexity_reward/std": 0.11789087951183319, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 286, "step_time": 67.21419882029295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 192.5625, "completions/mean_terminated_length": 191.9373779296875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.225553517928347, "epoch": 0.16353276353276353, "frac_reward_zero_std": 0.03125, "grad_norm": 0.048764538019895554, "kl": 0.04085266450420022, "learning_rate": 4.940365624455964e-06, "loss": 0.00020437099738046527, "num_tokens": 59203903.0, "reward": 2.3472657203674316, "reward_std": 0.5218558311462402, "rewards/code_complexity_reward/mean": 0.845898449420929, "rewards/code_complexity_reward/std": 0.0994248241186142, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 287, "step_time": 61.18749143742025 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 187.8828125, "completions/mean_terminated_length": 186.6117706298828, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23389996378682554, "epoch": 0.1641025641025641, "frac_reward_zero_std": 0.078125, "grad_norm": 0.041061174124479294, "kl": 0.04068573712720536, "learning_rate": 4.939280864971891e-06, "loss": 0.00020360833150334656, "num_tokens": 59369491.0, "reward": 2.2885255813598633, "reward_std": 0.5407186150550842, "rewards/code_complexity_reward/mean": 0.845507800579071, "rewards/code_complexity_reward/std": 0.13677632808685303, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 288, "step_time": 65.98832699283957 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 192.626953125, "completions/mean_terminated_length": 191.37452697753906, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.22754805674776435, "epoch": 0.16467236467236468, "frac_reward_zero_std": 0.078125, "grad_norm": 0.05222293734550476, "kl": 0.038831935002235696, "learning_rate": 4.938186449490175e-06, "loss": 0.0001942484814208001, "num_tokens": 59536876.0, "reward": 2.3334474563598633, "reward_std": 0.5567880272865295, "rewards/code_complexity_reward/mean": 0.8372070789337158, "rewards/code_complexity_reward/std": 0.13565601408481598, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 289, "step_time": 48.75495899375528 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 467.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 178.8984375, "completions/mean_terminated_length": 178.8984375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.21990815154276788, "epoch": 0.16524216524216523, "frac_reward_zero_std": 0.09375, "grad_norm": 0.042501408606767654, "kl": 0.041097964538494125, "learning_rate": 4.937082382343108e-06, "loss": 0.00020539753313641995, "num_tokens": 59697616.0, "reward": 2.351318359375, "reward_std": 0.5311822295188904, "rewards/code_complexity_reward/mean": 0.8550781011581421, "rewards/code_complexity_reward/std": 0.11709795147180557, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 290, "step_time": 47.17442773189396 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 197.255859375, "completions/mean_terminated_length": 195.40078735351562, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2198158719111234, "epoch": 0.1658119658119658, "frac_reward_zero_std": 0.140625, "grad_norm": 0.038606807589530945, "kl": 0.03915498359128833, "learning_rate": 4.935968667901185e-06, "loss": 0.00019571598386391997, "num_tokens": 59870051.0, "reward": 2.367138624191284, "reward_std": 0.5572884678840637, "rewards/code_complexity_reward/mean": 0.831347644329071, "rewards/code_complexity_reward/std": 0.13163408637046814, "rewards/code_execution_reward/mean": 0.439453125, "rewards/code_execution_reward/std": 0.49680593609809875, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 291, "step_time": 57.65884728729725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 486.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 193.357421875, "completions/mean_terminated_length": 193.357421875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22019778098911047, "epoch": 0.16638176638176638, "frac_reward_zero_std": 0.03125, "grad_norm": 0.04453340917825699, "kl": 0.03816460934467614, "learning_rate": 4.934845310573093e-06, "loss": 0.00019081411301158369, "num_tokens": 60037426.0, "reward": 2.2659668922424316, "reward_std": 0.5070059299468994, "rewards/code_complexity_reward/mean": 0.8332030773162842, "rewards/code_complexity_reward/std": 0.11244086176156998, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.019900046288967133, "step": 292, "step_time": 96.42388203274459 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 192.537109375, "completions/mean_terminated_length": 190.0216522216797, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2201454455498606, "epoch": 0.16695156695156696, "frac_reward_zero_std": 0.109375, "grad_norm": 0.07095185667276382, "kl": 0.03656018209585454, "learning_rate": 4.93371231480569e-06, "loss": 0.00018274685135111213, "num_tokens": 60207773.0, "reward": 2.2657227516174316, "reward_std": 0.5341900587081909, "rewards/code_complexity_reward/mean": 0.835156261920929, "rewards/code_complexity_reward/std": 0.14515790343284607, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 293, "step_time": 68.9343585781753 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 183.337890625, "completions/mean_terminated_length": 182.0490264892578, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.22677868115715683, "epoch": 0.1675213675213675, "frac_reward_zero_std": 0.015625, "grad_norm": 0.051760897040367126, "kl": 0.04440317818080075, "learning_rate": 4.9325696850839884e-06, "loss": 0.0002221267786808312, "num_tokens": 60370394.0, "reward": 2.315722703933716, "reward_std": 0.5304263830184937, "rewards/code_complexity_reward/mean": 0.858691394329071, "rewards/code_complexity_reward/std": 0.1174069344997406, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 294, "step_time": 60.23211861215532 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 503.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 194.25, "completions/mean_terminated_length": 194.25, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.22946100076660514, "epoch": 0.16809116809116809, "frac_reward_zero_std": 0.015625, "grad_norm": 0.046525903046131134, "kl": 0.039882982906419784, "learning_rate": 4.931417425931137e-06, "loss": 0.00019956647884100676, "num_tokens": 60536314.0, "reward": 2.288818359375, "reward_std": 0.5270154476165771, "rewards/code_complexity_reward/mean": 0.827441394329071, "rewards/code_complexity_reward/std": 0.12384059280157089, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.024042511358857155, "step": 295, "step_time": 65.0567777287215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 413.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 188.302734375, "completions/mean_terminated_length": 188.302734375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22563186334446073, "epoch": 0.16866096866096866, "frac_reward_zero_std": 0.09375, "grad_norm": 0.05474657565355301, "kl": 0.041816925193415955, "learning_rate": 4.9302555419084035e-06, "loss": 0.00020894659974146634, "num_tokens": 60700933.0, "reward": 2.3931641578674316, "reward_std": 0.5393286347389221, "rewards/code_complexity_reward/mean": 0.8421874642372131, "rewards/code_complexity_reward/std": 0.10130024701356888, "rewards/code_execution_reward/mean": 0.453125, "rewards/code_execution_reward/std": 0.4982847273349762, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 296, "step_time": 43.740755155682564 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 451.0, "completions/mean_length": 179.265625, "completions/mean_terminated_length": 178.61448669433594, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2225768577773124, "epoch": 0.16923076923076924, "frac_reward_zero_std": 0.0625, "grad_norm": 0.05217895284295082, "kl": 0.0723863486200571, "learning_rate": 4.929084037615154e-06, "loss": 0.00036220031324774027, "num_tokens": 60858165.0, "reward": 2.3187501430511475, "reward_std": 0.5313761234283447, "rewards/code_complexity_reward/mean": 0.857128918170929, "rewards/code_complexity_reward/std": 0.11366065591573715, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 297, "step_time": 58.983289455994964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 192.822265625, "completions/mean_terminated_length": 192.19764709472656, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.21882506320253015, "epoch": 0.1698005698005698, "frac_reward_zero_std": 0.078125, "grad_norm": 0.04564080759882927, "kl": 0.043432618578663096, "learning_rate": 4.927902917688839e-06, "loss": 0.00021709699649363756, "num_tokens": 61025754.0, "reward": 2.2374024391174316, "reward_std": 0.49202045798301697, "rewards/code_complexity_reward/mean": 0.8375976085662842, "rewards/code_complexity_reward/std": 0.12511242926120758, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 298, "step_time": 64.04298048932105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 181.275390625, "completions/mean_terminated_length": 179.32614135742188, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22675036755390465, "epoch": 0.17037037037037037, "frac_reward_zero_std": 0.109375, "grad_norm": 0.04922235384583473, "kl": 0.04406607695273124, "learning_rate": 4.9267121868049716e-06, "loss": 0.00022060202900320292, "num_tokens": 61187895.0, "reward": 2.2764649391174316, "reward_std": 0.5314362645149231, "rewards/code_complexity_reward/mean": 0.8431640863418579, "rewards/code_complexity_reward/std": 0.1322770118713379, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 299, "step_time": 67.59007748868316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 510.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 186.423828125, "completions/mean_terminated_length": 186.423828125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23030899139121175, "epoch": 0.17094017094017094, "frac_reward_zero_std": 0.109375, "grad_norm": 0.06968192011117935, "kl": 0.046035854902584106, "learning_rate": 4.925511849677113e-06, "loss": 0.0002303261135239154, "num_tokens": 61350472.0, "reward": 2.3026857376098633, "reward_std": 0.520047128200531, "rewards/code_complexity_reward/mean": 0.842089831829071, "rewards/code_complexity_reward/std": 0.11125914752483368, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.019864002242684364, "step": 300, "step_time": 55.587306876666844 }, { "epoch": 0.17094017094017094, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0025, "eval_completions/max_length": 254.75, "eval_completions/max_terminated_length": 252.07, "eval_completions/mean_length": 184.99125, "eval_completions/mean_terminated_length": 184.5125003051758, "eval_completions/min_length": 128.85, "eval_completions/min_terminated_length": 128.85, "eval_entropy": 0.2269045141339302, "eval_frac_reward_zero_std": 0.06, "eval_kl": 0.0460474915523082, "eval_loss": -0.0018519493751227856, "eval_num_tokens": 61350472.0, "eval_reward": 2.2387187802791595, "eval_reward_std": 0.29302072973921894, "eval_rewards/code_complexity_reward/mean": 0.8316874974966049, "eval_rewards/code_complexity_reward/std": 0.0733561915718019, "eval_rewards/code_execution_reward/mean": 0.31625, "eval_rewards/code_execution_reward/std": 0.2200625157356262, "eval_rewards/code_syntax_reward/mean": 0.49125, "eval_rewards/code_syntax_reward/std": 0.020812198668718338, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.49953125, "eval_rewards/xmlcount_reward_func/std": 0.0013258251920342445, "eval_runtime": 1205.712, "eval_samples_per_second": 0.083, "eval_steps_per_second": 0.011, "step": 300 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 194.87890625, "completions/mean_terminated_length": 193.00982666015625, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.21903980476781726, "epoch": 0.17150997150997152, "frac_reward_zero_std": 0.03125, "grad_norm": 0.06481245160102844, "kl": 0.043856118369149044, "learning_rate": 4.924301911056847e-06, "loss": 0.00021928080241195858, "num_tokens": 61519730.0, "reward": 2.2978515625, "reward_std": 0.55023592710495, "rewards/code_complexity_reward/mean": 0.8311523795127869, "rewards/code_complexity_reward/std": 0.14045368134975433, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 301, "step_time": 67.04367528762668 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 186.099609375, "completions/mean_terminated_length": 183.53346252441406, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.23120431043207645, "epoch": 0.17207977207977207, "frac_reward_zero_std": 0.078125, "grad_norm": 0.05572855472564697, "kl": 0.04312630320782773, "learning_rate": 4.923082375733767e-06, "loss": 0.00021565816132351756, "num_tokens": 61685013.0, "reward": 2.1865234375, "reward_std": 0.5104739665985107, "rewards/code_complexity_reward/mean": 0.8321288824081421, "rewards/code_complexity_reward/std": 0.144832581281662, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 302, "step_time": 109.12754524219781 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 177.626953125, "completions/mean_terminated_length": 176.97259521484375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.22373731224797666, "epoch": 0.17264957264957265, "frac_reward_zero_std": 0.125, "grad_norm": 0.04587939381599426, "kl": 0.046310561563586816, "learning_rate": 4.921853248535459e-06, "loss": 0.00023154873633757234, "num_tokens": 61843350.0, "reward": 2.2705078125, "reward_std": 0.5199275612831116, "rewards/code_complexity_reward/mean": 0.8614257574081421, "rewards/code_complexity_reward/std": 0.13721196353435516, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 303, "step_time": 55.42139623593539 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 181.041015625, "completions/mean_terminated_length": 180.39334106445312, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.2273602255154401, "epoch": 0.17321937321937322, "frac_reward_zero_std": 0.078125, "grad_norm": 0.05218443274497986, "kl": 0.04840376149513759, "learning_rate": 4.920614534327472e-06, "loss": 0.00024194683646783233, "num_tokens": 62003611.0, "reward": 2.286914110183716, "reward_std": 0.5394623875617981, "rewards/code_complexity_reward/mean": 0.846484363079071, "rewards/code_complexity_reward/std": 0.13748276233673096, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 304, "step_time": 67.06744588818401 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 167.46484375, "completions/mean_terminated_length": 166.7906036376953, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22375843208283186, "epoch": 0.1737891737891738, "frac_reward_zero_std": 0.109375, "grad_norm": 0.056332603096961975, "kl": 0.0569593520485796, "learning_rate": 4.9193662380133115e-06, "loss": 0.00028496465529315174, "num_tokens": 62155257.0, "reward": 2.3556153774261475, "reward_std": 0.5253123044967651, "rewards/code_complexity_reward/mean": 0.8505859375, "rewards/code_complexity_reward/std": 0.11446142196655273, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 305, "step_time": 56.192959459498525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 180.193359375, "completions/mean_terminated_length": 179.54403686523438, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22892334684729576, "epoch": 0.17435897435897435, "frac_reward_zero_std": 0.03125, "grad_norm": 0.05688875913619995, "kl": 0.06179052434163168, "learning_rate": 4.91810836453441e-06, "loss": 0.0003092230181209743, "num_tokens": 62315884.0, "reward": 2.2828125953674316, "reward_std": 0.5275089144706726, "rewards/code_complexity_reward/mean": 0.8441406488418579, "rewards/code_complexity_reward/std": 0.1266566962003708, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.019055325537919998, "step": 306, "step_time": 128.20818171091378 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 170.580078125, "completions/mean_terminated_length": 170.580078125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21836352674290538, "epoch": 0.17492877492877493, "frac_reward_zero_std": 0.109375, "grad_norm": 0.0477229580283165, "kl": 0.05170431238366291, "learning_rate": 4.916840918870115e-06, "loss": 0.00025876174913719296, "num_tokens": 62472157.0, "reward": 2.3221192359924316, "reward_std": 0.5383895635604858, "rewards/code_complexity_reward/mean": 0.842968761920929, "rewards/code_complexity_reward/std": 0.12484578788280487, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 307, "step_time": 41.52801481448114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 180.81640625, "completions/mean_terminated_length": 180.1682891845703, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.22361960704438388, "epoch": 0.1754985754985755, "frac_reward_zero_std": 0.09375, "grad_norm": 0.0501810647547245, "kl": 0.050867858255514875, "learning_rate": 4.9155639060376635e-06, "loss": 0.0002544290036894381, "num_tokens": 62634263.0, "reward": 2.2837891578674316, "reward_std": 0.5214746594429016, "rewards/code_complexity_reward/mean": 0.8395507335662842, "rewards/code_complexity_reward/std": 0.1194540485739708, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 308, "step_time": 68.53437728714198 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 476.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 183.775390625, "completions/mean_terminated_length": 183.775390625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2241826660465449, "epoch": 0.17606837606837608, "frac_reward_zero_std": 0.125, "grad_norm": 0.042217519134283066, "kl": 0.05442167443106882, "learning_rate": 4.914277331092165e-06, "loss": 0.00027228723047301173, "num_tokens": 62797028.0, "reward": 2.2619142532348633, "reward_std": 0.4981125295162201, "rewards/code_complexity_reward/mean": 0.839648425579071, "rewards/code_complexity_reward/std": 0.11274147033691406, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 309, "step_time": 65.02018330339342 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 180.970703125, "completions/mean_terminated_length": 179.0196533203125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2282206234522164, "epoch": 0.17663817663817663, "frac_reward_zero_std": 0.046875, "grad_norm": 0.04758700728416443, "kl": 0.04667403289931826, "learning_rate": 4.9129811991265835e-06, "loss": 0.00023333357239607722, "num_tokens": 62959517.0, "reward": 2.2702150344848633, "reward_std": 0.5403023958206177, "rewards/code_complexity_reward/mean": 0.828417956829071, "rewards/code_complexity_reward/std": 0.13475365936756134, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 310, "step_time": 55.94664412178099 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 171.65234375, "completions/mean_terminated_length": 170.98629760742188, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23226869385689497, "epoch": 0.1772079772079772, "frac_reward_zero_std": 0.140625, "grad_norm": 0.04685474932193756, "kl": 0.05736338303540833, "learning_rate": 4.911675515271711e-06, "loss": 0.00028691801708191633, "num_tokens": 63114947.0, "reward": 2.2532715797424316, "reward_std": 0.5040786266326904, "rewards/code_complexity_reward/mean": 0.8566405773162842, "rewards/code_complexity_reward/std": 0.1158430427312851, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 311, "step_time": 54.74848656542599 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 480.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 180.232421875, "completions/mean_terminated_length": 180.232421875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23341849213466048, "epoch": 0.17777777777777778, "frac_reward_zero_std": 0.0625, "grad_norm": 0.05345488339662552, "kl": 0.05094422469846904, "learning_rate": 4.910360284696154e-06, "loss": 0.0002546000760048628, "num_tokens": 63277810.0, "reward": 2.191210985183716, "reward_std": 0.48209714889526367, "rewards/code_complexity_reward/mean": 0.8379882574081421, "rewards/code_complexity_reward/std": 0.11285849660634995, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.019099153578281403, "step": 312, "step_time": 57.01496776472777 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 445.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 163.1796875, "completions/mean_terminated_length": 163.1796875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21539380494505167, "epoch": 0.17834757834757833, "frac_reward_zero_std": 0.15625, "grad_norm": 0.04842689260840416, "kl": 0.060161614790558815, "learning_rate": 4.909035512606307e-06, "loss": 0.000300683022942394, "num_tokens": 63426966.0, "reward": 2.451171875, "reward_std": 0.5508631467819214, "rewards/code_complexity_reward/mean": 0.852832019329071, "rewards/code_complexity_reward/std": 0.12616108357906342, "rewards/code_execution_reward/mean": 0.50390625, "rewards/code_execution_reward/std": 0.5004737377166748, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 313, "step_time": 50.24381101410836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 425.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 170.2265625, "completions/mean_terminated_length": 170.2265625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22747485851868987, "epoch": 0.1789173789173789, "frac_reward_zero_std": 0.109375, "grad_norm": 0.052895765751600266, "kl": 0.05691657541319728, "learning_rate": 4.907701204246339e-06, "loss": 0.000284614012343809, "num_tokens": 63581162.0, "reward": 2.2431640625, "reward_std": 0.49232080578804016, "rewards/code_complexity_reward/mean": 0.8438476324081421, "rewards/code_complexity_reward/std": 0.11727599054574966, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 314, "step_time": 62.28182325512171 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 413.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 159.40234375, "completions/mean_terminated_length": 159.40234375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2348092265892774, "epoch": 0.1794871794871795, "frac_reward_zero_std": 0.09375, "grad_norm": 0.06697177141904831, "kl": 0.061680460697971284, "learning_rate": 4.906357364898167e-06, "loss": 0.0003086141077801585, "num_tokens": 63730816.0, "reward": 2.3392090797424316, "reward_std": 0.5137369632720947, "rewards/code_complexity_reward/mean": 0.870312511920929, "rewards/code_complexity_reward/std": 0.09460753947496414, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 315, "step_time": 49.4837089786306 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 179.869140625, "completions/mean_terminated_length": 179.21917724609375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22944269189611077, "epoch": 0.18005698005698006, "frac_reward_zero_std": 0.09375, "grad_norm": 0.056896794587373734, "kl": 0.053012956806924194, "learning_rate": 4.905003999881435e-06, "loss": 0.0002649601665325463, "num_tokens": 63893557.0, "reward": 2.182177782058716, "reward_std": 0.4776599705219269, "rewards/code_complexity_reward/mean": 0.8350585699081421, "rewards/code_complexity_reward/std": 0.13284628093242645, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 316, "step_time": 49.475193310528994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 510.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 169.71875, "completions/mean_terminated_length": 169.71875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22196836164221168, "epoch": 0.18062678062678061, "frac_reward_zero_std": 0.09375, "grad_norm": 0.053050071001052856, "kl": 0.060797358222771436, "learning_rate": 4.903641114553497e-06, "loss": 0.00030405662255361676, "num_tokens": 64046293.0, "reward": 2.3870606422424316, "reward_std": 0.558487057685852, "rewards/code_complexity_reward/mean": 0.844921886920929, "rewards/code_complexity_reward/std": 0.134701207280159, "rewards/code_execution_reward/mean": 0.44921875, "rewards/code_execution_reward/std": 0.497901052236557, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 317, "step_time": 57.490817406214774 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 173.74609375, "completions/mean_terminated_length": 173.0841522216797, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22765909205190837, "epoch": 0.1811965811965812, "frac_reward_zero_std": 0.15625, "grad_norm": 0.049626197665929794, "kl": 0.065591666061664, "learning_rate": 4.902268714309392e-06, "loss": 0.0003281787212472409, "num_tokens": 64204123.0, "reward": 2.3216309547424316, "reward_std": 0.53204745054245, "rewards/code_complexity_reward/mean": 0.846386730670929, "rewards/code_complexity_reward/std": 0.11117426306009293, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 318, "step_time": 77.64516551513225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 473.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 174.607421875, "completions/mean_terminated_length": 174.607421875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2354029077105224, "epoch": 0.18176638176638177, "frac_reward_zero_std": 0.09375, "grad_norm": 0.047871146351099014, "kl": 0.05638060421915725, "learning_rate": 4.900886804581827e-06, "loss": 0.00028179268701933324, "num_tokens": 64364026.0, "reward": 2.2717771530151367, "reward_std": 0.5058143734931946, "rewards/code_complexity_reward/mean": 0.8485351800918579, "rewards/code_complexity_reward/std": 0.10521063208580017, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 319, "step_time": 64.8181341746822 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 176.671875, "completions/mean_terminated_length": 176.01565551757812, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2197324950248003, "epoch": 0.18233618233618235, "frac_reward_zero_std": 0.109375, "grad_norm": 0.0480693057179451, "kl": 0.06211975938640535, "learning_rate": 4.8994953908411485e-06, "loss": 0.00031043536728248, "num_tokens": 64522178.0, "reward": 2.3512697219848633, "reward_std": 0.5179421901702881, "rewards/code_complexity_reward/mean": 0.8586913347244263, "rewards/code_complexity_reward/std": 0.0954320952296257, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 320, "step_time": 69.75335902348161 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 170.716796875, "completions/mean_terminated_length": 170.04891967773438, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23368879896588624, "epoch": 0.1829059829059829, "frac_reward_zero_std": 0.125, "grad_norm": 0.04826509207487106, "kl": 0.06126207503257319, "learning_rate": 4.898094478595329e-06, "loss": 0.0003064447664655745, "num_tokens": 64677481.0, "reward": 2.2728028297424316, "reward_std": 0.5083488821983337, "rewards/code_complexity_reward/mean": 0.854687511920929, "rewards/code_complexity_reward/std": 0.11398105323314667, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 321, "step_time": 75.08879533596337 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 162.603515625, "completions/mean_terminated_length": 161.91976928710938, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2335685295984149, "epoch": 0.18347578347578347, "frac_reward_zero_std": 0.046875, "grad_norm": 0.05858930945396423, "kl": 0.06230722420150414, "learning_rate": 4.896684073389939e-06, "loss": 0.00031153965392149985, "num_tokens": 64829358.0, "reward": 2.3456544876098633, "reward_std": 0.5625236630439758, "rewards/code_complexity_reward/mean": 0.8386719226837158, "rewards/code_complexity_reward/std": 0.13801993429660797, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 322, "step_time": 59.972717598080635 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 162.3046875, "completions/mean_terminated_length": 161.62034606933594, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22775287460535765, "epoch": 0.18404558404558405, "frac_reward_zero_std": 0.125, "grad_norm": 0.07779685407876968, "kl": 0.07005880540236831, "learning_rate": 4.895264180808128e-06, "loss": 0.0003502894251141697, "num_tokens": 64978730.0, "reward": 2.3068361282348633, "reward_std": 0.49920642375946045, "rewards/code_complexity_reward/mean": 0.8596680164337158, "rewards/code_complexity_reward/std": 0.09472013264894485, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 323, "step_time": 48.795407637022436 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 171.59765625, "completions/mean_terminated_length": 170.93150329589844, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23686456680297852, "epoch": 0.18461538461538463, "frac_reward_zero_std": 0.0625, "grad_norm": 0.05603732168674469, "kl": 0.06449330237228423, "learning_rate": 4.893834806470601e-06, "loss": 0.00032259352155961096, "num_tokens": 65135500.0, "reward": 2.2435545921325684, "reward_std": 0.5284007787704468, "rewards/code_complexity_reward/mean": 0.8373046517372131, "rewards/code_complexity_reward/std": 0.14753340184688568, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.025862684473395348, "step": 324, "step_time": 67.20016886480153 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 414.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 162.66796875, "completions/mean_terminated_length": 162.66796875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.21806276799179614, "epoch": 0.18518518518518517, "frac_reward_zero_std": 0.109375, "grad_norm": 0.05446573346853256, "kl": 0.06701166072161868, "learning_rate": 4.892395956035598e-06, "loss": 0.0003350671613588929, "num_tokens": 65288154.0, "reward": 2.2962403297424316, "reward_std": 0.5069012641906738, "rewards/code_complexity_reward/mean": 0.858105480670929, "rewards/code_complexity_reward/std": 0.08840688318014145, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 325, "step_time": 47.82805980369449 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 462.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 168.833984375, "completions/mean_terminated_length": 168.833984375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24036600743420422, "epoch": 0.18575498575498575, "frac_reward_zero_std": 0.09375, "grad_norm": 0.056976016610860825, "kl": 0.0752033342141658, "learning_rate": 4.890947635198871e-06, "loss": 0.00037614625762216747, "num_tokens": 65443357.0, "reward": 2.1610350608825684, "reward_std": 0.43345907330513, "rewards/code_complexity_reward/mean": 0.84716796875, "rewards/code_complexity_reward/std": 0.10287827998399734, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 326, "step_time": 64.5646508699283 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 475.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 159.111328125, "completions/mean_terminated_length": 159.111328125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2262644183356315, "epoch": 0.18632478632478633, "frac_reward_zero_std": 0.046875, "grad_norm": 0.07333357632160187, "kl": 0.07720596145372838, "learning_rate": 4.889489849693657e-06, "loss": 0.0003858787822537124, "num_tokens": 65591894.0, "reward": 2.30908203125, "reward_std": 0.5318068265914917, "rewards/code_complexity_reward/mean": 0.8589843511581421, "rewards/code_complexity_reward/std": 0.12808747589588165, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 327, "step_time": 72.9247330930084 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 467.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 163.501953125, "completions/mean_terminated_length": 163.501953125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2318716668523848, "epoch": 0.1868945868945869, "frac_reward_zero_std": 0.140625, "grad_norm": 0.05758746340870857, "kl": 0.07949286041548476, "learning_rate": 4.888022605290665e-06, "loss": 0.0003974221181124449, "num_tokens": 65744927.0, "reward": 2.2699220180511475, "reward_std": 0.5097159147262573, "rewards/code_complexity_reward/mean": 0.8495117425918579, "rewards/code_complexity_reward/std": 0.10683470219373703, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 328, "step_time": 64.4861447699368 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 158.33203125, "completions/mean_terminated_length": 158.33203125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23692346340976655, "epoch": 0.18746438746438746, "frac_reward_zero_std": 0.046875, "grad_norm": 0.062056150287389755, "kl": 0.0779337968560867, "learning_rate": 4.8865459077980434e-06, "loss": 0.0003901051532011479, "num_tokens": 65893529.0, "reward": 2.306933641433716, "reward_std": 0.5068641304969788, "rewards/code_complexity_reward/mean": 0.8534179925918579, "rewards/code_complexity_reward/std": 0.10025913268327713, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 329, "step_time": 41.462985397316515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 166.19921875, "completions/mean_terminated_length": 165.5225067138672, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23237967863678932, "epoch": 0.18803418803418803, "frac_reward_zero_std": 0.078125, "grad_norm": 0.05323782563209534, "kl": 0.08640384988393635, "learning_rate": 4.885059763061363e-06, "loss": 0.0004318933351896703, "num_tokens": 66050551.0, "reward": 2.327392578125, "reward_std": 0.5355538129806519, "rewards/code_complexity_reward/mean": 0.8418945074081421, "rewards/code_complexity_reward/std": 0.12686079740524292, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 330, "step_time": 55.275924714282155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 373.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 159.580078125, "completions/mean_terminated_length": 159.580078125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.21961741079576313, "epoch": 0.1886039886039886, "frac_reward_zero_std": 0.125, "grad_norm": 0.05810258537530899, "kl": 0.08891169872367755, "learning_rate": 4.88356417696359e-06, "loss": 0.000444636243628338, "num_tokens": 66199328.0, "reward": 2.3031249046325684, "reward_std": 0.5588901042938232, "rewards/code_complexity_reward/mean": 0.8388671875, "rewards/code_complexity_reward/std": 0.14263759553432465, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 331, "step_time": 48.794891310855746 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 466.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 148.51171875, "completions/mean_terminated_length": 148.51171875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22811482870019972, "epoch": 0.1891737891737892, "frac_reward_zero_std": 0.15625, "grad_norm": 0.061670929193496704, "kl": 0.09481391101144254, "learning_rate": 4.882059155425069e-06, "loss": 0.0004740238655358553, "num_tokens": 66342526.0, "reward": 2.455273389816284, "reward_std": 0.5413743257522583, "rewards/code_complexity_reward/mean": 0.8681640625, "rewards/code_complexity_reward/std": 0.10904382169246674, "rewards/code_execution_reward/mean": 0.4921875, "rewards/code_execution_reward/std": 0.5004279017448425, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.023378821089863777, "step": 332, "step_time": 71.60440878011286 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 502.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 153.1875, "completions/mean_terminated_length": 153.1875, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.2484284471720457, "epoch": 0.18974358974358974, "frac_reward_zero_std": 0.0625, "grad_norm": 0.06268219649791718, "kl": 0.08933454437647015, "learning_rate": 4.880544704403489e-06, "loss": 0.0004468684201128781, "num_tokens": 66490806.0, "reward": 2.2859864234924316, "reward_std": 0.5117935538291931, "rewards/code_complexity_reward/mean": 0.8639647960662842, "rewards/code_complexity_reward/std": 0.11293669790029526, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 333, "step_time": 58.06958099734038 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 463.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 155.767578125, "completions/mean_terminated_length": 155.767578125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2330259014852345, "epoch": 0.1903133903133903, "frac_reward_zero_std": 0.125, "grad_norm": 0.06000392511487007, "kl": 0.08579599263612181, "learning_rate": 4.87902082989387e-06, "loss": 0.00042899869731627405, "num_tokens": 66638319.0, "reward": 2.2828125953674316, "reward_std": 0.5108580589294434, "rewards/code_complexity_reward/mean": 0.851757824420929, "rewards/code_complexity_reward/std": 0.1150888204574585, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 334, "step_time": 52.078295200131834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 499.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 165.0390625, "completions/mean_terminated_length": 165.0390625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24049691087566316, "epoch": 0.1908831908831909, "frac_reward_zero_std": 0.109375, "grad_norm": 0.07016944885253906, "kl": 0.08759037707932293, "learning_rate": 4.877487537928536e-06, "loss": 0.0004380405880510807, "num_tokens": 66795515.0, "reward": 2.248974561691284, "reward_std": 0.5381667613983154, "rewards/code_complexity_reward/mean": 0.8386719226837158, "rewards/code_complexity_reward/std": 0.14932526648044586, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 335, "step_time": 48.36954453587532 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 158.044921875, "completions/mean_terminated_length": 156.65687561035156, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23018305259756744, "epoch": 0.19145299145299147, "frac_reward_zero_std": 0.171875, "grad_norm": 0.05215616896748543, "kl": 0.08913674193900079, "learning_rate": 4.875944834577086e-06, "loss": 0.000445568875875324, "num_tokens": 66942890.0, "reward": 2.319287061691284, "reward_std": 0.5503671169281006, "rewards/code_complexity_reward/mean": 0.857714831829071, "rewards/code_complexity_reward/std": 0.1396600902080536, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 336, "step_time": 49.108992100693285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 422.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 149.228515625, "completions/mean_terminated_length": 149.228515625, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.22310131834819913, "epoch": 0.19202279202279202, "frac_reward_zero_std": 0.109375, "grad_norm": 0.06984200328588486, "kl": 0.10069286427460611, "learning_rate": 4.87439272594638e-06, "loss": 0.0005035186186432838, "num_tokens": 67088415.0, "reward": 2.3546876907348633, "reward_std": 0.5072469115257263, "rewards/code_complexity_reward/mean": 0.8748047351837158, "rewards/code_complexity_reward/std": 0.08663055300712585, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.023378821089863777, "step": 337, "step_time": 43.417031462304294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 411.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 162.955078125, "completions/mean_terminated_length": 162.955078125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22410673275589943, "epoch": 0.1925925925925926, "frac_reward_zero_std": 0.171875, "grad_norm": 0.06408800184726715, "kl": 0.09187903790734708, "learning_rate": 4.872831218180504e-06, "loss": 0.0004594514612108469, "num_tokens": 67243848.0, "reward": 2.339160203933716, "reward_std": 0.539557695388794, "rewards/code_complexity_reward/mean": 0.8377929925918579, "rewards/code_complexity_reward/std": 0.12951214611530304, "rewards/code_execution_reward/mean": 0.40625, "rewards/code_execution_reward/std": 0.49161264300346375, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 338, "step_time": 71.64142351783812 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 161.185546875, "completions/mean_terminated_length": 161.185546875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2345005120150745, "epoch": 0.19316239316239317, "frac_reward_zero_std": 0.046875, "grad_norm": 0.05411020293831825, "kl": 0.08506119751837105, "learning_rate": 4.871260317460756e-06, "loss": 0.0004253623483236879, "num_tokens": 67395599.0, "reward": 2.248730421066284, "reward_std": 0.5096148252487183, "rewards/code_complexity_reward/mean": 0.841601550579071, "rewards/code_complexity_reward/std": 0.12558190524578094, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.028089813888072968, "step": 339, "step_time": 47.17930252943188 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 440.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 145.1796875, "completions/mean_terminated_length": 145.1796875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22852203552611172, "epoch": 0.19373219373219372, "frac_reward_zero_std": 0.125, "grad_norm": 0.058884572237730026, "kl": 0.09909282822627574, "learning_rate": 4.869680030005611e-06, "loss": 0.0004954496398568153, "num_tokens": 67535227.0, "reward": 2.3614258766174316, "reward_std": 0.5000697374343872, "rewards/code_complexity_reward/mean": 0.8600585460662842, "rewards/code_complexity_reward/std": 0.08736966550350189, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 340, "step_time": 51.39852469600737 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 139.064453125, "completions/mean_terminated_length": 139.064453125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23710773815400898, "epoch": 0.1943019943019943, "frac_reward_zero_std": 0.15625, "grad_norm": 0.07234074920415878, "kl": 0.12616520561277866, "learning_rate": 4.868090362070706e-06, "loss": 0.0006309489835985005, "num_tokens": 67671060.0, "reward": 2.4090821743011475, "reward_std": 0.5337627530097961, "rewards/code_complexity_reward/mean": 0.87060546875, "rewards/code_complexity_reward/std": 0.1090351864695549, "rewards/code_execution_reward/mean": 0.44140625, "rewards/code_execution_reward/std": 0.4970405399799347, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 341, "step_time": 46.01080149784684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 155.4375, "completions/mean_terminated_length": 154.73973083496094, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23840538901276886, "epoch": 0.19487179487179487, "frac_reward_zero_std": 0.109375, "grad_norm": 0.05471871793270111, "kl": 0.10010669939219952, "learning_rate": 4.866491319948809e-06, "loss": 0.0005004426930099726, "num_tokens": 67819908.0, "reward": 2.282275438308716, "reward_std": 0.522857666015625, "rewards/code_complexity_reward/mean": 0.8539062142372131, "rewards/code_complexity_reward/std": 0.1180996522307396, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 342, "step_time": 59.378247554413974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 383.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 151.69140625, "completions/mean_terminated_length": 151.69140625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2294992571696639, "epoch": 0.19544159544159545, "frac_reward_zero_std": 0.171875, "grad_norm": 0.04970858618617058, "kl": 0.09624269435880706, "learning_rate": 4.864882909969797e-06, "loss": 0.00048116646939888597, "num_tokens": 67965862.0, "reward": 2.343310594558716, "reward_std": 0.4925476610660553, "rewards/code_complexity_reward/mean": 0.8822265863418579, "rewards/code_complexity_reward/std": 0.07774035632610321, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.052975114434957504, "step": 343, "step_time": 46.930439413525164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 144.990234375, "completions/mean_terminated_length": 144.990234375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.239232930354774, "epoch": 0.196011396011396, "frac_reward_zero_std": 0.203125, "grad_norm": 0.06214551627635956, "kl": 0.10309218621114269, "learning_rate": 4.863265138500629e-06, "loss": 0.0005155018297955394, "num_tokens": 68108873.0, "reward": 2.3194823265075684, "reward_std": 0.4982823431491852, "rewards/code_complexity_reward/mean": 0.865234375, "rewards/code_complexity_reward/std": 0.10122702270746231, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 344, "step_time": 48.62362785451114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 458.0, "completions/mean_length": 151.20703125, "completions/mean_terminated_length": 150.5009765625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.224079177249223, "epoch": 0.19658119658119658, "frac_reward_zero_std": 0.140625, "grad_norm": 0.059596240520477295, "kl": 0.10302799928467721, "learning_rate": 4.8616380119453244e-06, "loss": 0.0005149564240127802, "num_tokens": 68254851.0, "reward": 2.3068361282348633, "reward_std": 0.50467449426651, "rewards/code_complexity_reward/mean": 0.8567382097244263, "rewards/code_complexity_reward/std": 0.11247310787439346, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 345, "step_time": 58.8192940633744 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 149.328125, "completions/mean_terminated_length": 149.328125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2323521226644516, "epoch": 0.19715099715099715, "frac_reward_zero_std": 0.078125, "grad_norm": 0.0645022913813591, "kl": 0.09946052468148991, "learning_rate": 4.8600015367449304e-06, "loss": 0.0004971304442733526, "num_tokens": 68401331.0, "reward": 2.3341798782348633, "reward_std": 0.533314049243927, "rewards/code_complexity_reward/mean": 0.857714831829071, "rewards/code_complexity_reward/std": 0.12607289850711823, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 346, "step_time": 95.23103968705982 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 147.275390625, "completions/mean_terminated_length": 147.275390625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2155011638533324, "epoch": 0.19772079772079773, "frac_reward_zero_std": 0.15625, "grad_norm": 0.07251669466495514, "kl": 0.10179346543736756, "learning_rate": 4.858355719377506e-06, "loss": 0.0005090250633656979, "num_tokens": 68544688.0, "reward": 2.4100098609924316, "reward_std": 0.5524970889091492, "rewards/code_complexity_reward/mean": 0.8581054210662842, "rewards/code_complexity_reward/std": 0.12515243887901306, "rewards/code_execution_reward/mean": 0.455078125, "rewards/code_execution_reward/std": 0.4984649419784546, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 347, "step_time": 59.973932416178286 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 154.37890625, "completions/mean_terminated_length": 153.67906188964844, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23219784372486174, "epoch": 0.19829059829059828, "frac_reward_zero_std": 0.109375, "grad_norm": 0.05963036045432091, "kl": 0.09421143512008712, "learning_rate": 4.85670056635809e-06, "loss": 0.0004711585643235594, "num_tokens": 68692170.0, "reward": 2.333789348602295, "reward_std": 0.5305384397506714, "rewards/code_complexity_reward/mean": 0.8626952767372131, "rewards/code_complexity_reward/std": 0.11626510322093964, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 348, "step_time": 76.92646268568933 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 414.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 156.947265625, "completions/mean_terminated_length": 156.947265625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.221119096968323, "epoch": 0.19886039886039886, "frac_reward_zero_std": 0.171875, "grad_norm": 0.05397028475999832, "kl": 0.0973411209997721, "learning_rate": 4.855036084238675e-06, "loss": 0.0004866704111918807, "num_tokens": 68838103.0, "reward": 2.3066892623901367, "reward_std": 0.5195214748382568, "rewards/code_complexity_reward/mean": 0.8480468988418579, "rewards/code_complexity_reward/std": 0.11719191819429398, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 349, "step_time": 60.21091307140887 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 407.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 150.302734375, "completions/mean_terminated_length": 150.302734375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2292138240300119, "epoch": 0.19943019943019943, "frac_reward_zero_std": 0.1875, "grad_norm": 0.05296735465526581, "kl": 0.09218397224321961, "learning_rate": 4.853362279608186e-06, "loss": 0.00046056712744757533, "num_tokens": 68985042.0, "reward": 2.3117189407348633, "reward_std": 0.5391696691513062, "rewards/code_complexity_reward/mean": 0.852343738079071, "rewards/code_complexity_reward/std": 0.140150785446167, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 350, "step_time": 60.06627430021763 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 440.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 154.40234375, "completions/mean_terminated_length": 154.40234375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23418272729031742, "epoch": 0.2, "frac_reward_zero_std": 0.171875, "grad_norm": 0.057674624025821686, "kl": 0.13017030025366694, "learning_rate": 4.851679159092449e-06, "loss": 0.0006511928513646126, "num_tokens": 69135488.0, "reward": 2.325000047683716, "reward_std": 0.49647486209869385, "rewards/code_complexity_reward/mean": 0.8695312738418579, "rewards/code_complexity_reward/std": 0.08913244307041168, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 351, "step_time": 51.424807409755886 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 442.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 141.80859375, "completions/mean_terminated_length": 141.80859375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2397836863528937, "epoch": 0.20056980056980056, "frac_reward_zero_std": 0.15625, "grad_norm": 0.057100530713796616, "kl": 0.12557518715038896, "learning_rate": 4.849986729354169e-06, "loss": 0.0006278535001911223, "num_tokens": 69274902.0, "reward": 2.3310546875, "reward_std": 0.5391520261764526, "rewards/code_complexity_reward/mean": 0.8677734136581421, "rewards/code_complexity_reward/std": 0.11954955011606216, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 352, "step_time": 52.84126944001764 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 480.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 161.9921875, "completions/mean_terminated_length": 161.9921875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23569034691900015, "epoch": 0.20113960113960114, "frac_reward_zero_std": 0.125, "grad_norm": 0.07084424793720245, "kl": 0.08694440737599507, "learning_rate": 4.848284997092902e-06, "loss": 0.0004344282206147909, "num_tokens": 69430930.0, "reward": 2.2867677211761475, "reward_std": 0.497893750667572, "rewards/code_complexity_reward/mean": 0.8623046875, "rewards/code_complexity_reward/std": 0.12044015526771545, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 353, "step_time": 55.82749448250979 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 454.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 142.51171875, "completions/mean_terminated_length": 142.51171875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23933249013498425, "epoch": 0.20170940170940171, "frac_reward_zero_std": 0.25, "grad_norm": 0.053901560604572296, "kl": 0.10093678336124867, "learning_rate": 4.846573969045028e-06, "loss": 0.0005049030878581107, "num_tokens": 69568816.0, "reward": 2.328662157058716, "reward_std": 0.5301940441131592, "rewards/code_complexity_reward/mean": 0.8705078363418579, "rewards/code_complexity_reward/std": 0.12640340626239777, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 354, "step_time": 63.73285282123834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 453.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 146.125, "completions/mean_terminated_length": 146.125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23228923068381846, "epoch": 0.2022792022792023, "frac_reward_zero_std": 0.125, "grad_norm": 0.06359187513589859, "kl": 0.11074511747574434, "learning_rate": 4.844853651983723e-06, "loss": 0.0005540888523682952, "num_tokens": 69711104.0, "reward": 2.3064942359924316, "reward_std": 0.5113946795463562, "rewards/code_complexity_reward/mean": 0.8727538585662842, "rewards/code_complexity_reward/std": 0.11061486601829529, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 355, "step_time": 51.22994702216238 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 146.96484375, "completions/mean_terminated_length": 145.53334045410156, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2367454944178462, "epoch": 0.20284900284900284, "frac_reward_zero_std": 0.1875, "grad_norm": 0.06718358397483826, "kl": 0.09823031554697081, "learning_rate": 4.8431240527189385e-06, "loss": 0.0004912135773338377, "num_tokens": 69852374.0, "reward": 2.284472703933716, "reward_std": 0.5344566702842712, "rewards/code_complexity_reward/mean": 0.8607421517372131, "rewards/code_complexity_reward/std": 0.13323402404785156, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.019099153578281403, "step": 356, "step_time": 50.63188935443759 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 159.349609375, "completions/mean_terminated_length": 156.5728302001953, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.23367707105353475, "epoch": 0.20341880341880342, "frac_reward_zero_std": 0.140625, "grad_norm": 0.057034365832805634, "kl": 0.09964394144481048, "learning_rate": 4.841385178097365e-06, "loss": 0.0004982081591151655, "num_tokens": 70003857.0, "reward": 2.29150390625, "reward_std": 0.5312735438346863, "rewards/code_complexity_reward/mean": 0.8584960699081421, "rewards/code_complexity_reward/std": 0.12977129220962524, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 357, "step_time": 57.38605337776244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 156.99609375, "completions/mean_terminated_length": 155.6039276123047, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2268886943347752, "epoch": 0.203988603988604, "frac_reward_zero_std": 0.109375, "grad_norm": 0.05658808350563049, "kl": 0.10350929177366197, "learning_rate": 4.839637035002412e-06, "loss": 0.0005178060382604599, "num_tokens": 70153455.0, "reward": 2.287109613418579, "reward_std": 0.5226520895957947, "rewards/code_complexity_reward/mean": 0.8542969226837158, "rewards/code_complexity_reward/std": 0.13018536567687988, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 358, "step_time": 66.46239480376244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 159.26953125, "completions/mean_terminated_length": 157.8862762451172, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23978349845856428, "epoch": 0.20455840455840457, "frac_reward_zero_std": 0.125, "grad_norm": 0.06592671573162079, "kl": 0.13302609813399613, "learning_rate": 4.837879630354181e-06, "loss": 0.0006644241511821747, "num_tokens": 70308153.0, "reward": 2.2564451694488525, "reward_std": 0.5376686453819275, "rewards/code_complexity_reward/mean": 0.8495117425918579, "rewards/code_complexity_reward/std": 0.1424373835325241, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 359, "step_time": 76.57112909667194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 148.470703125, "completions/mean_terminated_length": 148.470703125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2307477192953229, "epoch": 0.20512820512820512, "frac_reward_zero_std": 0.171875, "grad_norm": 0.05658668279647827, "kl": 0.09760747570544481, "learning_rate": 4.836112971109431e-06, "loss": 0.0004879760090261698, "num_tokens": 70452746.0, "reward": 2.3380861282348633, "reward_std": 0.5451481938362122, "rewards/code_complexity_reward/mean": 0.867968738079071, "rewards/code_complexity_reward/std": 0.12980255484580994, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 360, "step_time": 49.69948194269091 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 397.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 144.587890625, "completions/mean_terminated_length": 144.587890625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22705362271517515, "epoch": 0.2056980056980057, "frac_reward_zero_std": 0.15625, "grad_norm": 0.06435813009738922, "kl": 0.10361670097336173, "learning_rate": 4.834337064261559e-06, "loss": 0.0005179772269912064, "num_tokens": 70595871.0, "reward": 2.3896484375, "reward_std": 0.5161484479904175, "rewards/code_complexity_reward/mean": 0.884082019329071, "rewards/code_complexity_reward/std": 0.09334790706634521, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 361, "step_time": 53.78294870629907 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 145.630859375, "completions/mean_terminated_length": 145.630859375, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.23833679454401135, "epoch": 0.20626780626780628, "frac_reward_zero_std": 0.140625, "grad_norm": 0.06624864041805267, "kl": 0.11590484448242933, "learning_rate": 4.832551916840568e-06, "loss": 0.0005798835190944374, "num_tokens": 70736138.0, "reward": 2.3341798782348633, "reward_std": 0.535550057888031, "rewards/code_complexity_reward/mean": 0.868945300579071, "rewards/code_complexity_reward/std": 0.11990071833133698, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 362, "step_time": 52.325816456228495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 152.087890625, "completions/mean_terminated_length": 150.67648315429688, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.2282409332692623, "epoch": 0.20683760683760682, "frac_reward_zero_std": 0.171875, "grad_norm": 0.055910222232341766, "kl": 0.10350898315664381, "learning_rate": 4.830757535913042e-06, "loss": 0.0005174755351617932, "num_tokens": 70879439.0, "reward": 2.300585985183716, "reward_std": 0.5237950086593628, "rewards/code_complexity_reward/mean": 0.8622070550918579, "rewards/code_complexity_reward/std": 0.12071383744478226, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 363, "step_time": 66.22473243251443 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 142.19140625, "completions/mean_terminated_length": 141.46771240234375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23261913331225514, "epoch": 0.2074074074074074, "frac_reward_zero_std": 0.1875, "grad_norm": 0.05968683585524559, "kl": 0.10438659164356068, "learning_rate": 4.828953928582112e-06, "loss": 0.0005220535094849765, "num_tokens": 71018993.0, "reward": 2.31396484375, "reward_std": 0.5045079588890076, "rewards/code_complexity_reward/mean": 0.8687499761581421, "rewards/code_complexity_reward/std": 0.10246472805738449, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 364, "step_time": 125.18809331394732 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 505.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 158.3671875, "completions/mean_terminated_length": 158.3671875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.22778274351730943, "epoch": 0.20797720797720798, "frac_reward_zero_std": 0.125, "grad_norm": 0.0597526952624321, "kl": 0.09007815778022632, "learning_rate": 4.827141101987437e-06, "loss": 0.0004501529037952423, "num_tokens": 71168965.0, "reward": 2.321533203125, "reward_std": 0.5049015283584595, "rewards/code_complexity_reward/mean": 0.8679687976837158, "rewards/code_complexity_reward/std": 0.09399469196796417, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 365, "step_time": 55.30121125560254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 156.59765625, "completions/mean_terminated_length": 155.9021453857422, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2333914889022708, "epoch": 0.20854700854700856, "frac_reward_zero_std": 0.234375, "grad_norm": 0.05414281040430069, "kl": 0.09275628463365138, "learning_rate": 4.825319063305167e-06, "loss": 0.0004638570244424045, "num_tokens": 71316383.0, "reward": 2.3685548305511475, "reward_std": 0.5155791640281677, "rewards/code_complexity_reward/mean": 0.8685547113418579, "rewards/code_complexity_reward/std": 0.103403240442276, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 366, "step_time": 57.139071800746024 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 464.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 147.18359375, "completions/mean_terminated_length": 147.18359375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23894177516922355, "epoch": 0.2091168091168091, "frac_reward_zero_std": 0.171875, "grad_norm": 0.062093593180179596, "kl": 0.10227628494612873, "learning_rate": 4.823487819747922e-06, "loss": 0.0005110683850944042, "num_tokens": 71459725.0, "reward": 2.3283205032348633, "reward_std": 0.5050321221351624, "rewards/code_complexity_reward/mean": 0.8787108659744263, "rewards/code_complexity_reward/std": 0.10080836713314056, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 367, "step_time": 56.2068176548928 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 155.0078125, "completions/mean_terminated_length": 154.3092041015625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24238335737027228, "epoch": 0.20968660968660968, "frac_reward_zero_std": 0.171875, "grad_norm": 0.05929528921842575, "kl": 0.1017161826021038, "learning_rate": 4.821647378564756e-06, "loss": 0.0005085846642032266, "num_tokens": 71609945.0, "reward": 2.2590818405151367, "reward_std": 0.518269956111908, "rewards/code_complexity_reward/mean": 0.8597655892372131, "rewards/code_complexity_reward/std": 0.13086473941802979, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 368, "step_time": 68.34525067079812 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 499.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 146.03125, "completions/mean_terminated_length": 146.03125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22790051973424852, "epoch": 0.21025641025641026, "frac_reward_zero_std": 0.1875, "grad_norm": 0.05446599796414375, "kl": 0.1144609444309026, "learning_rate": 4.819797747041135e-06, "loss": 0.0005723665235564113, "num_tokens": 71752385.0, "reward": 2.339062452316284, "reward_std": 0.5033814311027527, "rewards/code_complexity_reward/mean": 0.8757811784744263, "rewards/code_complexity_reward/std": 0.09935887157917023, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 369, "step_time": 47.34277326799929 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 451.0, "completions/mean_length": 155.673828125, "completions/mean_terminated_length": 154.9765167236328, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2364111423958093, "epoch": 0.21082621082621084, "frac_reward_zero_std": 0.15625, "grad_norm": 0.06180352717638016, "kl": 0.0938318723347038, "learning_rate": 4.817938932498904e-06, "loss": 0.0004691376816481352, "num_tokens": 71901946.0, "reward": 2.299267530441284, "reward_std": 0.5207774043083191, "rewards/code_complexity_reward/mean": 0.861621081829071, "rewards/code_complexity_reward/std": 0.11660858243703842, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 370, "step_time": 50.022393234074116 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 147.07421875, "completions/mean_terminated_length": 147.07421875, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.2388331894762814, "epoch": 0.21139601139601139, "frac_reward_zero_std": 0.09375, "grad_norm": 0.06027046591043472, "kl": 0.10249761486193165, "learning_rate": 4.816070942296262e-06, "loss": 0.0005123636801727116, "num_tokens": 72046312.0, "reward": 2.280468702316284, "reward_std": 0.5202639102935791, "rewards/code_complexity_reward/mean": 0.864062488079071, "rewards/code_complexity_reward/std": 0.12689848244190216, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.03121940791606903, "step": 371, "step_time": 57.02936505898833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 427.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 144.142578125, "completions/mean_terminated_length": 144.142578125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23651538277044892, "epoch": 0.21196581196581196, "frac_reward_zero_std": 0.21875, "grad_norm": 0.05768340080976486, "kl": 0.10920307965716347, "learning_rate": 4.814193783827725e-06, "loss": 0.0005461292457766831, "num_tokens": 72189009.0, "reward": 2.3105955123901367, "reward_std": 0.5105408430099487, "rewards/code_complexity_reward/mean": 0.8719726800918579, "rewards/code_complexity_reward/std": 0.10952939093112946, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 372, "step_time": 52.413762249052525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 156.771484375, "completions/mean_terminated_length": 156.07632446289062, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2302998041268438, "epoch": 0.21253561253561254, "frac_reward_zero_std": 0.0625, "grad_norm": 0.06568866223096848, "kl": 0.09577057213755324, "learning_rate": 4.812307464524107e-06, "loss": 0.0004787927900906652, "num_tokens": 72334484.0, "reward": 2.255419969558716, "reward_std": 0.51805579662323, "rewards/code_complexity_reward/mean": 0.8604491949081421, "rewards/code_complexity_reward/std": 0.12745894491672516, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.025282343849539757, "step": 373, "step_time": 64.14088142383844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 336.0, "completions/max_terminated_length": 336.0, "completions/mean_length": 144.97265625, "completions/mean_terminated_length": 144.97265625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2300473835784942, "epoch": 0.21310541310541312, "frac_reward_zero_std": 0.140625, "grad_norm": 0.06847607344388962, "kl": 0.10110017948318273, "learning_rate": 4.8104119918524825e-06, "loss": 0.0005057142116129398, "num_tokens": 72476862.0, "reward": 2.312744379043579, "reward_std": 0.5078011155128479, "rewards/code_complexity_reward/mean": 0.8765624761581421, "rewards/code_complexity_reward/std": 0.10560374706983566, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 374, "step_time": 95.48400118015707 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 417.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 154.30859375, "completions/mean_terminated_length": 154.30859375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23605824867263436, "epoch": 0.21367521367521367, "frac_reward_zero_std": 0.125, "grad_norm": 0.06280485540628433, "kl": 0.09797648567473516, "learning_rate": 4.808507373316163e-06, "loss": 0.0004898573970422149, "num_tokens": 72623068.0, "reward": 2.3014650344848633, "reward_std": 0.49422580003738403, "rewards/code_complexity_reward/mean": 0.873339831829071, "rewards/code_complexity_reward/std": 0.0987553521990776, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 375, "step_time": 51.932272852398455 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 158.974609375, "completions/mean_terminated_length": 158.28375244140625, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.23026850819587708, "epoch": 0.21424501424501424, "frac_reward_zero_std": 0.09375, "grad_norm": 0.062454257160425186, "kl": 0.09334030729951337, "learning_rate": 4.80659361645466e-06, "loss": 0.00046661688247695565, "num_tokens": 72773631.0, "reward": 2.2274904251098633, "reward_std": 0.5314382314682007, "rewards/code_complexity_reward/mean": 0.83740234375, "rewards/code_complexity_reward/std": 0.1497306525707245, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 376, "step_time": 58.043671385385096 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 142.82421875, "completions/mean_terminated_length": 142.82421875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23520746594294906, "epoch": 0.21481481481481482, "frac_reward_zero_std": 0.171875, "grad_norm": 0.05405892804265022, "kl": 0.1059423114056699, "learning_rate": 4.804670728843665e-06, "loss": 0.0005296836025081575, "num_tokens": 72914949.0, "reward": 2.3213868141174316, "reward_std": 0.5397113561630249, "rewards/code_complexity_reward/mean": 0.872753918170929, "rewards/code_complexity_reward/std": 0.12974773347377777, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 377, "step_time": 57.93982150219381 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 383.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 151.62109375, "completions/mean_terminated_length": 151.62109375, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.23453700100071728, "epoch": 0.2153846153846154, "frac_reward_zero_std": 0.171875, "grad_norm": 0.0607321672141552, "kl": 0.10028167878044769, "learning_rate": 4.802738718095007e-06, "loss": 0.0005014139460399747, "num_tokens": 73062139.0, "reward": 2.313720703125, "reward_std": 0.533279538154602, "rewards/code_complexity_reward/mean": 0.8692382574081421, "rewards/code_complexity_reward/std": 0.12285738438367844, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 378, "step_time": 88.41903579421341 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 149.732421875, "completions/mean_terminated_length": 149.732421875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24245142936706543, "epoch": 0.21595441595441595, "frac_reward_zero_std": 0.140625, "grad_norm": 0.05878087133169174, "kl": 0.10280436422908679, "learning_rate": 4.800797591856637e-06, "loss": 0.0005140528082847595, "num_tokens": 73208466.0, "reward": 2.2903809547424316, "reward_std": 0.5507081151008606, "rewards/code_complexity_reward/mean": 0.858593761920929, "rewards/code_complexity_reward/std": 0.14105547964572906, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 379, "step_time": 44.62781197577715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 145.783203125, "completions/mean_terminated_length": 145.783203125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2390093521680683, "epoch": 0.21652421652421652, "frac_reward_zero_std": 0.21875, "grad_norm": 0.059238556772470474, "kl": 0.1046454164898023, "learning_rate": 4.798847357812583e-06, "loss": 0.0005232612602412701, "num_tokens": 73352523.0, "reward": 2.3384766578674316, "reward_std": 0.5214517712593079, "rewards/code_complexity_reward/mean": 0.8791015148162842, "rewards/code_complexity_reward/std": 0.12497659772634506, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 380, "step_time": 49.5003852462396 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 143.26171875, "completions/mean_terminated_length": 142.5401153564453, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22964544035494328, "epoch": 0.2170940170940171, "frac_reward_zero_std": 0.171875, "grad_norm": 0.06376112252473831, "kl": 0.12055987888015807, "learning_rate": 4.796888023682932e-06, "loss": 0.0006028059870004654, "num_tokens": 73493585.0, "reward": 2.2685060501098633, "reward_std": 0.5069822072982788, "rewards/code_complexity_reward/mean": 0.867968738079071, "rewards/code_complexity_reward/std": 0.12794236838817596, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 381, "step_time": 59.13478536531329 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 495.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 151.86328125, "completions/mean_terminated_length": 151.86328125, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.24602180696092546, "epoch": 0.21766381766381768, "frac_reward_zero_std": 0.140625, "grad_norm": 0.06692658364772797, "kl": 0.09973249165341258, "learning_rate": 4.794919597223791e-06, "loss": 0.000498540117405355, "num_tokens": 73640675.0, "reward": 2.3249998092651367, "reward_std": 0.5115304589271545, "rewards/code_complexity_reward/mean": 0.8604491949081421, "rewards/code_complexity_reward/std": 0.11104904860258102, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 382, "step_time": 63.704593162983656 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 147.865234375, "completions/mean_terminated_length": 147.865234375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23704657168127596, "epoch": 0.21823361823361823, "frac_reward_zero_std": 0.203125, "grad_norm": 0.060335323214530945, "kl": 0.10662510863039643, "learning_rate": 4.7929420862272605e-06, "loss": 0.0005330734420567751, "num_tokens": 73782894.0, "reward": 2.2697267532348633, "reward_std": 0.48260498046875, "rewards/code_complexity_reward/mean": 0.8874999284744263, "rewards/code_complexity_reward/std": 0.10217784345149994, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 383, "step_time": 57.15840154886246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 152.712890625, "completions/mean_terminated_length": 151.30392456054688, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2424948054831475, "epoch": 0.2188034188034188, "frac_reward_zero_std": 0.1875, "grad_norm": 0.06483165174722672, "kl": 0.10853293602121994, "learning_rate": 4.790955498521403e-06, "loss": 0.0005426938878372312, "num_tokens": 73931883.0, "reward": 2.23193359375, "reward_std": 0.5218027830123901, "rewards/code_complexity_reward/mean": 0.8570312261581421, "rewards/code_complexity_reward/std": 0.14738215506076813, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 384, "step_time": 67.6604725997895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 473.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 144.66796875, "completions/mean_terminated_length": 144.66796875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23518654983490705, "epoch": 0.21937321937321938, "frac_reward_zero_std": 0.25, "grad_norm": 0.07011569291353226, "kl": 0.10588565794751048, "learning_rate": 4.788959841970211e-06, "loss": 0.0005293108988553286, "num_tokens": 74077409.0, "reward": 2.2406249046325684, "reward_std": 0.4914807975292206, "rewards/code_complexity_reward/mean": 0.8698241710662842, "rewards/code_complexity_reward/std": 0.1162550300359726, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 385, "step_time": 52.48955264594406 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 397.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 145.51171875, "completions/mean_terminated_length": 145.51171875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22778192232362926, "epoch": 0.21994301994301993, "frac_reward_zero_std": 0.1875, "grad_norm": 0.05736496299505234, "kl": 0.11570135795045644, "learning_rate": 4.786955124473575e-06, "loss": 0.0005789014976471663, "num_tokens": 74220871.0, "reward": 2.347900390625, "reward_std": 0.5300710201263428, "rewards/code_complexity_reward/mean": 0.8633788824081421, "rewards/code_complexity_reward/std": 0.1118764877319336, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 386, "step_time": 49.61705166101456 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 154.515625, "completions/mean_terminated_length": 153.11373901367188, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23862961563281715, "epoch": 0.2205128205128205, "frac_reward_zero_std": 0.140625, "grad_norm": 0.054498206824064255, "kl": 0.09683412604499608, "learning_rate": 4.784941353967255e-06, "loss": 0.0004844882641918957, "num_tokens": 74371191.0, "reward": 2.3221678733825684, "reward_std": 0.5298587083816528, "rewards/code_complexity_reward/mean": 0.868847668170929, "rewards/code_complexity_reward/std": 0.13267987966537476, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 387, "step_time": 66.367371478118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 383.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 147.505859375, "completions/mean_terminated_length": 147.505859375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.22552934614941478, "epoch": 0.22108262108262108, "frac_reward_zero_std": 0.15625, "grad_norm": 0.059239357709884644, "kl": 0.10043734248029068, "learning_rate": 4.7829185384228495e-06, "loss": 0.0005021824035793543, "num_tokens": 74515738.0, "reward": 2.3795409202575684, "reward_std": 0.5259068012237549, "rewards/code_complexity_reward/mean": 0.87353515625, "rewards/code_complexity_reward/std": 0.09841106086969376, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 388, "step_time": 46.677604511380196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 399.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 155.830078125, "completions/mean_terminated_length": 155.830078125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23077873815782368, "epoch": 0.22165242165242166, "frac_reward_zero_std": 0.1875, "grad_norm": 0.05800984054803848, "kl": 0.09384048980427906, "learning_rate": 4.780886685847759e-06, "loss": 0.00046931602992117405, "num_tokens": 74663643.0, "reward": 2.2461915016174316, "reward_std": 0.4881473779678345, "rewards/code_complexity_reward/mean": 0.8615233898162842, "rewards/code_complexity_reward/std": 0.12116533517837524, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 389, "step_time": 69.50827107019722 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 148.583984375, "completions/mean_terminated_length": 148.583984375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23318591946735978, "epoch": 0.2222222222222222, "frac_reward_zero_std": 0.1875, "grad_norm": 0.05914246290922165, "kl": 0.09923542273463681, "learning_rate": 4.778845804285158e-06, "loss": 0.0004961935919709504, "num_tokens": 74808254.0, "reward": 2.2594728469848633, "reward_std": 0.5005995631217957, "rewards/code_complexity_reward/mean": 0.873828113079071, "rewards/code_complexity_reward/std": 0.12445022910833359, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 390, "step_time": 48.800742011517286 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 433.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 165.341796875, "completions/mean_terminated_length": 165.341796875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23137721163220704, "epoch": 0.2227920227920228, "frac_reward_zero_std": 0.140625, "grad_norm": 0.06648840010166168, "kl": 0.09086785221006721, "learning_rate": 4.776795901813966e-06, "loss": 0.00045445235446095467, "num_tokens": 74963933.0, "reward": 2.255175828933716, "reward_std": 0.5120323896408081, "rewards/code_complexity_reward/mean": 0.8412109613418579, "rewards/code_complexity_reward/std": 0.13873519003391266, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 391, "step_time": 56.46448375377804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 155.787109375, "completions/mean_terminated_length": 154.39019775390625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.23077476304024458, "epoch": 0.22336182336182336, "frac_reward_zero_std": 0.1875, "grad_norm": 0.0767553523182869, "kl": 0.10012833401560783, "learning_rate": 4.774736986548807e-06, "loss": 0.0005004761042073369, "num_tokens": 75113256.0, "reward": 2.2643556594848633, "reward_std": 0.5120576620101929, "rewards/code_complexity_reward/mean": 0.862109363079071, "rewards/code_complexity_reward/std": 0.12897247076034546, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 392, "step_time": 58.05604039039463 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 350.0, "completions/max_terminated_length": 350.0, "completions/mean_length": 143.228515625, "completions/mean_terminated_length": 143.228515625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24102211510762572, "epoch": 0.22393162393162394, "frac_reward_zero_std": 0.171875, "grad_norm": 0.054083727300167084, "kl": 0.10318055015522987, "learning_rate": 4.772669066639988e-06, "loss": 0.0005162570159882307, "num_tokens": 75255381.0, "reward": 2.3717775344848633, "reward_std": 0.5419120192527771, "rewards/code_complexity_reward/mean": 0.870410144329071, "rewards/code_complexity_reward/std": 0.11648986488580704, "rewards/code_execution_reward/mean": 0.40625, "rewards/code_execution_reward/std": 0.49161264300346375, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 393, "step_time": 38.72660363651812 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 414.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 156.412109375, "completions/mean_terminated_length": 156.412109375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23246066085994244, "epoch": 0.2245014245014245, "frac_reward_zero_std": 0.109375, "grad_norm": 0.06032125651836395, "kl": 0.09730935998959467, "learning_rate": 4.7705921502734555e-06, "loss": 0.0004867383686359972, "num_tokens": 75405720.0, "reward": 2.2562012672424316, "reward_std": 0.5013161301612854, "rewards/code_complexity_reward/mean": 0.8646484613418579, "rewards/code_complexity_reward/std": 0.12147282809019089, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 394, "step_time": 45.10282030608505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 471.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 142.458984375, "completions/mean_terminated_length": 142.458984375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2220670550595969, "epoch": 0.22507122507122507, "frac_reward_zero_std": 0.28125, "grad_norm": 0.054095618426799774, "kl": 0.09671357076149434, "learning_rate": 4.768506245670773e-06, "loss": 0.0004833896819036454, "num_tokens": 75549331.0, "reward": 2.3844728469848633, "reward_std": 0.5328982472419739, "rewards/code_complexity_reward/mean": 0.8743164539337158, "rewards/code_complexity_reward/std": 0.11416129022836685, "rewards/code_execution_reward/mean": 0.4140625, "rewards/code_execution_reward/std": 0.49304109811782837, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 395, "step_time": 46.71663156989962 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 382.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 144.392578125, "completions/mean_terminated_length": 144.392578125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2222644668072462, "epoch": 0.22564102564102564, "frac_reward_zero_std": 0.125, "grad_norm": 0.059584278613328934, "kl": 0.09283798281103373, "learning_rate": 4.766411361089083e-06, "loss": 0.0004641050472855568, "num_tokens": 75690276.0, "reward": 2.30615234375, "reward_std": 0.48075103759765625, "rewards/code_complexity_reward/mean": 0.8838866949081421, "rewards/code_complexity_reward/std": 0.08577452600002289, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 396, "step_time": 40.42865873966366 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 437.0, "completions/mean_length": 145.65234375, "completions/mean_terminated_length": 144.9354248046875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2310574329458177, "epoch": 0.22621082621082622, "frac_reward_zero_std": 0.125, "grad_norm": 0.0596124492585659, "kl": 0.10186422237893566, "learning_rate": 4.764307504821076e-06, "loss": 0.0005095530650578439, "num_tokens": 75833970.0, "reward": 2.293896436691284, "reward_std": 0.5083000659942627, "rewards/code_complexity_reward/mean": 0.863085925579071, "rewards/code_complexity_reward/std": 0.11651598662137985, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 397, "step_time": 58.57821589987725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 457.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 146.23046875, "completions/mean_terminated_length": 146.23046875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.22417170903645456, "epoch": 0.22678062678062677, "frac_reward_zero_std": 0.171875, "grad_norm": 0.061599042266607285, "kl": 0.09282808873103932, "learning_rate": 4.7621946851949565e-06, "loss": 0.0004642562416847795, "num_tokens": 75977736.0, "reward": 2.3265626430511475, "reward_std": 0.5279076099395752, "rewards/code_complexity_reward/mean": 0.8720703125, "rewards/code_complexity_reward/std": 0.12481378763914108, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 398, "step_time": 63.301699684001505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 145.064453125, "completions/mean_terminated_length": 144.34637451171875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23352391039952636, "epoch": 0.22735042735042735, "frac_reward_zero_std": 0.234375, "grad_norm": 0.054934870451688766, "kl": 0.0983636777382344, "learning_rate": 4.760072910574413e-06, "loss": 0.000491857819724828, "num_tokens": 76119713.0, "reward": 2.3792481422424316, "reward_std": 0.5358026027679443, "rewards/code_complexity_reward/mean": 0.8788085579872131, "rewards/code_complexity_reward/std": 0.10568546503782272, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 399, "step_time": 67.6245833504945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 138.69921875, "completions/mean_terminated_length": 138.69921875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.230343195842579, "epoch": 0.22792022792022792, "frac_reward_zero_std": 0.15625, "grad_norm": 0.06333411484956741, "kl": 0.09574662847444415, "learning_rate": 4.757942189358579e-06, "loss": 0.0004787701473105699, "num_tokens": 76257943.0, "reward": 2.315722942352295, "reward_std": 0.5168415307998657, "rewards/code_complexity_reward/mean": 0.8846679925918579, "rewards/code_complexity_reward/std": 0.10437722504138947, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 400, "step_time": 76.49051422812045 }, { "epoch": 0.22792022792022792, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.00125, "eval_completions/max_length": 209.48, "eval_completions/max_terminated_length": 207.19, "eval_completions/mean_length": 149.26375, "eval_completions/mean_terminated_length": 148.91910720825194, "eval_completions/min_length": 104.91, "eval_completions/min_terminated_length": 104.91, "eval_entropy": 0.22828682094812394, "eval_frac_reward_zero_std": 0.21, "eval_kl": 0.09034340467303992, "eval_loss": 0.001160123967565596, "eval_num_tokens": 76257943.0, "eval_reward": 2.242843818664551, "eval_reward_std": 0.22474357288330793, "eval_rewards/code_complexity_reward/mean": 0.8717499941587448, "eval_rewards/code_complexity_reward/std": 0.05462637720629573, "eval_rewards/code_execution_reward/mean": 0.27625, "eval_rewards/code_execution_reward/std": 0.17604609519243242, "eval_rewards/code_syntax_reward/mean": 0.495, "eval_rewards/code_syntax_reward/std": 0.014142135381698609, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.49984375, "eval_rewards/xmlcount_reward_func/std": 0.0004419417306780815, "eval_runtime": 996.2596, "eval_samples_per_second": 0.1, "eval_steps_per_second": 0.013, "step": 400 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 372.0, "completions/max_terminated_length": 372.0, "completions/mean_length": 146.57421875, "completions/mean_terminated_length": 146.57421875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22583085135556757, "epoch": 0.2284900284900285, "frac_reward_zero_std": 0.046875, "grad_norm": 0.06648116558790207, "kl": 0.0900637965532951, "learning_rate": 4.7558025299820056e-06, "loss": 0.00045029696775600314, "num_tokens": 76401381.0, "reward": 2.28857421875, "reward_std": 0.5381723642349243, "rewards/code_complexity_reward/mean": 0.8575195074081421, "rewards/code_complexity_reward/std": 0.14033471047878265, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 401, "step_time": 59.295409094542265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 393.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 150.3203125, "completions/mean_terminated_length": 150.3203125, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.23731005494482815, "epoch": 0.22905982905982905, "frac_reward_zero_std": 0.171875, "grad_norm": 0.06680215150117874, "kl": 0.09433183132205158, "learning_rate": 4.753653940914627e-06, "loss": 0.0004715590039268136, "num_tokens": 76548505.0, "reward": 2.2190918922424316, "reward_std": 0.48294597864151, "rewards/code_complexity_reward/mean": 0.8644530773162842, "rewards/code_complexity_reward/std": 0.13158707320690155, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 402, "step_time": 109.09996950533241 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 155.19921875, "completions/mean_terminated_length": 154.5009765625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2338043455965817, "epoch": 0.22962962962962963, "frac_reward_zero_std": 0.234375, "grad_norm": 0.04923326149582863, "kl": 0.08646597003098577, "learning_rate": 4.751496430661725e-06, "loss": 0.00043202858068980277, "num_tokens": 76697767.0, "reward": 2.279247999191284, "reward_std": 0.4901078939437866, "rewards/code_complexity_reward/mean": 0.862109363079071, "rewards/code_complexity_reward/std": 0.11381930857896805, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 403, "step_time": 57.67293477896601 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 145.05859375, "completions/mean_terminated_length": 145.05859375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.22928040963597596, "epoch": 0.2301994301994302, "frac_reward_zero_std": 0.234375, "grad_norm": 0.05363072082400322, "kl": 0.10243148310109973, "learning_rate": 4.749330007763896e-06, "loss": 0.0005124016897752881, "num_tokens": 76840173.0, "reward": 2.388378858566284, "reward_std": 0.5330897569656372, "rewards/code_complexity_reward/mean": 0.871386706829071, "rewards/code_complexity_reward/std": 0.10259179770946503, "rewards/code_execution_reward/mean": 0.41796875, "rewards/code_execution_reward/std": 0.4937073290348053, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 404, "step_time": 80.23264241404831 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 156.232421875, "completions/mean_terminated_length": 155.5362091064453, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2228116039186716, "epoch": 0.23076923076923078, "frac_reward_zero_std": 0.171875, "grad_norm": 0.06846494972705841, "kl": 0.09087286109570414, "learning_rate": 4.747154680797017e-06, "loss": 0.0004541347734630108, "num_tokens": 76990268.0, "reward": 2.2374510765075684, "reward_std": 0.5205230712890625, "rewards/code_complexity_reward/mean": 0.865234375, "rewards/code_complexity_reward/std": 0.13979417085647583, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 405, "step_time": 59.487401072867215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 446.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 142.908203125, "completions/mean_terminated_length": 142.908203125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22534334543161094, "epoch": 0.23133903133903133, "frac_reward_zero_std": 0.1875, "grad_norm": 0.05635723099112511, "kl": 0.09723316028248519, "learning_rate": 4.744970458372215e-06, "loss": 0.0004863155772909522, "num_tokens": 77132117.0, "reward": 2.3431639671325684, "reward_std": 0.5174944996833801, "rewards/code_complexity_reward/mean": 0.880859375, "rewards/code_complexity_reward/std": 0.1136687621474266, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 406, "step_time": 62.35770201869309 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 146.767578125, "completions/mean_terminated_length": 146.05284118652344, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.21701563335955143, "epoch": 0.2319088319088319, "frac_reward_zero_std": 0.359375, "grad_norm": 0.05357591062784195, "kl": 0.09613490127958357, "learning_rate": 4.742777349135825e-06, "loss": 0.0004806377983186394, "num_tokens": 77277566.0, "reward": 2.400439500808716, "reward_std": 0.5436216592788696, "rewards/code_complexity_reward/mean": 0.8709960579872131, "rewards/code_complexity_reward/std": 0.1400083303451538, "rewards/code_execution_reward/mean": 0.43359375, "rewards/code_execution_reward/std": 0.4960552453994751, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 407, "step_time": 67.34623160772026 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 136.236328125, "completions/mean_terminated_length": 136.236328125, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.22334759403020144, "epoch": 0.23247863247863249, "frac_reward_zero_std": 0.203125, "grad_norm": 0.07225152105093002, "kl": 0.09725460287882015, "learning_rate": 4.740575361769365e-06, "loss": 0.000486302946228534, "num_tokens": 77414015.0, "reward": 2.345996141433716, "reward_std": 0.543398916721344, "rewards/code_complexity_reward/mean": 0.8807617425918579, "rewards/code_complexity_reward/std": 0.13557878136634827, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 408, "step_time": 94.75342650618404 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 475.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 141.373046875, "completions/mean_terminated_length": 141.373046875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23323393473401666, "epoch": 0.23304843304843303, "frac_reward_zero_std": 0.265625, "grad_norm": 0.05641573294997215, "kl": 0.09523543703835458, "learning_rate": 4.7383645049894965e-06, "loss": 0.0004760123265441507, "num_tokens": 77553830.0, "reward": 2.362548828125, "reward_std": 0.5289144515991211, "rewards/code_complexity_reward/mean": 0.887011706829071, "rewards/code_complexity_reward/std": 0.12148081511259079, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 409, "step_time": 52.842526357620955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 139.90625, "completions/mean_terminated_length": 138.4470672607422, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23549190908670425, "epoch": 0.2336182336182336, "frac_reward_zero_std": 0.140625, "grad_norm": 0.06754958629608154, "kl": 0.10298446466913447, "learning_rate": 4.736144787547991e-06, "loss": 0.0005148157943040133, "num_tokens": 77692686.0, "reward": 2.2013673782348633, "reward_std": 0.45527520775794983, "rewards/code_complexity_reward/mean": 0.8782227039337158, "rewards/code_complexity_reward/std": 0.12239202857017517, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 410, "step_time": 49.14226136635989 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 482.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 146.28125, "completions/mean_terminated_length": 146.28125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23066630237735808, "epoch": 0.2341880341880342, "frac_reward_zero_std": 0.171875, "grad_norm": 0.07670249044895172, "kl": 0.09063864056952298, "learning_rate": 4.733916218231693e-06, "loss": 0.00045327056432142854, "num_tokens": 77836494.0, "reward": 2.188037157058716, "reward_std": 0.48967793583869934, "rewards/code_complexity_reward/mean": 0.8578125238418579, "rewards/code_complexity_reward/std": 0.13737651705741882, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 411, "step_time": 56.18050103727728 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 428.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 152.150390625, "completions/mean_terminated_length": 152.150390625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2237495796289295, "epoch": 0.23475783475783477, "frac_reward_zero_std": 0.265625, "grad_norm": 0.06520096212625504, "kl": 0.08800314273685217, "learning_rate": 4.731678805862492e-06, "loss": 0.0004399816389195621, "num_tokens": 77980699.0, "reward": 2.3578126430511475, "reward_std": 0.5074759721755981, "rewards/code_complexity_reward/mean": 0.87939453125, "rewards/code_complexity_reward/std": 0.08560949563980103, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 412, "step_time": 60.93395243678242 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 487.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 140.978515625, "completions/mean_terminated_length": 140.978515625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22980292071588337, "epoch": 0.23532763532763531, "frac_reward_zero_std": 0.21875, "grad_norm": 0.05799083411693573, "kl": 0.10077406710479409, "learning_rate": 4.729432559297278e-06, "loss": 0.0005038646631874144, "num_tokens": 78120472.0, "reward": 2.3506836891174316, "reward_std": 0.5593337416648865, "rewards/code_complexity_reward/mean": 0.8747069835662842, "rewards/code_complexity_reward/std": 0.14402621984481812, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 413, "step_time": 86.06194609031081 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 454.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 143.7578125, "completions/mean_terminated_length": 143.7578125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.22510950220748782, "epoch": 0.2358974358974359, "frac_reward_zero_std": 0.21875, "grad_norm": 0.09394004940986633, "kl": 0.09590468532405794, "learning_rate": 4.727177487427916e-06, "loss": 0.0004795065033249557, "num_tokens": 78261412.0, "reward": 2.295654296875, "reward_std": 0.49938997626304626, "rewards/code_complexity_reward/mean": 0.8843750357627869, "rewards/code_complexity_reward/std": 0.09906608611345291, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 414, "step_time": 71.83795385435224 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 507.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 150.37109375, "completions/mean_terminated_length": 150.37109375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22263735975138843, "epoch": 0.23646723646723647, "frac_reward_zero_std": 0.125, "grad_norm": 0.06241513043642044, "kl": 0.10389766737353057, "learning_rate": 4.724913599181204e-06, "loss": 0.0005198987200856209, "num_tokens": 78406730.0, "reward": 2.3270020484924316, "reward_std": 0.5516517162322998, "rewards/code_complexity_reward/mean": 0.857128918170929, "rewards/code_complexity_reward/std": 0.13571205735206604, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 415, "step_time": 51.926522457040846 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 138.875, "completions/mean_terminated_length": 138.14480590820312, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2209162760991603, "epoch": 0.23703703703703705, "frac_reward_zero_std": 0.265625, "grad_norm": 0.05465070158243179, "kl": 0.10788019426399842, "learning_rate": 4.722640903518842e-06, "loss": 0.000539567437954247, "num_tokens": 78544602.0, "reward": 2.3848633766174316, "reward_std": 0.5561284422874451, "rewards/code_complexity_reward/mean": 0.877636730670929, "rewards/code_complexity_reward/std": 0.1366458684206009, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 416, "step_time": 55.48052672762424 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 144.609375, "completions/mean_terminated_length": 143.89041137695312, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.21956589072942734, "epoch": 0.2376068376068376, "frac_reward_zero_std": 0.1875, "grad_norm": 0.0524759441614151, "kl": 0.09602105047088116, "learning_rate": 4.7203594094373885e-06, "loss": 0.00047993205953389406, "num_tokens": 78687138.0, "reward": 2.3411622047424316, "reward_std": 0.512472927570343, "rewards/code_complexity_reward/mean": 0.8830077648162842, "rewards/code_complexity_reward/std": 0.1011883094906807, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 417, "step_time": 48.62041681353003 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 142.484375, "completions/mean_terminated_length": 142.484375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.22818947536870837, "epoch": 0.23817663817663817, "frac_reward_zero_std": 0.21875, "grad_norm": 0.05485553666949272, "kl": 0.1001977218547836, "learning_rate": 4.7180691259682394e-06, "loss": 0.0005011949688196182, "num_tokens": 78827258.0, "reward": 2.3222169876098633, "reward_std": 0.5239899158477783, "rewards/code_complexity_reward/mean": 0.87353515625, "rewards/code_complexity_reward/std": 0.12134198099374771, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 418, "step_time": 51.59447749797255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 419.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 139.017578125, "completions/mean_terminated_length": 139.017578125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23110702028498054, "epoch": 0.23874643874643875, "frac_reward_zero_std": 0.265625, "grad_norm": 0.057774610817432404, "kl": 0.0981419743038714, "learning_rate": 4.7157700621775795e-06, "loss": 0.0004907081602141261, "num_tokens": 78969315.0, "reward": 2.2372071743011475, "reward_std": 0.47986605763435364, "rewards/code_complexity_reward/mean": 0.88818359375, "rewards/code_complexity_reward/std": 0.12079328298568726, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 419, "step_time": 54.3717844709754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 483.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 136.32421875, "completions/mean_terminated_length": 136.32421875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22602722560986876, "epoch": 0.23931623931623933, "frac_reward_zero_std": 0.296875, "grad_norm": 0.058479536324739456, "kl": 0.10021981556201354, "learning_rate": 4.71346222716635e-06, "loss": 0.0005012318724766374, "num_tokens": 79107345.0, "reward": 2.3106446266174316, "reward_std": 0.49631616473197937, "rewards/code_complexity_reward/mean": 0.897167980670929, "rewards/code_complexity_reward/std": 0.08988352119922638, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 420, "step_time": 63.678219897672534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 147.78515625, "completions/mean_terminated_length": 147.07240295410156, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2249027667567134, "epoch": 0.23988603988603988, "frac_reward_zero_std": 0.21875, "grad_norm": 0.05543098971247673, "kl": 0.10320048278663307, "learning_rate": 4.711145630070214e-06, "loss": 0.0005161979352124035, "num_tokens": 79250003.0, "reward": 2.282275438308716, "reward_std": 0.5369922518730164, "rewards/code_complexity_reward/mean": 0.8768554329872131, "rewards/code_complexity_reward/std": 0.14237459003925323, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 421, "step_time": 75.31759471446276 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 481.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 145.48828125, "completions/mean_terminated_length": 145.48828125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22710414952598512, "epoch": 0.24045584045584045, "frac_reward_zero_std": 0.21875, "grad_norm": 0.057551898062229156, "kl": 0.09724232135340571, "learning_rate": 4.7088202800595214e-06, "loss": 0.00048648216761648655, "num_tokens": 79392749.0, "reward": 2.315624952316284, "reward_std": 0.4906064569950104, "rewards/code_complexity_reward/mean": 0.887499988079071, "rewards/code_complexity_reward/std": 0.09194569289684296, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 422, "step_time": 54.567393156699836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 133.130859375, "completions/mean_terminated_length": 132.38943481445312, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2318042942788452, "epoch": 0.24102564102564103, "frac_reward_zero_std": 0.265625, "grad_norm": 0.060653626918792725, "kl": 0.10651173302903771, "learning_rate": 4.70648618633927e-06, "loss": 0.0005322953220456839, "num_tokens": 79528624.0, "reward": 2.2544922828674316, "reward_std": 0.5125068426132202, "rewards/code_complexity_reward/mean": 0.8827148079872131, "rewards/code_complexity_reward/std": 0.1404809057712555, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 423, "step_time": 59.37510277889669 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 133.26953125, "completions/mean_terminated_length": 132.52838134765625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24155122158117592, "epoch": 0.2415954415954416, "frac_reward_zero_std": 0.234375, "grad_norm": 0.06365375965833664, "kl": 0.11030927969841287, "learning_rate": 4.704143358149067e-06, "loss": 0.0005516852252185345, "num_tokens": 79669058.0, "reward": 2.2692384719848633, "reward_std": 0.5164311528205872, "rewards/code_complexity_reward/mean": 0.885058581829071, "rewards/code_complexity_reward/std": 0.1342019885778427, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 424, "step_time": 52.006918606348336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 366.0, "completions/max_terminated_length": 366.0, "completions/mean_length": 139.263671875, "completions/mean_terminated_length": 139.263671875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.21898980368860066, "epoch": 0.24216524216524216, "frac_reward_zero_std": 0.171875, "grad_norm": 0.06287528574466705, "kl": 0.10494987247511744, "learning_rate": 4.701791804763102e-06, "loss": 0.0005246583605185151, "num_tokens": 79807809.0, "reward": 2.3263673782348633, "reward_std": 0.5335680246353149, "rewards/code_complexity_reward/mean": 0.8806641101837158, "rewards/code_complexity_reward/std": 0.13159099221229553, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 425, "step_time": 48.42579513788223 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 145.44921875, "completions/mean_terminated_length": 145.44921875, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.23414009460248053, "epoch": 0.24273504273504273, "frac_reward_zero_std": 0.21875, "grad_norm": 0.08115114271640778, "kl": 0.10673984565073624, "learning_rate": 4.699431535490097e-06, "loss": 0.0005337934708222747, "num_tokens": 79953119.0, "reward": 2.2152345180511475, "reward_std": 0.46671223640441895, "rewards/code_complexity_reward/mean": 0.884765625, "rewards/code_complexity_reward/std": 0.11283719539642334, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 426, "step_time": 49.95651541650295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 139.271484375, "completions/mean_terminated_length": 139.271484375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22382417088374496, "epoch": 0.2433048433048433, "frac_reward_zero_std": 0.25, "grad_norm": 0.06967084109783173, "kl": 0.11063602718058974, "learning_rate": 4.697062559673279e-06, "loss": 0.0005533998482860625, "num_tokens": 80093818.0, "reward": 2.338135004043579, "reward_std": 0.5299375057220459, "rewards/code_complexity_reward/mean": 0.8850586414337158, "rewards/code_complexity_reward/std": 0.12424486130475998, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 427, "step_time": 47.36469525657594 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 483.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 140.44921875, "completions/mean_terminated_length": 140.44921875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23621180979534984, "epoch": 0.2438746438746439, "frac_reward_zero_std": 0.171875, "grad_norm": 0.06935219466686249, "kl": 0.0986549936933443, "learning_rate": 4.694684886690341e-06, "loss": 0.000493349798489362, "num_tokens": 80236128.0, "reward": 2.2764649391174316, "reward_std": 0.481078177690506, "rewards/code_complexity_reward/mean": 0.8786132335662842, "rewards/code_complexity_reward/std": 0.09604013711214066, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 428, "step_time": 46.5137081425637 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 442.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 139.212890625, "completions/mean_terminated_length": 139.212890625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.22477357531897724, "epoch": 0.24444444444444444, "frac_reward_zero_std": 0.21875, "grad_norm": 0.0600622221827507, "kl": 0.10857976239640266, "learning_rate": 4.692298525953402e-06, "loss": 0.0005428571021184325, "num_tokens": 80373829.0, "reward": 2.300097703933716, "reward_std": 0.5280349850654602, "rewards/code_complexity_reward/mean": 0.8753905892372131, "rewards/code_complexity_reward/std": 0.13300620019435883, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.029158055782318115, "step": 429, "step_time": 53.64259947743267 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 141.103515625, "completions/mean_terminated_length": 141.103515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22561671561561525, "epoch": 0.245014245014245, "frac_reward_zero_std": 0.1875, "grad_norm": 0.060978252440690994, "kl": 0.11361921369098127, "learning_rate": 4.689903486908975e-06, "loss": 0.0005679384921677411, "num_tokens": 80512426.0, "reward": 2.3213868141174316, "reward_std": 0.5298216342926025, "rewards/code_complexity_reward/mean": 0.8698241710662842, "rewards/code_complexity_reward/std": 0.13821113109588623, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 430, "step_time": 111.13550876546651 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 146.10546875, "completions/mean_terminated_length": 145.38943481445312, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2240348970517516, "epoch": 0.2455840455840456, "frac_reward_zero_std": 0.296875, "grad_norm": 0.04626726359128952, "kl": 0.0949850354809314, "learning_rate": 4.687499779037921e-06, "loss": 0.00047493085730820894, "num_tokens": 80655064.0, "reward": 2.3743653297424316, "reward_std": 0.5253164172172546, "rewards/code_complexity_reward/mean": 0.882031261920929, "rewards/code_complexity_reward/std": 0.0973191186785698, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 431, "step_time": 55.61028370261192 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 139.41015625, "completions/mean_terminated_length": 138.68101501464844, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23363318806514144, "epoch": 0.24615384615384617, "frac_reward_zero_std": 0.234375, "grad_norm": 0.05927174910902977, "kl": 0.1061317806597799, "learning_rate": 4.685087411855426e-06, "loss": 0.0005306899547576904, "num_tokens": 80794458.0, "reward": 2.306689500808716, "reward_std": 0.5085765719413757, "rewards/code_complexity_reward/mean": 0.8788086175918579, "rewards/code_complexity_reward/std": 0.11306590586900711, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 432, "step_time": 58.08565575256944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 379.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 143.1328125, "completions/mean_terminated_length": 143.1328125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.22770019643940032, "epoch": 0.24672364672364672, "frac_reward_zero_std": 0.21875, "grad_norm": 0.05942457541823387, "kl": 0.10290005232673138, "learning_rate": 4.682666394910943e-06, "loss": 0.0005146568291820586, "num_tokens": 80939678.0, "reward": 2.3211426734924316, "reward_std": 0.4998958110809326, "rewards/code_complexity_reward/mean": 0.8854491710662842, "rewards/code_complexity_reward/std": 0.09930380433797836, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 433, "step_time": 49.362389077432454 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 139.56640625, "completions/mean_terminated_length": 136.6338653564453, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.229784571332857, "epoch": 0.2472934472934473, "frac_reward_zero_std": 0.28125, "grad_norm": 0.09329322725534439, "kl": 0.18269786494784057, "learning_rate": 4.6802367377881754e-06, "loss": 0.0009125444339588284, "num_tokens": 81082232.0, "reward": 2.2348146438598633, "reward_std": 0.5464752316474915, "rewards/code_complexity_reward/mean": 0.866503894329071, "rewards/code_complexity_reward/std": 0.17552292346954346, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 434, "step_time": 59.257620316930115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 477.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 133.84375, "completions/mean_terminated_length": 133.84375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22715220390819013, "epoch": 0.24786324786324787, "frac_reward_zero_std": 0.234375, "grad_norm": 0.05662503466010094, "kl": 0.12848471850156784, "learning_rate": 4.6777984501050215e-06, "loss": 0.0006424580933526158, "num_tokens": 81220928.0, "reward": 2.2428712844848633, "reward_std": 0.49672403931617737, "rewards/code_complexity_reward/mean": 0.8782227039337158, "rewards/code_complexity_reward/std": 0.1377032846212387, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 435, "step_time": 48.18227670621127 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 128.48046875, "completions/mean_terminated_length": 128.48046875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22071651765145361, "epoch": 0.24843304843304842, "frac_reward_zero_std": 0.25, "grad_norm": 0.06259174644947052, "kl": 0.11831236770376563, "learning_rate": 4.67535154151355e-06, "loss": 0.0005919005488976836, "num_tokens": 81353894.0, "reward": 2.3453125953674316, "reward_std": 0.523972749710083, "rewards/code_complexity_reward/mean": 0.8966796398162842, "rewards/code_complexity_reward/std": 0.12186864018440247, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 436, "step_time": 65.91456545051187 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 136.556640625, "completions/mean_terminated_length": 135.82191467285156, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23105978150852025, "epoch": 0.249002849002849, "frac_reward_zero_std": 0.25, "grad_norm": 0.05556010454893112, "kl": 0.11574989149812609, "learning_rate": 4.672896021699951e-06, "loss": 0.0005787305417470634, "num_tokens": 81493595.0, "reward": 2.324267864227295, "reward_std": 0.5090503692626953, "rewards/code_complexity_reward/mean": 0.8924804925918579, "rewards/code_complexity_reward/std": 0.1061563789844513, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 437, "step_time": 58.04619440808892 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 490.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 141.580078125, "completions/mean_terminated_length": 141.580078125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22411458054557443, "epoch": 0.24957264957264957, "frac_reward_zero_std": 0.171875, "grad_norm": 0.05904000997543335, "kl": 0.10270321508869529, "learning_rate": 4.670431900384507e-06, "loss": 0.0005135225364938378, "num_tokens": 81637212.0, "reward": 2.248486280441284, "reward_std": 0.5146474242210388, "rewards/code_complexity_reward/mean": 0.866992175579071, "rewards/code_complexity_reward/std": 0.13470390439033508, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 438, "step_time": 74.85706945788115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 420.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 128.798828125, "completions/mean_terminated_length": 128.798828125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24333108612336218, "epoch": 0.2501424501424501, "frac_reward_zero_std": 0.171875, "grad_norm": 0.06444047391414642, "kl": 0.1155524974456057, "learning_rate": 4.667959187321545e-06, "loss": 0.0005778888007625937, "num_tokens": 81772781.0, "reward": 2.3372559547424316, "reward_std": 0.513968288898468, "rewards/code_complexity_reward/mean": 0.8957030773162842, "rewards/code_complexity_reward/std": 0.10845423489809036, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 439, "step_time": 50.110731456428766 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 463.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 141.958984375, "completions/mean_terminated_length": 141.958984375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23142813006415963, "epoch": 0.25071225071225073, "frac_reward_zero_std": 0.234375, "grad_norm": 0.07104143500328064, "kl": 0.10315441398415715, "learning_rate": 4.665477892299408e-06, "loss": 0.0005158439162187278, "num_tokens": 81913424.0, "reward": 2.2884278297424316, "reward_std": 0.5167584419250488, "rewards/code_complexity_reward/mean": 0.8773437738418579, "rewards/code_complexity_reward/std": 0.13481295108795166, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 440, "step_time": 112.98949063010514 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 133.599609375, "completions/mean_terminated_length": 132.11569213867188, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2304819985292852, "epoch": 0.2512820512820513, "frac_reward_zero_std": 0.171875, "grad_norm": 0.0580725222826004, "kl": 0.11377331358380616, "learning_rate": 4.662988025140407e-06, "loss": 0.0005687740631401539, "num_tokens": 82049387.0, "reward": 2.2809572219848633, "reward_std": 0.5120143294334412, "rewards/code_complexity_reward/mean": 0.88232421875, "rewards/code_complexity_reward/std": 0.128574401140213, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 441, "step_time": 49.38840660918504 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 454.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 135.154296875, "completions/mean_terminated_length": 135.154296875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23104103468358517, "epoch": 0.2518518518518518, "frac_reward_zero_std": 0.203125, "grad_norm": 0.06918158382177353, "kl": 0.12407486431766301, "learning_rate": 4.660489595700788e-06, "loss": 0.0006202106596902013, "num_tokens": 82184906.0, "reward": 2.31982421875, "reward_std": 0.5071818828582764, "rewards/code_complexity_reward/mean": 0.8858398199081421, "rewards/code_complexity_reward/std": 0.11377561092376709, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 442, "step_time": 45.317901426926255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 416.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 138.337890625, "completions/mean_terminated_length": 138.337890625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23050527554005384, "epoch": 0.25242165242165243, "frac_reward_zero_std": 0.140625, "grad_norm": 0.06984768807888031, "kl": 0.12437799258623272, "learning_rate": 4.657982613870691e-06, "loss": 0.0006217072950676084, "num_tokens": 82325439.0, "reward": 2.288867235183716, "reward_std": 0.5331873893737793, "rewards/code_complexity_reward/mean": 0.8744140863418579, "rewards/code_complexity_reward/std": 0.149099200963974, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 443, "step_time": 52.71964042261243 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 400.0, "completions/max_terminated_length": 400.0, "completions/mean_length": 128.232421875, "completions/mean_terminated_length": 128.232421875, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.23506158427335322, "epoch": 0.252991452991453, "frac_reward_zero_std": 0.25, "grad_norm": 0.06443700194358826, "kl": 0.1279867998091504, "learning_rate": 4.655467089574111e-06, "loss": 0.0006399613921530545, "num_tokens": 82459534.0, "reward": 2.3412599563598633, "reward_std": 0.5040472745895386, "rewards/code_complexity_reward/mean": 0.9003905653953552, "rewards/code_complexity_reward/std": 0.09461965411901474, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 444, "step_time": 50.31448798440397 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 359.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 131.0234375, "completions/mean_terminated_length": 131.0234375, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.23249834030866623, "epoch": 0.2535612535612536, "frac_reward_zero_std": 0.234375, "grad_norm": 0.07172799110412598, "kl": 0.11747592512983829, "learning_rate": 4.652943032768857e-06, "loss": 0.0005871765315532684, "num_tokens": 82593802.0, "reward": 2.3033204078674316, "reward_std": 0.48543232679367065, "rewards/code_complexity_reward/mean": 0.901562511920929, "rewards/code_complexity_reward/std": 0.1015414372086525, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 445, "step_time": 40.64056274201721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 397.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 128.74609375, "completions/mean_terminated_length": 128.74609375, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.22869372414425015, "epoch": 0.25413105413105413, "frac_reward_zero_std": 0.3125, "grad_norm": 0.07970990240573883, "kl": 0.11908794392365962, "learning_rate": 4.650410453446519e-06, "loss": 0.0005957202520221472, "num_tokens": 82731472.0, "reward": 2.267627239227295, "reward_std": 0.4878745973110199, "rewards/code_complexity_reward/mean": 0.8890624642372131, "rewards/code_complexity_reward/std": 0.1099167913198471, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 446, "step_time": 41.7183015011251 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 144.5234375, "completions/mean_terminated_length": 143.80430603027344, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2227871399372816, "epoch": 0.2547008547008547, "frac_reward_zero_std": 0.234375, "grad_norm": 0.060294799506664276, "kl": 0.12008353730197996, "learning_rate": 4.647869361632417e-06, "loss": 0.000600283732637763, "num_tokens": 82874012.0, "reward": 2.281982421875, "reward_std": 0.5129715800285339, "rewards/code_complexity_reward/mean": 0.8746093511581421, "rewards/code_complexity_reward/std": 0.13512276113033295, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 447, "step_time": 48.7774148741737 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 129.302734375, "completions/mean_terminated_length": 129.302734375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23033503140322864, "epoch": 0.2552706552706553, "frac_reward_zero_std": 0.3125, "grad_norm": 0.059538327157497406, "kl": 0.12063561019022018, "learning_rate": 4.645319767385573e-06, "loss": 0.0006032423116266727, "num_tokens": 83007663.0, "reward": 2.2791993618011475, "reward_std": 0.4675915539264679, "rewards/code_complexity_reward/mean": 0.8984375, "rewards/code_complexity_reward/std": 0.09171926975250244, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 448, "step_time": 66.69999111257493 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 148.28515625, "completions/mean_terminated_length": 148.28515625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22111745574511588, "epoch": 0.25584045584045584, "frac_reward_zero_std": 0.15625, "grad_norm": 0.060818862169981, "kl": 0.11088821128942072, "learning_rate": 4.642761680798666e-06, "loss": 0.0005543787265196443, "num_tokens": 83150233.0, "reward": 2.321582317352295, "reward_std": 0.531684935092926, "rewards/code_complexity_reward/mean": 0.8749023675918579, "rewards/code_complexity_reward/std": 0.1269080936908722, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 449, "step_time": 49.786273340694606 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 128.333984375, "completions/mean_terminated_length": 126.82942199707031, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23183574341237545, "epoch": 0.2564102564102564, "frac_reward_zero_std": 0.203125, "grad_norm": 0.05854504182934761, "kl": 0.11875278397928923, "learning_rate": 4.64019511199799e-06, "loss": 0.0005936100496910512, "num_tokens": 83282356.0, "reward": 2.31884765625, "reward_std": 0.5211465954780579, "rewards/code_complexity_reward/mean": 0.88720703125, "rewards/code_complexity_reward/std": 0.10830219089984894, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.025862684473395348, "step": 450, "step_time": 56.760582524351776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 128.146484375, "completions/mean_terminated_length": 127.39530181884766, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.22808955959044397, "epoch": 0.256980056980057, "frac_reward_zero_std": 0.1875, "grad_norm": 0.05679795891046524, "kl": 0.12262952094897628, "learning_rate": 4.637620071143418e-06, "loss": 0.0006130142137408257, "num_tokens": 83413127.0, "reward": 2.364551067352295, "reward_std": 0.5236114859580994, "rewards/code_complexity_reward/mean": 0.9017578363418579, "rewards/code_complexity_reward/std": 0.11639915406703949, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 451, "step_time": 48.35464141797274 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 128.55078125, "completions/mean_terminated_length": 127.8003921508789, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23111938312649727, "epoch": 0.25754985754985754, "frac_reward_zero_std": 0.28125, "grad_norm": 0.0646485909819603, "kl": 0.12523296975996345, "learning_rate": 4.635036568428358e-06, "loss": 0.0006259871297515929, "num_tokens": 83548361.0, "reward": 2.242480516433716, "reward_std": 0.4550968408584595, "rewards/code_complexity_reward/mean": 0.9052734375, "rewards/code_complexity_reward/std": 0.07939054816961288, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 452, "step_time": 68.68334537371993 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 125.486328125, "completions/mean_terminated_length": 124.72994232177734, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23423432116396725, "epoch": 0.25811965811965815, "frac_reward_zero_std": 0.25, "grad_norm": 0.062110163271427155, "kl": 0.12485947145614773, "learning_rate": 4.632444614079718e-06, "loss": 0.0006242475938051939, "num_tokens": 83683626.0, "reward": 2.2589845657348633, "reward_std": 0.5090951919555664, "rewards/code_complexity_reward/mean": 0.8840819597244263, "rewards/code_complexity_reward/std": 0.13184501230716705, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 453, "step_time": 71.49214257858694 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 492.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 133.583984375, "completions/mean_terminated_length": 133.583984375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2263591610826552, "epoch": 0.2586894586894587, "frac_reward_zero_std": 0.21875, "grad_norm": 0.05969427153468132, "kl": 0.12800349306780845, "learning_rate": 4.629844218357859e-06, "loss": 0.0006399884005077183, "num_tokens": 83818509.0, "reward": 2.3451170921325684, "reward_std": 0.5351336002349854, "rewards/code_complexity_reward/mean": 0.88818359375, "rewards/code_complexity_reward/std": 0.1287909299135208, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 454, "step_time": 47.331105067394674 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 137.94140625, "completions/mean_terminated_length": 137.94140625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2237241673283279, "epoch": 0.25925925925925924, "frac_reward_zero_std": 0.25, "grad_norm": 0.05522114038467407, "kl": 0.11876331840176135, "learning_rate": 4.627235391556559e-06, "loss": 0.0005936659290455282, "num_tokens": 83956879.0, "reward": 2.2982423305511475, "reward_std": 0.5233462452888489, "rewards/code_complexity_reward/mean": 0.8779296875, "rewards/code_complexity_reward/std": 0.11911780387163162, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 455, "step_time": 53.744420900940895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 506.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 136.814453125, "completions/mean_terminated_length": 136.814453125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23461113451048732, "epoch": 0.25982905982905985, "frac_reward_zero_std": 0.265625, "grad_norm": 0.06844348460435867, "kl": 0.11642711400054395, "learning_rate": 4.62461814400297e-06, "loss": 0.000581983127631247, "num_tokens": 84097192.0, "reward": 2.2721192836761475, "reward_std": 0.5059772729873657, "rewards/code_complexity_reward/mean": 0.88818359375, "rewards/code_complexity_reward/std": 0.12284152209758759, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 456, "step_time": 59.46102934796363 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 129.669921875, "completions/mean_terminated_length": 129.669921875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22923592780716717, "epoch": 0.2603988603988604, "frac_reward_zero_std": 0.3125, "grad_norm": 0.05556744709610939, "kl": 0.11894171440508217, "learning_rate": 4.621992486057578e-06, "loss": 0.0005947979516349733, "num_tokens": 84232679.0, "reward": 2.3209962844848633, "reward_std": 0.5112947821617126, "rewards/code_complexity_reward/mean": 0.899707019329071, "rewards/code_complexity_reward/std": 0.10919072479009628, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 457, "step_time": 44.27096753567457 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 420.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 133.3828125, "completions/mean_terminated_length": 133.3828125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23318831203505397, "epoch": 0.26096866096866095, "frac_reward_zero_std": 0.3125, "grad_norm": 0.06288889050483704, "kl": 0.12215739290695637, "learning_rate": 4.6193584281141645e-06, "loss": 0.0006107388762757182, "num_tokens": 84371651.0, "reward": 2.353564739227295, "reward_std": 0.5198724269866943, "rewards/code_complexity_reward/mean": 0.8910156488418579, "rewards/code_complexity_reward/std": 0.11538814753293991, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.025282343849539757, "step": 458, "step_time": 85.42614769749343 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 120.490234375, "completions/mean_terminated_length": 120.490234375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2232190789654851, "epoch": 0.26153846153846155, "frac_reward_zero_std": 0.375, "grad_norm": 0.05694479122757912, "kl": 0.12527544447220862, "learning_rate": 4.616715980599759e-06, "loss": 0.0006264598923735321, "num_tokens": 84502638.0, "reward": 2.4041991233825684, "reward_std": 0.544036328792572, "rewards/code_complexity_reward/mean": 0.89794921875, "rewards/code_complexity_reward/std": 0.12187660485506058, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 459, "step_time": 47.11620935611427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 119.23046875, "completions/mean_terminated_length": 118.46183776855469, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23859772575087845, "epoch": 0.2621082621082621, "frac_reward_zero_std": 0.328125, "grad_norm": 0.06713374704122543, "kl": 0.1328044079709798, "learning_rate": 4.614065153974602e-06, "loss": 0.000664037128444761, "num_tokens": 84632948.0, "reward": 2.355224609375, "reward_std": 0.5702407360076904, "rewards/code_complexity_reward/mean": 0.8960937261581421, "rewards/code_complexity_reward/std": 0.1607053130865097, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 460, "step_time": 50.794071392156184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 453.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 123.87890625, "completions/mean_terminated_length": 123.87890625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23838531807996333, "epoch": 0.26267806267806265, "frac_reward_zero_std": 0.328125, "grad_norm": 0.062361665070056915, "kl": 0.12683966720942408, "learning_rate": 4.611405958732106e-06, "loss": 0.0006345354486256838, "num_tokens": 84763366.0, "reward": 2.4645018577575684, "reward_std": 0.5160068869590759, "rewards/code_complexity_reward/mean": 0.90771484375, "rewards/code_complexity_reward/std": 0.06557030230760574, "rewards/code_execution_reward/mean": 0.45703125, "rewards/code_execution_reward/std": 0.49863746762275696, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 461, "step_time": 45.51734190713614 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 462.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 123.021484375, "completions/mean_terminated_length": 123.021484375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.22412985493429005, "epoch": 0.26324786324786326, "frac_reward_zero_std": 0.28125, "grad_norm": 0.06400682032108307, "kl": 0.13589996460359544, "learning_rate": 4.6087384053988076e-06, "loss": 0.0006796946981921792, "num_tokens": 84893337.0, "reward": 2.3421876430511475, "reward_std": 0.5162975192070007, "rewards/code_complexity_reward/mean": 0.90283203125, "rewards/code_complexity_reward/std": 0.11384915560483932, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 462, "step_time": 45.92490171268582 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 123.48046875, "completions/mean_terminated_length": 123.48046875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2331359211821109, "epoch": 0.2638176638176638, "frac_reward_zero_std": 0.28125, "grad_norm": 0.06406661868095398, "kl": 0.12800032913219184, "learning_rate": 4.606062504534332e-06, "loss": 0.0006399651756510139, "num_tokens": 85022919.0, "reward": 2.3421876430511475, "reward_std": 0.5028855800628662, "rewards/code_complexity_reward/mean": 0.9013671875, "rewards/code_complexity_reward/std": 0.09309888631105423, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 463, "step_time": 45.44594192598015 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 462.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 131.27734375, "completions/mean_terminated_length": 131.27734375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22786477115005255, "epoch": 0.2643874643874644, "frac_reward_zero_std": 0.140625, "grad_norm": 0.07002715766429901, "kl": 0.12715877417940646, "learning_rate": 4.603378266731347e-06, "loss": 0.0006356704980134964, "num_tokens": 85159101.0, "reward": 2.256640672683716, "reward_std": 0.4801386296749115, "rewards/code_complexity_reward/mean": 0.8958984613418579, "rewards/code_complexity_reward/std": 0.1107383668422699, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 464, "step_time": 45.36945752892643 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 122.341796875, "completions/mean_terminated_length": 122.341796875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21836265036836267, "epoch": 0.26495726495726496, "frac_reward_zero_std": 0.328125, "grad_norm": 0.060322169214487076, "kl": 0.13430766994133592, "learning_rate": 4.600685702615522e-06, "loss": 0.0006715765921398997, "num_tokens": 85289348.0, "reward": 2.4057130813598633, "reward_std": 0.5192930102348328, "rewards/code_complexity_reward/mean": 0.904589831829071, "rewards/code_complexity_reward/std": 0.09499936550855637, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 465, "step_time": 45.40675258357078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 407.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 135.033203125, "completions/mean_terminated_length": 135.033203125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22836453933268785, "epoch": 0.2655270655270655, "frac_reward_zero_std": 0.25, "grad_norm": 0.06687171757221222, "kl": 0.1656886914279312, "learning_rate": 4.597984822845488e-06, "loss": 0.0008273756830021739, "num_tokens": 85432941.0, "reward": 2.2748537063598633, "reward_std": 0.5243473052978516, "rewards/code_complexity_reward/mean": 0.8801757097244263, "rewards/code_complexity_reward/std": 0.14270441234111786, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 466, "step_time": 50.829293549992144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 505.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 118.962890625, "completions/mean_terminated_length": 118.962890625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23778114095330238, "epoch": 0.2660968660968661, "frac_reward_zero_std": 0.21875, "grad_norm": 0.07228327542543411, "kl": 0.13441120518837124, "learning_rate": 4.595275638112792e-06, "loss": 0.0006719405064359307, "num_tokens": 85564330.0, "reward": 2.3024415969848633, "reward_std": 0.48699212074279785, "rewards/code_complexity_reward/mean": 0.906542956829071, "rewards/code_complexity_reward/std": 0.0912039577960968, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 467, "step_time": 78.0337559338659 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 421.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 129.984375, "completions/mean_terminated_length": 129.984375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23186091450043023, "epoch": 0.26666666666666666, "frac_reward_zero_std": 0.28125, "grad_norm": 0.07332908362150192, "kl": 0.12049636105075479, "learning_rate": 4.592558159141859e-06, "loss": 0.0006025757757015526, "num_tokens": 85703898.0, "reward": 2.2545900344848633, "reward_std": 0.48559829592704773, "rewards/code_complexity_reward/mean": 0.890917956829071, "rewards/code_complexity_reward/std": 0.11484923958778381, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 468, "step_time": 71.11481999792159 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 126.568359375, "completions/mean_terminated_length": 124.29666137695312, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.23642562655732036, "epoch": 0.2672364672364672, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06547810137271881, "kl": 0.13610127684660256, "learning_rate": 4.5898323966899455e-06, "loss": 0.0006807397585362196, "num_tokens": 85836877.0, "reward": 2.2818360328674316, "reward_std": 0.5153511166572571, "rewards/code_complexity_reward/mean": 0.889355480670929, "rewards/code_complexity_reward/std": 0.13702519237995148, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 469, "step_time": 58.68666387908161 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 125.76953125, "completions/mean_terminated_length": 125.76953125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2318358647171408, "epoch": 0.2678062678062678, "frac_reward_zero_std": 0.359375, "grad_norm": 0.05554869771003723, "kl": 0.12833900714758784, "learning_rate": 4.5870983615470986e-06, "loss": 0.0006417976110242307, "num_tokens": 85968495.0, "reward": 2.285693407058716, "reward_std": 0.5180184841156006, "rewards/code_complexity_reward/mean": 0.8900390863418579, "rewards/code_complexity_reward/std": 0.13497261703014374, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 470, "step_time": 46.90777441859245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 426.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 130.201171875, "completions/mean_terminated_length": 130.201171875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22658627689816058, "epoch": 0.26837606837606837, "frac_reward_zero_std": 0.28125, "grad_norm": 0.058642879128456116, "kl": 0.13010272558312863, "learning_rate": 4.584356064536112e-06, "loss": 0.000650572357699275, "num_tokens": 86102198.0, "reward": 2.3163576126098633, "reward_std": 0.49128642678260803, "rewards/code_complexity_reward/mean": 0.899218738079071, "rewards/code_complexity_reward/std": 0.08929695934057236, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 471, "step_time": 46.63273444864899 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 116.69921875, "completions/mean_terminated_length": 116.69921875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2321798645425588, "epoch": 0.26894586894586897, "frac_reward_zero_std": 0.28125, "grad_norm": 0.06593184173107147, "kl": 0.1347513647051528, "learning_rate": 4.5816055165124875e-06, "loss": 0.000673608505167067, "num_tokens": 86232476.0, "reward": 2.3473143577575684, "reward_std": 0.501981258392334, "rewards/code_complexity_reward/mean": 0.917675793170929, "rewards/code_complexity_reward/std": 0.0942322313785553, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 472, "step_time": 46.73142416961491 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 314.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 123.091796875, "completions/mean_terminated_length": 123.091796875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23930921452119946, "epoch": 0.2695156695156695, "frac_reward_zero_std": 0.328125, "grad_norm": 0.07017837464809418, "kl": 0.13249407126568258, "learning_rate": 4.578846728364387e-06, "loss": 0.0006623239605687559, "num_tokens": 86364595.0, "reward": 2.2654786109924316, "reward_std": 0.4485526978969574, "rewards/code_complexity_reward/mean": 0.9095703363418579, "rewards/code_complexity_reward/std": 0.0681471973657608, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 473, "step_time": 56.398513617925346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 122.77734375, "completions/mean_terminated_length": 122.77734375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23620624281466007, "epoch": 0.27008547008547007, "frac_reward_zero_std": 0.296875, "grad_norm": 0.06197541952133179, "kl": 0.13642219617031515, "learning_rate": 4.57607971101259e-06, "loss": 0.000681632780469954, "num_tokens": 86493953.0, "reward": 2.2880859375, "reward_std": 0.49229350686073303, "rewards/code_complexity_reward/mean": 0.9019531011581421, "rewards/code_complexity_reward/std": 0.10729776322841644, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 474, "step_time": 36.261502750217915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 389.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 132.771484375, "completions/mean_terminated_length": 132.771484375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24126631463877857, "epoch": 0.2706552706552707, "frac_reward_zero_std": 0.234375, "grad_norm": 0.06473872810602188, "kl": 0.12714207789395005, "learning_rate": 4.573304475410455e-06, "loss": 0.0006358071113936603, "num_tokens": 86631684.0, "reward": 2.232715129852295, "reward_std": 0.475617915391922, "rewards/code_complexity_reward/mean": 0.8973632454872131, "rewards/code_complexity_reward/std": 0.12250596284866333, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 475, "step_time": 42.431706331670284 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 423.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 128.921875, "completions/mean_terminated_length": 128.921875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22953874920494854, "epoch": 0.2712250712250712, "frac_reward_zero_std": 0.265625, "grad_norm": 0.08748714625835419, "kl": 0.1390841763932258, "learning_rate": 4.5705210325438694e-06, "loss": 0.0006954483687877655, "num_tokens": 86766396.0, "reward": 2.1793947219848633, "reward_std": 0.4624873697757721, "rewards/code_complexity_reward/mean": 0.880175769329071, "rewards/code_complexity_reward/std": 0.14386531710624695, "rewards/code_execution_reward/mean": 0.20703125, "rewards/code_execution_reward/std": 0.40557438135147095, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 476, "step_time": 49.35815437696874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 463.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 136.326171875, "completions/mean_terminated_length": 136.326171875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2362077629659325, "epoch": 0.2717948717948718, "frac_reward_zero_std": 0.21875, "grad_norm": 0.06050344556570053, "kl": 0.12316807825118303, "learning_rate": 4.567729393431211e-06, "loss": 0.000615730183199048, "num_tokens": 86906795.0, "reward": 2.2530274391174316, "reward_std": 0.5293680429458618, "rewards/code_complexity_reward/mean": 0.8820312023162842, "rewards/code_complexity_reward/std": 0.15624436736106873, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 477, "step_time": 53.78161941282451 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 121.271484375, "completions/mean_terminated_length": 121.271484375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22530547343194485, "epoch": 0.2723646723646724, "frac_reward_zero_std": 0.296875, "grad_norm": 0.06095538288354874, "kl": 0.134990761638619, "learning_rate": 4.564929569123303e-06, "loss": 0.0006750613683834672, "num_tokens": 87034678.0, "reward": 2.3525390625, "reward_std": 0.5012834668159485, "rewards/code_complexity_reward/mean": 0.9087890386581421, "rewards/code_complexity_reward/std": 0.09990649670362473, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 478, "step_time": 37.67435675859451 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 463.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 124.451171875, "completions/mean_terminated_length": 124.451171875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22205977723933756, "epoch": 0.2729344729344729, "frac_reward_zero_std": 0.3125, "grad_norm": 0.06791117787361145, "kl": 0.1387692765565589, "learning_rate": 4.562121570703369e-06, "loss": 0.0006939460872672498, "num_tokens": 87165453.0, "reward": 2.275390625, "reward_std": 0.4853830933570862, "rewards/code_complexity_reward/mean": 0.9068359136581421, "rewards/code_complexity_reward/std": 0.11866901814937592, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 479, "step_time": 52.61596376076341 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 377.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 131.384765625, "completions/mean_terminated_length": 131.384765625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22709761001169682, "epoch": 0.27350427350427353, "frac_reward_zero_std": 0.3125, "grad_norm": 0.06239312142133713, "kl": 0.1324883826309815, "learning_rate": 4.559305409286993e-06, "loss": 0.0006623535882681608, "num_tokens": 87300298.0, "reward": 2.33154296875, "reward_std": 0.5054484605789185, "rewards/code_complexity_reward/mean": 0.898242175579071, "rewards/code_complexity_reward/std": 0.097205750644207, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 480, "step_time": 46.35068217664957 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 377.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 134.0625, "completions/mean_terminated_length": 134.0625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23209154466167092, "epoch": 0.2740740740740741, "frac_reward_zero_std": 0.234375, "grad_norm": 0.06639091670513153, "kl": 0.1304163602180779, "learning_rate": 4.556481096022067e-06, "loss": 0.0006523339543491602, "num_tokens": 87438706.0, "reward": 2.2806642055511475, "reward_std": 0.5173337459564209, "rewards/code_complexity_reward/mean": 0.8830077648162842, "rewards/code_complexity_reward/std": 0.13679422438144684, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 481, "step_time": 55.34264897275716 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 463.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 139.541015625, "completions/mean_terminated_length": 139.541015625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22678197897039354, "epoch": 0.27464387464387463, "frac_reward_zero_std": 0.203125, "grad_norm": 0.08696722984313965, "kl": 0.13020424940623343, "learning_rate": 4.553648642088759e-06, "loss": 0.0006512682884931564, "num_tokens": 87579743.0, "reward": 2.285888910293579, "reward_std": 0.5098850131034851, "rewards/code_complexity_reward/mean": 0.8833984136581421, "rewards/code_complexity_reward/std": 0.1270921677350998, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 482, "step_time": 54.716035888530314 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 128.42578125, "completions/mean_terminated_length": 128.42578125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23656411631964147, "epoch": 0.27521367521367524, "frac_reward_zero_std": 0.265625, "grad_norm": 0.06243158131837845, "kl": 0.13333281816449016, "learning_rate": 4.550808058699458e-06, "loss": 0.0006664065294899046, "num_tokens": 87714881.0, "reward": 2.2775392532348633, "reward_std": 0.4818865954875946, "rewards/code_complexity_reward/mean": 0.8911132216453552, "rewards/code_complexity_reward/std": 0.11136113107204437, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 483, "step_time": 50.83459915779531 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 121.20703125, "completions/mean_terminated_length": 121.20703125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23653333680704236, "epoch": 0.2757834757834758, "frac_reward_zero_std": 0.3125, "grad_norm": 0.06253166496753693, "kl": 0.1431149272248149, "learning_rate": 4.547959357098733e-06, "loss": 0.0007154073682613671, "num_tokens": 87849371.0, "reward": 2.2762207984924316, "reward_std": 0.5076442956924438, "rewards/code_complexity_reward/mean": 0.8919922113418579, "rewards/code_complexity_reward/std": 0.13068513572216034, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 484, "step_time": 69.50646604411304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 473.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 123.9921875, "completions/mean_terminated_length": 123.9921875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.23234407254494727, "epoch": 0.27635327635327633, "frac_reward_zero_std": 0.359375, "grad_norm": 0.05877547711133957, "kl": 0.16710898792371154, "learning_rate": 4.545102548563294e-06, "loss": 0.0008356470498256385, "num_tokens": 87982839.0, "reward": 2.243457317352295, "reward_std": 0.47291481494903564, "rewards/code_complexity_reward/mean": 0.9056640267372131, "rewards/code_complexity_reward/std": 0.11732149124145508, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 485, "step_time": 54.693020707927644 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 441.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 119.052734375, "completions/mean_terminated_length": 119.052734375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23940442921593785, "epoch": 0.27692307692307694, "frac_reward_zero_std": 0.3125, "grad_norm": 0.06506418436765671, "kl": 0.13923190243076533, "learning_rate": 4.5422376444019375e-06, "loss": 0.0006963605992496014, "num_tokens": 88114906.0, "reward": 2.335253953933716, "reward_std": 0.5205017924308777, "rewards/code_complexity_reward/mean": 0.9090819954872131, "rewards/code_complexity_reward/std": 0.12231674790382385, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 486, "step_time": 51.08251268975437 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 127.88671875, "completions/mean_terminated_length": 127.88671875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22518274909816682, "epoch": 0.2774928774928775, "frac_reward_zero_std": 0.296875, "grad_norm": 0.059725321829319, "kl": 0.14405507605988532, "learning_rate": 4.539364655955511e-06, "loss": 0.0007198095554485917, "num_tokens": 88248768.0, "reward": 2.288379192352295, "reward_std": 0.48415878415107727, "rewards/code_complexity_reward/mean": 0.9100586175918579, "rewards/code_complexity_reward/std": 0.09753014147281647, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 487, "step_time": 67.06768747977912 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 125.66015625, "completions/mean_terminated_length": 124.90410614013672, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2364388071000576, "epoch": 0.27806267806267804, "frac_reward_zero_std": 0.1875, "grad_norm": 0.07084959745407104, "kl": 0.13438485097140074, "learning_rate": 4.536483594596861e-06, "loss": 0.000671890564262867, "num_tokens": 88382122.0, "reward": 2.3021974563598633, "reward_std": 0.5488747954368591, "rewards/code_complexity_reward/mean": 0.886035144329071, "rewards/code_complexity_reward/std": 0.15284420549869537, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 488, "step_time": 58.30244265682995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 388.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 129.962890625, "completions/mean_terminated_length": 129.962890625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22644004272297025, "epoch": 0.27863247863247864, "frac_reward_zero_std": 0.34375, "grad_norm": 0.05811511352658272, "kl": 0.14769906329456717, "learning_rate": 4.533594471730793e-06, "loss": 0.0007384378695860505, "num_tokens": 88515679.0, "reward": 2.31640625, "reward_std": 0.5075211524963379, "rewards/code_complexity_reward/mean": 0.8951171636581421, "rewards/code_complexity_reward/std": 0.10485094785690308, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 489, "step_time": 72.48512971960008 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 454.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 123.23046875, "completions/mean_terminated_length": 123.23046875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22739246557466686, "epoch": 0.2792022792022792, "frac_reward_zero_std": 0.25, "grad_norm": 0.0647667720913887, "kl": 0.13805393932852894, "learning_rate": 4.530697298794022e-06, "loss": 0.0006901668384671211, "num_tokens": 88650541.0, "reward": 2.3257813453674316, "reward_std": 0.5235542058944702, "rewards/code_complexity_reward/mean": 0.9025390148162842, "rewards/code_complexity_reward/std": 0.12240814417600632, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 490, "step_time": 44.61355273518711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 115.408203125, "completions/mean_terminated_length": 114.63209533691406, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2320376387797296, "epoch": 0.2797720797720798, "frac_reward_zero_std": 0.25, "grad_norm": 0.06493986397981644, "kl": 0.14075108990073204, "learning_rate": 4.527792087255133e-06, "loss": 0.0007036728784441948, "num_tokens": 88779294.0, "reward": 2.3699707984924316, "reward_std": 0.5369725227355957, "rewards/code_complexity_reward/mean": 0.9010741710662842, "rewards/code_complexity_reward/std": 0.12478481233119965, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 491, "step_time": 47.45057494007051 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 120.232421875, "completions/mean_terminated_length": 120.232421875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22701682476326823, "epoch": 0.28034188034188035, "frac_reward_zero_std": 0.21875, "grad_norm": 0.07242612540721893, "kl": 0.1388201735680923, "learning_rate": 4.524878848614529e-06, "loss": 0.0006942523177713156, "num_tokens": 88907605.0, "reward": 2.338818311691284, "reward_std": 0.5184447765350342, "rewards/code_complexity_reward/mean": 0.89794921875, "rewards/code_complexity_reward/std": 0.11531753838062286, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 492, "step_time": 69.90818912908435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 441.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 126.517578125, "completions/mean_terminated_length": 126.517578125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23032844555564225, "epoch": 0.2809116809116809, "frac_reward_zero_std": 0.25, "grad_norm": 0.06308289617300034, "kl": 0.13765139260794967, "learning_rate": 4.521957594404389e-06, "loss": 0.0006887298659421504, "num_tokens": 89043342.0, "reward": 2.285400390625, "reward_std": 0.48651382327079773, "rewards/code_complexity_reward/mean": 0.9097656011581421, "rewards/code_complexity_reward/std": 0.10436811298131943, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 493, "step_time": 51.96369194705039 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 120.423828125, "completions/mean_terminated_length": 118.88824462890625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23136902111582458, "epoch": 0.2814814814814815, "frac_reward_zero_std": 0.25, "grad_norm": 0.06525687873363495, "kl": 0.14288292196579278, "learning_rate": 4.519028336188625e-06, "loss": 0.0007140798261389136, "num_tokens": 89172919.0, "reward": 2.27490234375, "reward_std": 0.5123116970062256, "rewards/code_complexity_reward/mean": 0.8960937261581421, "rewards/code_complexity_reward/std": 0.13434036076068878, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 494, "step_time": 87.66745663620532 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 425.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 128.701171875, "completions/mean_terminated_length": 128.701171875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22596744517795742, "epoch": 0.28205128205128205, "frac_reward_zero_std": 0.375, "grad_norm": 0.063666433095932, "kl": 0.1369365000864491, "learning_rate": 4.516091085562828e-06, "loss": 0.0006847893819212914, "num_tokens": 89307374.0, "reward": 2.2748048305511475, "reward_std": 0.4713514447212219, "rewards/code_complexity_reward/mean": 0.8994140625, "rewards/code_complexity_reward/std": 0.10369705408811569, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 495, "step_time": 44.334700358100235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 112.62109375, "completions/mean_terminated_length": 112.62109375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.22751927585341036, "epoch": 0.2826210826210826, "frac_reward_zero_std": 0.421875, "grad_norm": 0.059618059545755386, "kl": 0.15937443228904158, "learning_rate": 4.5131458541542326e-06, "loss": 0.000797344371676445, "num_tokens": 89431972.0, "reward": 2.323535442352295, "reward_std": 0.4999488890171051, "rewards/code_complexity_reward/mean": 0.9149414300918579, "rewards/code_complexity_reward/std": 0.10090779513120651, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 496, "step_time": 47.872854026034474 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 127.494140625, "completions/mean_terminated_length": 127.494140625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22154513513669372, "epoch": 0.2831908831908832, "frac_reward_zero_std": 0.25, "grad_norm": 0.07714511454105377, "kl": 0.13744940178003162, "learning_rate": 4.510192653621662e-06, "loss": 0.0006872155936434865, "num_tokens": 89563433.0, "reward": 2.3133788108825684, "reward_std": 0.5356943607330322, "rewards/code_complexity_reward/mean": 0.88623046875, "rewards/code_complexity_reward/std": 0.13901683688163757, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 497, "step_time": 55.886468663811684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 320.0, "completions/max_terminated_length": 320.0, "completions/mean_length": 116.3828125, "completions/mean_terminated_length": 116.3828125, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.2285718924831599, "epoch": 0.28376068376068375, "frac_reward_zero_std": 0.328125, "grad_norm": 0.07688754051923752, "kl": 0.16581619263160974, "learning_rate": 4.507231495655488e-06, "loss": 0.0008289825636893511, "num_tokens": 89690197.0, "reward": 2.352783203125, "reward_std": 0.5185061693191528, "rewards/code_complexity_reward/mean": 0.903124988079071, "rewards/code_complexity_reward/std": 0.10626186430454254, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 498, "step_time": 75.35488211363554 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 112.533203125, "completions/mean_terminated_length": 111.75146484375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23044854612089694, "epoch": 0.28433048433048436, "frac_reward_zero_std": 0.375, "grad_norm": 0.06364596635103226, "kl": 0.1479298210470006, "learning_rate": 4.50426239197758e-06, "loss": 0.0007399230380542576, "num_tokens": 89816350.0, "reward": 2.297119379043579, "reward_std": 0.5223271250724792, "rewards/code_complexity_reward/mean": 0.9014648199081421, "rewards/code_complexity_reward/std": 0.13415241241455078, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 499, "step_time": 68.60948755871505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 121.716796875, "completions/mean_terminated_length": 120.95303344726562, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23101773345842957, "epoch": 0.2849002849002849, "frac_reward_zero_std": 0.265625, "grad_norm": 0.06763343513011932, "kl": 0.13854926463682204, "learning_rate": 4.5012853543412616e-06, "loss": 0.0006929626688361168, "num_tokens": 89947709.0, "reward": 2.268603801727295, "reward_std": 0.4882643520832062, "rewards/code_complexity_reward/mean": 0.9071289300918579, "rewards/code_complexity_reward/std": 0.1159617230296135, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 500, "step_time": 67.40322264190763 }, { "epoch": 0.2849002849002849, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.00125, "eval_completions/max_length": 176.86, "eval_completions/max_terminated_length": 175.81, "eval_completions/mean_length": 120.96125, "eval_completions/mean_terminated_length": 120.7689288330078, "eval_completions/min_length": 84.64, "eval_completions/min_terminated_length": 84.64, "eval_entropy": 0.22914338536560536, "eval_frac_reward_zero_std": 0.25, "eval_kl": 0.15098337724804878, "eval_loss": -0.0010163236875087023, "eval_num_tokens": 89947709.0, "eval_reward": 2.2965626072883607, "eval_reward_std": 0.2420611686259508, "eval_rewards/code_complexity_reward/mean": 0.9029374903440476, "eval_rewards/code_complexity_reward/std": 0.0492585277184844, "eval_rewards/code_execution_reward/mean": 0.3, "eval_rewards/code_execution_reward/std": 0.19413636416196822, "eval_rewards/code_syntax_reward/mean": 0.494375, "eval_rewards/code_syntax_reward/std": 0.015909902304410934, "eval_rewards/reasoning_present_reward_func/mean": 0.0998750015348196, "eval_rewards/reasoning_present_reward_func/std": 0.000353553406894207, "eval_rewards/xmlcount_reward_func/mean": 0.499375, "eval_rewards/xmlcount_reward_func/std": 0.001767766885459423, "eval_runtime": 819.6883, "eval_samples_per_second": 0.122, "eval_steps_per_second": 0.016, "step": 500 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 417.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 124.5625, "completions/mean_terminated_length": 124.5625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2252666843123734, "epoch": 0.28547008547008546, "frac_reward_zero_std": 0.3125, "grad_norm": 0.08283236622810364, "kl": 0.22328285651747137, "learning_rate": 4.498300394531265e-06, "loss": 0.0011173875536769629, "num_tokens": 90080197.0, "reward": 2.2904300689697266, "reward_std": 0.5092934966087341, "rewards/code_complexity_reward/mean": 0.8994140625, "rewards/code_complexity_reward/std": 0.12058097869157791, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 501, "step_time": 42.75103440042585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 484.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 124.064453125, "completions/mean_terminated_length": 124.064453125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.21430037706159055, "epoch": 0.28603988603988606, "frac_reward_zero_std": 0.28125, "grad_norm": 0.06578958034515381, "kl": 0.14847930578980595, "learning_rate": 4.49530752436368e-06, "loss": 0.0007428044336847961, "num_tokens": 90211982.0, "reward": 2.4283204078674316, "reward_std": 0.530238926410675, "rewards/code_complexity_reward/mean": 0.9005858898162842, "rewards/code_complexity_reward/std": 0.11103342473506927, "rewards/code_execution_reward/mean": 0.431640625, "rewards/code_execution_reward/std": 0.4957893490791321, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 502, "step_time": 48.28209474589676 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 417.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 127.388671875, "completions/mean_terminated_length": 127.388671875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23257483425550163, "epoch": 0.2866096866096866, "frac_reward_zero_std": 0.1875, "grad_norm": 0.07234786450862885, "kl": 0.14960340003017336, "learning_rate": 4.492306755685913e-06, "loss": 0.0007478470215573907, "num_tokens": 90346293.0, "reward": 2.3020997047424316, "reward_std": 0.5159319043159485, "rewards/code_complexity_reward/mean": 0.8952147960662842, "rewards/code_complexity_reward/std": 0.1277976632118225, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 503, "step_time": 47.802190117537975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 478.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 116.005859375, "completions/mean_terminated_length": 116.005859375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23041672352701426, "epoch": 0.28717948717948716, "frac_reward_zero_std": 0.25, "grad_norm": 0.08498305082321167, "kl": 0.15488928311970085, "learning_rate": 4.489298100376633e-06, "loss": 0.0007741822628304362, "num_tokens": 90473664.0, "reward": 2.3405275344848633, "reward_std": 0.5150125026702881, "rewards/code_complexity_reward/mean": 0.909472644329071, "rewards/code_complexity_reward/std": 0.10936176031827927, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 504, "step_time": 47.17815569974482 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 126.28515625, "completions/mean_terminated_length": 126.28515625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22615258605219424, "epoch": 0.28774928774928776, "frac_reward_zero_std": 0.28125, "grad_norm": 0.1292046308517456, "kl": 0.15697883546818048, "learning_rate": 4.486281570345732e-06, "loss": 0.0007850665133446455, "num_tokens": 90608466.0, "reward": 2.314941644668579, "reward_std": 0.4988601803779602, "rewards/code_complexity_reward/mean": 0.9063476324081421, "rewards/code_complexity_reward/std": 0.10498224198818207, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 505, "step_time": 65.34417738672346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 484.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 124.998046875, "completions/mean_terminated_length": 124.998046875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24117243173532188, "epoch": 0.2883190883190883, "frac_reward_zero_std": 0.328125, "grad_norm": 0.061331089586019516, "kl": 0.14283982291817665, "learning_rate": 4.483257177534273e-06, "loss": 0.000714137451723218, "num_tokens": 90742121.0, "reward": 2.296191453933716, "reward_std": 0.4929916262626648, "rewards/code_complexity_reward/mean": 0.9039062261581421, "rewards/code_complexity_reward/std": 0.10133344680070877, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 506, "step_time": 56.61509765870869 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 136.923828125, "completions/mean_terminated_length": 136.923828125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2305109165608883, "epoch": 0.28888888888888886, "frac_reward_zero_std": 0.15625, "grad_norm": 0.06544279307126999, "kl": 0.14065126376226544, "learning_rate": 4.480224933914444e-06, "loss": 0.0007032605353742838, "num_tokens": 90881130.0, "reward": 2.1686036586761475, "reward_std": 0.4863140285015106, "rewards/code_complexity_reward/mean": 0.877636730670929, "rewards/code_complexity_reward/std": 0.160426527261734, "rewards/code_execution_reward/mean": 0.208984375, "rewards/code_execution_reward/std": 0.40698084235191345, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.492919921875, "rewards/xmlcount_reward_func/std": 0.056758660823106766, "step": 507, "step_time": 66.74091090075672 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 376.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 125.970703125, "completions/mean_terminated_length": 125.970703125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22450733999721706, "epoch": 0.28945868945868947, "frac_reward_zero_std": 0.28125, "grad_norm": 0.06103334575891495, "kl": 0.1456145925913006, "learning_rate": 4.477184851489511e-06, "loss": 0.000728082493878901, "num_tokens": 91014115.0, "reward": 2.2571778297424316, "reward_std": 0.47636765241622925, "rewards/code_complexity_reward/mean": 0.897656261920929, "rewards/code_complexity_reward/std": 0.1236843690276146, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 508, "step_time": 48.34026720561087 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 376.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 117.10546875, "completions/mean_terminated_length": 117.10546875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21699745673686266, "epoch": 0.29002849002849, "frac_reward_zero_std": 0.390625, "grad_norm": 0.05736244469881058, "kl": 0.15996414865367115, "learning_rate": 4.474136942293771e-06, "loss": 0.000799809000454843, "num_tokens": 91144433.0, "reward": 2.3583009243011475, "reward_std": 0.5592964887619019, "rewards/code_complexity_reward/mean": 0.8984375, "rewards/code_complexity_reward/std": 0.14513471722602844, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 509, "step_time": 40.3738081837073 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 494.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 115.03515625, "completions/mean_terminated_length": 115.03515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23070245143026114, "epoch": 0.2905982905982906, "frac_reward_zero_std": 0.359375, "grad_norm": 0.05869653448462486, "kl": 0.15436650824267417, "learning_rate": 4.471081218392502e-06, "loss": 0.0007719952845945954, "num_tokens": 91271923.0, "reward": 2.3874025344848633, "reward_std": 0.5311182737350464, "rewards/code_complexity_reward/mean": 0.909472644329071, "rewards/code_complexity_reward/std": 0.1100752055644989, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 510, "step_time": 62.80863618478179 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 351.0, "completions/max_terminated_length": 351.0, "completions/mean_length": 113.3125, "completions/mean_terminated_length": 113.3125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.232912078499794, "epoch": 0.29116809116809117, "frac_reward_zero_std": 0.28125, "grad_norm": 0.06511753797531128, "kl": 0.16096101456787437, "learning_rate": 4.468017691881918e-06, "loss": 0.0008047277806326747, "num_tokens": 91398259.0, "reward": 2.3446288108825684, "reward_std": 0.5281280279159546, "rewards/code_complexity_reward/mean": 0.90380859375, "rewards/code_complexity_reward/std": 0.133301243185997, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 511, "step_time": 48.179333448410034 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 118.705078125, "completions/mean_terminated_length": 118.705078125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2328944206237793, "epoch": 0.2917378917378917, "frac_reward_zero_std": 0.3125, "grad_norm": 0.13435904681682587, "kl": 0.30293766304384917, "learning_rate": 4.464946374889121e-06, "loss": 0.0015147051308304071, "num_tokens": 91527124.0, "reward": 2.325000286102295, "reward_std": 0.47371211647987366, "rewards/code_complexity_reward/mean": 0.9154297113418579, "rewards/code_complexity_reward/std": 0.08009418845176697, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 512, "step_time": 77.8723966917023 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 116.47265625, "completions/mean_terminated_length": 116.47265625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23718419531360269, "epoch": 0.2923076923076923, "frac_reward_zero_std": 0.359375, "grad_norm": 0.06490057706832886, "kl": 0.15794319962151349, "learning_rate": 4.461867279572049e-06, "loss": 0.0007900488562881947, "num_tokens": 91653926.0, "reward": 2.2740235328674316, "reward_std": 0.49783191084861755, "rewards/code_complexity_reward/mean": 0.9035155773162842, "rewards/code_complexity_reward/std": 0.12302189320325851, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 513, "step_time": 46.7815976459533 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 368.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 121.80859375, "completions/mean_terminated_length": 121.80859375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2302235784009099, "epoch": 0.2928774928774929, "frac_reward_zero_std": 0.296875, "grad_norm": 0.06395449489355087, "kl": 0.15698404714930803, "learning_rate": 4.458780418119433e-06, "loss": 0.0007850406691431999, "num_tokens": 91784188.0, "reward": 2.3053712844848633, "reward_std": 0.505532443523407, "rewards/code_complexity_reward/mean": 0.904589831829071, "rewards/code_complexity_reward/std": 0.11387768387794495, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 514, "step_time": 39.614679091610014 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 463.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 126.84375, "completions/mean_terminated_length": 126.84375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21952164638787508, "epoch": 0.2934472934472934, "frac_reward_zero_std": 0.296875, "grad_norm": 0.09401550143957138, "kl": 0.23106962605379522, "learning_rate": 4.455685802750747e-06, "loss": 0.0011562006548047066, "num_tokens": 91917004.0, "reward": 2.274951457977295, "reward_std": 0.45992401242256165, "rewards/code_complexity_reward/mean": 0.9125000238418579, "rewards/code_complexity_reward/std": 0.08973760157823563, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 515, "step_time": 61.23562391474843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 381.0, "completions/max_terminated_length": 381.0, "completions/mean_length": 112.033203125, "completions/mean_terminated_length": 112.033203125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2331848624162376, "epoch": 0.294017094017094, "frac_reward_zero_std": 0.375, "grad_norm": 0.06526991724967957, "kl": 0.1620735563337803, "learning_rate": 4.4525834457161565e-06, "loss": 0.0008101475541479886, "num_tokens": 92042965.0, "reward": 2.2417969703674316, "reward_std": 0.4542699456214905, "rewards/code_complexity_reward/mean": 0.916210949420929, "rewards/code_complexity_reward/std": 0.09742726385593414, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 516, "step_time": 57.43051228299737 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 117.41796875, "completions/mean_terminated_length": 116.64579010009766, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2316655230242759, "epoch": 0.2945868945868946, "frac_reward_zero_std": 0.296875, "grad_norm": 0.061643362045288086, "kl": 0.16045362874865532, "learning_rate": 4.449473359296476e-06, "loss": 0.0008020418463274837, "num_tokens": 92170107.0, "reward": 2.3056154251098633, "reward_std": 0.489347368478775, "rewards/code_complexity_reward/mean": 0.913867175579071, "rewards/code_complexity_reward/std": 0.10627607256174088, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 517, "step_time": 57.46647454239428 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 379.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 116.97265625, "completions/mean_terminated_length": 116.97265625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23003393248654902, "epoch": 0.2951566951566952, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06329713016748428, "kl": 0.15327770553994924, "learning_rate": 4.446355555803115e-06, "loss": 0.0007662561256438494, "num_tokens": 92295229.0, "reward": 2.31396484375, "reward_std": 0.49569112062454224, "rewards/code_complexity_reward/mean": 0.9146484136581421, "rewards/code_complexity_reward/std": 0.10260917246341705, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 518, "step_time": 87.30097070708871 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 442.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 120.572265625, "completions/mean_terminated_length": 120.572265625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2286268298048526, "epoch": 0.29572649572649573, "frac_reward_zero_std": 0.390625, "grad_norm": 0.05691300705075264, "kl": 0.15944159612990916, "learning_rate": 4.443230047578033e-06, "loss": 0.0007970191072672606, "num_tokens": 92425034.0, "reward": 2.3438477516174316, "reward_std": 0.48555076122283936, "rewards/code_complexity_reward/mean": 0.916699230670929, "rewards/code_complexity_reward/std": 0.07015109807252884, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 519, "step_time": 52.40504542272538 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 359.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 120.169921875, "completions/mean_terminated_length": 120.169921875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22760956059210002, "epoch": 0.2962962962962963, "frac_reward_zero_std": 0.421875, "grad_norm": 0.057339731603860855, "kl": 0.1446835232200101, "learning_rate": 4.440096846993686e-06, "loss": 0.0007233917713165283, "num_tokens": 92555377.0, "reward": 2.281298875808716, "reward_std": 0.4819878339767456, "rewards/code_complexity_reward/mean": 0.9100586175918579, "rewards/code_complexity_reward/std": 0.0996146947145462, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 520, "step_time": 39.35708335507661 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 440.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 111.822265625, "completions/mean_terminated_length": 111.822265625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22871284163556993, "epoch": 0.2968660968660969, "frac_reward_zero_std": 0.421875, "grad_norm": 0.05664236471056938, "kl": 0.1653678846778348, "learning_rate": 4.436955966452985e-06, "loss": 0.0008266378426924348, "num_tokens": 92678494.0, "reward": 2.313281297683716, "reward_std": 0.4956927001476288, "rewards/code_complexity_reward/mean": 0.9166015386581421, "rewards/code_complexity_reward/std": 0.10125326365232468, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 521, "step_time": 50.19511463586241 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 123.8359375, "completions/mean_terminated_length": 123.8359375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2326124096289277, "epoch": 0.29743589743589743, "frac_reward_zero_std": 0.359375, "grad_norm": 0.05881559103727341, "kl": 0.152428082190454, "learning_rate": 4.433807418389239e-06, "loss": 0.0007622986449860036, "num_tokens": 92812170.0, "reward": 2.310058832168579, "reward_std": 0.5309145450592041, "rewards/code_complexity_reward/mean": 0.8907226324081421, "rewards/code_complexity_reward/std": 0.1471412628889084, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 522, "step_time": 63.83754973951727 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 123.35546875, "completions/mean_terminated_length": 122.59490966796875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2342834216542542, "epoch": 0.298005698005698, "frac_reward_zero_std": 0.3125, "grad_norm": 0.06398825347423553, "kl": 0.16429192817304283, "learning_rate": 4.4306512152661085e-06, "loss": 0.0008217955473810434, "num_tokens": 92942264.0, "reward": 2.263232469558716, "reward_std": 0.5193015933036804, "rewards/code_complexity_reward/mean": 0.9007812738418579, "rewards/code_complexity_reward/std": 0.14847354590892792, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 523, "step_time": 55.18034799862653 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 404.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 125.89453125, "completions/mean_terminated_length": 125.89453125, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.23186835460364819, "epoch": 0.2985754985754986, "frac_reward_zero_std": 0.328125, "grad_norm": 0.06374917179346085, "kl": 0.15141806157771498, "learning_rate": 4.427487369577561e-06, "loss": 0.00075714779086411, "num_tokens": 93075562.0, "reward": 2.2525877952575684, "reward_std": 0.45919546484947205, "rewards/code_complexity_reward/mean": 0.908398449420929, "rewards/code_complexity_reward/std": 0.09486716240644455, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 524, "step_time": 58.41833021584898 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 351.0, "completions/max_terminated_length": 351.0, "completions/mean_length": 116.865234375, "completions/mean_terminated_length": 116.865234375, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.22488990053534508, "epoch": 0.29914529914529914, "frac_reward_zero_std": 0.25, "grad_norm": 0.06690513342618942, "kl": 0.15298588399309665, "learning_rate": 4.424315893847814e-06, "loss": 0.000764820899348706, "num_tokens": 93204925.0, "reward": 2.327685832977295, "reward_std": 0.5544618964195251, "rewards/code_complexity_reward/mean": 0.9029296636581421, "rewards/code_complexity_reward/std": 0.15382026135921478, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 525, "step_time": 38.112620355561376 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 336.0, "completions/max_terminated_length": 336.0, "completions/mean_length": 112.892578125, "completions/mean_terminated_length": 112.892578125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22402333049103618, "epoch": 0.29971509971509974, "frac_reward_zero_std": 0.296875, "grad_norm": 0.06611306965351105, "kl": 0.16549389925785363, "learning_rate": 4.421136800631289e-06, "loss": 0.0008274949504993856, "num_tokens": 93329574.0, "reward": 2.4245119094848633, "reward_std": 0.5230216383934021, "rewards/code_complexity_reward/mean": 0.918261706829071, "rewards/code_complexity_reward/std": 0.09365122765302658, "rewards/code_execution_reward/mean": 0.41015625, "rewards/code_execution_reward/std": 0.49234291911125183, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 526, "step_time": 77.29980084486306 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 306.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 110.658203125, "completions/mean_terminated_length": 110.658203125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2297386876307428, "epoch": 0.3002849002849003, "frac_reward_zero_std": 0.5, "grad_norm": 0.05580823868513107, "kl": 0.16650975798256695, "learning_rate": 4.417950102512564e-06, "loss": 0.0008324443479068577, "num_tokens": 93457079.0, "reward": 2.3819825649261475, "reward_std": 0.4995240867137909, "rewards/code_complexity_reward/mean": 0.923535168170929, "rewards/code_complexity_reward/std": 0.07354459911584854, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 527, "step_time": 54.833444345742464 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 122.015625, "completions/mean_terminated_length": 122.015625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2253142788540572, "epoch": 0.30085470085470084, "frac_reward_zero_std": 0.359375, "grad_norm": 0.0636884868144989, "kl": 0.16893831407651305, "learning_rate": 4.414755812106319e-06, "loss": 0.0008446778520010412, "num_tokens": 93585775.0, "reward": 2.3024415969848633, "reward_std": 0.4764707386493683, "rewards/code_complexity_reward/mean": 0.913867175579071, "rewards/code_complexity_reward/std": 0.09073027223348618, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 528, "step_time": 45.74589005950838 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 388.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 114.572265625, "completions/mean_terminated_length": 114.572265625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23925350070931017, "epoch": 0.30142450142450145, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05580594390630722, "kl": 0.16151292819995433, "learning_rate": 4.4115539420572885e-06, "loss": 0.0008076361264102161, "num_tokens": 93713804.0, "reward": 2.3114256858825684, "reward_std": 0.48800817131996155, "rewards/code_complexity_reward/mean": 0.91259765625, "rewards/code_complexity_reward/std": 0.09863297641277313, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 529, "step_time": 59.85753961559385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 324.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 121.275390625, "completions/mean_terminated_length": 121.275390625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23094148584641516, "epoch": 0.301994301994302, "frac_reward_zero_std": 0.40625, "grad_norm": 0.0764802023768425, "kl": 0.17987029976211488, "learning_rate": 4.4083445050402135e-06, "loss": 0.0008989175548776984, "num_tokens": 93848497.0, "reward": 2.2869629859924316, "reward_std": 0.5015799403190613, "rewards/code_complexity_reward/mean": 0.9010741710662842, "rewards/code_complexity_reward/std": 0.12168829888105392, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 530, "step_time": 74.86363331973553 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 111.701171875, "completions/mean_terminated_length": 110.91780853271484, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.250134106958285, "epoch": 0.30256410256410254, "frac_reward_zero_std": 0.359375, "grad_norm": 0.06477190554141998, "kl": 0.16530025389511138, "learning_rate": 4.4051275137597874e-06, "loss": 0.0008266451768577099, "num_tokens": 93974512.0, "reward": 2.3500001430511475, "reward_std": 0.526891827583313, "rewards/code_complexity_reward/mean": 0.91015625, "rewards/code_complexity_reward/std": 0.12451224774122238, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 531, "step_time": 48.45946398656815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 119.07421875, "completions/mean_terminated_length": 119.07421875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22659856779500842, "epoch": 0.30313390313390315, "frac_reward_zero_std": 0.375, "grad_norm": 0.06069967523217201, "kl": 0.1618099615443498, "learning_rate": 4.401902980950607e-06, "loss": 0.0008091225754469633, "num_tokens": 94104614.0, "reward": 2.2884767055511475, "reward_std": 0.47227850556373596, "rewards/code_complexity_reward/mean": 0.9111328125, "rewards/code_complexity_reward/std": 0.10108552128076553, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 532, "step_time": 47.75969701446593 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 117.783203125, "completions/mean_terminated_length": 117.783203125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2363805277273059, "epoch": 0.3037037037037037, "frac_reward_zero_std": 0.4375, "grad_norm": 0.057533130049705505, "kl": 0.16613512742333114, "learning_rate": 4.398670919377124e-06, "loss": 0.0008308254182338715, "num_tokens": 94236975.0, "reward": 2.30224609375, "reward_std": 0.4767693877220154, "rewards/code_complexity_reward/mean": 0.9141601324081421, "rewards/code_complexity_reward/std": 0.09038771688938141, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 533, "step_time": 45.850162377581 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 118.9921875, "completions/mean_terminated_length": 118.9921875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22694432828575373, "epoch": 0.30427350427350425, "frac_reward_zero_std": 0.40625, "grad_norm": 0.07444920390844345, "kl": 0.15339144342578948, "learning_rate": 4.395431341833592e-06, "loss": 0.0007668950711376965, "num_tokens": 94365171.0, "reward": 2.284619092941284, "reward_std": 0.49729111790657043, "rewards/code_complexity_reward/mean": 0.904589831829071, "rewards/code_complexity_reward/std": 0.1234034076333046, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 534, "step_time": 42.82245934382081 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 470.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 114.02734375, "completions/mean_terminated_length": 114.02734375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24019651906564832, "epoch": 0.30484330484330485, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05847647786140442, "kl": 0.17190027504693717, "learning_rate": 4.392184261144017e-06, "loss": 0.0008596311090514064, "num_tokens": 94491241.0, "reward": 2.2300782203674316, "reward_std": 0.4574386477470398, "rewards/code_complexity_reward/mean": 0.9083983898162842, "rewards/code_complexity_reward/std": 0.12310311198234558, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 535, "step_time": 53.620987514033914 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 437.0, "completions/max_terminated_length": 437.0, "completions/mean_length": 120.091796875, "completions/mean_terminated_length": 120.091796875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22948594158515334, "epoch": 0.3054131054131054, "frac_reward_zero_std": 0.359375, "grad_norm": 0.07302652299404144, "kl": 0.2191583268577233, "learning_rate": 4.388929690162108e-06, "loss": 0.0010964050889015198, "num_tokens": 94622112.0, "reward": 2.319140911102295, "reward_std": 0.5037480592727661, "rewards/code_complexity_reward/mean": 0.9105468988418579, "rewards/code_complexity_reward/std": 0.11418944597244263, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 536, "step_time": 60.257759902626276 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 120.34765625, "completions/mean_terminated_length": 120.34765625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23526834230870008, "epoch": 0.305982905982906, "frac_reward_zero_std": 0.40625, "grad_norm": 0.12817080318927765, "kl": 0.30573871347587556, "learning_rate": 4.385667641771222e-06, "loss": 0.0015304939588531852, "num_tokens": 94752050.0, "reward": 2.3187499046325684, "reward_std": 0.4990108013153076, "rewards/code_complexity_reward/mean": 0.9111328125, "rewards/code_complexity_reward/std": 0.10701009631156921, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 537, "step_time": 69.03331176191568 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 397.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 114.794921875, "completions/mean_terminated_length": 114.794921875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23546040779910982, "epoch": 0.30655270655270656, "frac_reward_zero_std": 0.34375, "grad_norm": 0.06201101839542389, "kl": 0.1608390067704022, "learning_rate": 4.382398128884319e-06, "loss": 0.0008040498942136765, "num_tokens": 94879249.0, "reward": 2.2382326126098633, "reward_std": 0.4706861674785614, "rewards/code_complexity_reward/mean": 0.9070311784744263, "rewards/code_complexity_reward/std": 0.12005124986171722, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 538, "step_time": 41.008510833606124 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 488.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 121.251953125, "completions/mean_terminated_length": 121.251953125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.231809162767604, "epoch": 0.3071225071225071, "frac_reward_zero_std": 0.359375, "grad_norm": 0.06296008825302124, "kl": 0.15062116691842675, "learning_rate": 4.379121164443904e-06, "loss": 0.0007530763396061957, "num_tokens": 95010754.0, "reward": 2.3185060024261475, "reward_std": 0.5360220074653625, "rewards/code_complexity_reward/mean": 0.8942382335662842, "rewards/code_complexity_reward/std": 0.1481490582227707, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 539, "step_time": 50.05148234590888 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 113.197265625, "completions/mean_terminated_length": 112.41683197021484, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22915161564014852, "epoch": 0.3076923076923077, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06663066893815994, "kl": 0.16416439588647336, "learning_rate": 4.37583676142198e-06, "loss": 0.0008208189974538982, "num_tokens": 95137023.0, "reward": 2.3507325649261475, "reward_std": 0.5331255197525024, "rewards/code_complexity_reward/mean": 0.904296875, "rewards/code_complexity_reward/std": 0.1319398134946823, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 540, "step_time": 75.73775878176093 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 299.0, "completions/max_terminated_length": 299.0, "completions/mean_length": 114.5234375, "completions/mean_terminated_length": 114.5234375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23650244297459722, "epoch": 0.30826210826210826, "frac_reward_zero_std": 0.390625, "grad_norm": 0.05364042893052101, "kl": 0.16635476949159056, "learning_rate": 4.37254493282e-06, "loss": 0.0008319623302668333, "num_tokens": 95264003.0, "reward": 2.3661623001098633, "reward_std": 0.4907049834728241, "rewards/code_complexity_reward/mean": 0.9238280653953552, "rewards/code_complexity_reward/std": 0.06007654592394829, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 541, "step_time": 36.39391146507114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 119.462890625, "completions/mean_terminated_length": 118.69471740722656, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23186192754656076, "epoch": 0.3088319088319088, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06331723928451538, "kl": 0.19423429516609758, "learning_rate": 4.369245691668806e-06, "loss": 0.0009716841159388423, "num_tokens": 95395112.0, "reward": 2.241748094558716, "reward_std": 0.4649654030799866, "rewards/code_complexity_reward/mean": 0.9091796875, "rewards/code_complexity_reward/std": 0.10963250696659088, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 542, "step_time": 59.32012976612896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 119.013671875, "completions/mean_terminated_length": 118.24462127685547, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22789787966758013, "epoch": 0.3094017094017094, "frac_reward_zero_std": 0.359375, "grad_norm": 0.0598786361515522, "kl": 0.1534998562419787, "learning_rate": 4.365939051028585e-06, "loss": 0.000767445657402277, "num_tokens": 95523839.0, "reward": 2.2948732376098633, "reward_std": 0.5006398558616638, "rewards/code_complexity_reward/mean": 0.908007800579071, "rewards/code_complexity_reward/std": 0.11362353712320328, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 543, "step_time": 57.81974368728697 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 109.875, "completions/mean_terminated_length": 109.08805847167969, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23406341765075922, "epoch": 0.30997150997150996, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05431102216243744, "kl": 0.16822322900407016, "learning_rate": 4.362625023988816e-06, "loss": 0.0008410609443672001, "num_tokens": 95648567.0, "reward": 2.316455125808716, "reward_std": 0.5132767558097839, "rewards/code_complexity_reward/mean": 0.9100586175918579, "rewards/code_complexity_reward/std": 0.12497121095657349, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 544, "step_time": 74.43628453463316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 383.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 111.8203125, "completions/mean_terminated_length": 111.8203125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23916938668116927, "epoch": 0.31054131054131057, "frac_reward_zero_std": 0.375, "grad_norm": 0.06190738454461098, "kl": 0.1697421371936798, "learning_rate": 4.3593036236682176e-06, "loss": 0.0008488856256008148, "num_tokens": 95774643.0, "reward": 2.3401856422424316, "reward_std": 0.4957970380783081, "rewards/code_complexity_reward/mean": 0.9183593392372131, "rewards/code_complexity_reward/std": 0.09239603579044342, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 545, "step_time": 39.07807002402842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 420.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 115.939453125, "completions/mean_terminated_length": 115.939453125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23163698962889612, "epoch": 0.3111111111111111, "frac_reward_zero_std": 0.3125, "grad_norm": 0.06220908835530281, "kl": 0.17115754738915712, "learning_rate": 4.355974863214694e-06, "loss": 0.0008559600682929158, "num_tokens": 95903436.0, "reward": 2.3326172828674316, "reward_std": 0.5154658555984497, "rewards/code_complexity_reward/mean": 0.9132812023162842, "rewards/code_complexity_reward/std": 0.11687120795249939, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 546, "step_time": 42.13742655236274 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 411.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 132.720703125, "completions/mean_terminated_length": 132.720703125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22571431589312851, "epoch": 0.31168091168091167, "frac_reward_zero_std": 0.296875, "grad_norm": 0.057490020990371704, "kl": 0.1476734661264345, "learning_rate": 4.352638755805287e-06, "loss": 0.0007384284399449825, "num_tokens": 96044045.0, "reward": 2.2819337844848633, "reward_std": 0.47283419966697693, "rewards/code_complexity_reward/mean": 0.8997069597244263, "rewards/code_complexity_reward/std": 0.10272658616304398, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 547, "step_time": 42.16905076242983 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 422.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 121.439453125, "completions/mean_terminated_length": 121.439453125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23588912468403578, "epoch": 0.31225071225071227, "frac_reward_zero_std": 0.3125, "grad_norm": 0.0672227218747139, "kl": 0.16285950399469584, "learning_rate": 4.349295314646119e-06, "loss": 0.0008147989283315837, "num_tokens": 96176358.0, "reward": 2.36376953125, "reward_std": 0.5099411606788635, "rewards/code_complexity_reward/mean": 0.9141601324081421, "rewards/code_complexity_reward/std": 0.09921254217624664, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 548, "step_time": 43.129221038892865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 113.1484375, "completions/mean_terminated_length": 113.1484375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22788158687762916, "epoch": 0.3128205128205128, "frac_reward_zero_std": 0.484375, "grad_norm": 0.051959965378046036, "kl": 0.15631633717566729, "learning_rate": 4.3459445529723456e-06, "loss": 0.0007815912831574678, "num_tokens": 96301306.0, "reward": 2.254199266433716, "reward_std": 0.4610217809677124, "rewards/code_complexity_reward/mean": 0.9198242425918579, "rewards/code_complexity_reward/std": 0.09478871524333954, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 549, "step_time": 37.149210115894675 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 116.353515625, "completions/mean_terminated_length": 116.353515625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23630855069495738, "epoch": 0.31339031339031337, "frac_reward_zero_std": 0.296875, "grad_norm": 0.06709733605384827, "kl": 0.1697618308244273, "learning_rate": 4.3425864840481e-06, "loss": 0.0008489431347697973, "num_tokens": 96428255.0, "reward": 2.3004884719848633, "reward_std": 0.48201993107795715, "rewards/code_complexity_reward/mean": 0.914355456829071, "rewards/code_complexity_reward/std": 0.08959560096263885, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 550, "step_time": 47.32696792297065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 371.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 120.1015625, "completions/mean_terminated_length": 120.1015625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23899825615808368, "epoch": 0.313960113960114, "frac_reward_zero_std": 0.359375, "grad_norm": 0.05580990016460419, "kl": 0.17371007020119578, "learning_rate": 4.3392211211664426e-06, "loss": 0.0008686658693477511, "num_tokens": 96558987.0, "reward": 2.3546388149261475, "reward_std": 0.5001460313796997, "rewards/code_complexity_reward/mean": 0.9169921875, "rewards/code_complexity_reward/std": 0.0858612135052681, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 551, "step_time": 48.619574531912804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 115.833984375, "completions/mean_terminated_length": 115.833984375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22479557781480253, "epoch": 0.3145299145299145, "frac_reward_zero_std": 0.375, "grad_norm": 0.05913609266281128, "kl": 0.1851979224011302, "learning_rate": 4.335848477649305e-06, "loss": 0.0009262352250516415, "num_tokens": 96689206.0, "reward": 2.27587890625, "reward_std": 0.48259249329566956, "rewards/code_complexity_reward/mean": 0.9190430045127869, "rewards/code_complexity_reward/std": 0.11400750279426575, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 552, "step_time": 42.42359284777194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 118.3671875, "completions/mean_terminated_length": 118.3671875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23407210782170296, "epoch": 0.31509971509971507, "frac_reward_zero_std": 0.328125, "grad_norm": 0.06539369374513626, "kl": 0.18254462734330446, "learning_rate": 4.3324685668474405e-06, "loss": 0.0009129554382525384, "num_tokens": 96817722.0, "reward": 2.276611328125, "reward_std": 0.47209930419921875, "rewards/code_complexity_reward/mean": 0.912402331829071, "rewards/code_complexity_reward/std": 0.10245274752378464, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 553, "step_time": 62.74498334340751 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 456.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 119.033203125, "completions/mean_terminated_length": 119.033203125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23145971191115677, "epoch": 0.3156695156695157, "frac_reward_zero_std": 0.421875, "grad_norm": 0.05753250792622566, "kl": 0.1713134569581598, "learning_rate": 4.329081402140373e-06, "loss": 0.0008566250326111913, "num_tokens": 96946067.0, "reward": 2.251708984375, "reward_std": 0.4747728109359741, "rewards/code_complexity_reward/mean": 0.908984363079071, "rewards/code_complexity_reward/std": 0.12044984847307205, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 554, "step_time": 54.464133565314114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 441.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 126.298828125, "completions/mean_terminated_length": 126.298828125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23254968062974513, "epoch": 0.3162393162393162, "frac_reward_zero_std": 0.359375, "grad_norm": 0.05355829373002052, "kl": 0.15901605656836182, "learning_rate": 4.325686996936336e-06, "loss": 0.0007951621664687991, "num_tokens": 97078388.0, "reward": 2.302783489227295, "reward_std": 0.5407497882843018, "rewards/code_complexity_reward/mean": 0.8936523199081421, "rewards/code_complexity_reward/std": 0.1529991775751114, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 555, "step_time": 51.378043266013265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 456.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 120.7265625, "completions/mean_terminated_length": 120.7265625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.21815654821693897, "epoch": 0.31680911680911683, "frac_reward_zero_std": 0.421875, "grad_norm": 0.05945544317364693, "kl": 0.1571457856334746, "learning_rate": 4.32228536467223e-06, "loss": 0.0007858219323679805, "num_tokens": 97210552.0, "reward": 2.2864747047424316, "reward_std": 0.5119500756263733, "rewards/code_complexity_reward/mean": 0.9041992425918579, "rewards/code_complexity_reward/std": 0.1305825561285019, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.0275954008102417, "step": 556, "step_time": 44.795490832068026 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 454.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 121.240234375, "completions/mean_terminated_length": 121.240234375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2456729826517403, "epoch": 0.3173789173789174, "frac_reward_zero_std": 0.484375, "grad_norm": 0.055338796228170395, "kl": 0.1686518577625975, "learning_rate": 4.31887651881356e-06, "loss": 0.0008431980386376381, "num_tokens": 97344419.0, "reward": 2.2051758766174316, "reward_std": 0.44131937623023987, "rewards/code_complexity_reward/mean": 0.9069335460662842, "rewards/code_complexity_reward/std": 0.1128087043762207, "rewards/code_execution_reward/mean": 0.203125, "rewards/code_execution_reward/std": 0.4027182459831238, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 557, "step_time": 52.63640658836812 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 443.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 116.5234375, "completions/mean_terminated_length": 116.5234375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2453936745878309, "epoch": 0.31794871794871793, "frac_reward_zero_std": 0.421875, "grad_norm": 0.059012312442064285, "kl": 0.1958548673428595, "learning_rate": 4.315460472854389e-06, "loss": 0.0009791525080800056, "num_tokens": 97477183.0, "reward": 2.3212404251098633, "reward_std": 0.4761475920677185, "rewards/code_complexity_reward/mean": 0.9246094226837158, "rewards/code_complexity_reward/std": 0.08246355503797531, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 558, "step_time": 45.32782872766256 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 498.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 118.341796875, "completions/mean_terminated_length": 118.341796875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24502379726618528, "epoch": 0.31851851851851853, "frac_reward_zero_std": 0.390625, "grad_norm": 0.05567416548728943, "kl": 0.17000641371123493, "learning_rate": 4.3120372403172816e-06, "loss": 0.0008499313844367862, "num_tokens": 97606398.0, "reward": 2.343994140625, "reward_std": 0.5055749416351318, "rewards/code_complexity_reward/mean": 0.915332019329071, "rewards/code_complexity_reward/std": 0.10376609861850739, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 559, "step_time": 48.10229159053415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 115.96484375, "completions/mean_terminated_length": 115.96484375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22859458671882749, "epoch": 0.3190883190883191, "frac_reward_zero_std": 0.5, "grad_norm": 0.04600020498037338, "kl": 0.16961073537822813, "learning_rate": 4.308606834753249e-06, "loss": 0.0008483415585942566, "num_tokens": 97733796.0, "reward": 2.344531297683716, "reward_std": 0.494954377412796, "rewards/code_complexity_reward/mean": 0.9154297113418579, "rewards/code_complexity_reward/std": 0.0894438698887825, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 560, "step_time": 41.06168116722256 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 120.2890625, "completions/mean_terminated_length": 120.2890625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24150822614319623, "epoch": 0.31965811965811963, "frac_reward_zero_std": 0.375, "grad_norm": 0.05320151522755623, "kl": 0.1615952936699614, "learning_rate": 4.3051692697417e-06, "loss": 0.0008080520783551037, "num_tokens": 97864304.0, "reward": 2.261718988418579, "reward_std": 0.5131818056106567, "rewards/code_complexity_reward/mean": 0.8941406011581421, "rewards/code_complexity_reward/std": 0.14556336402893066, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 561, "step_time": 44.436551864258945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 115.80859375, "completions/mean_terminated_length": 115.80859375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23491639364510775, "epoch": 0.32022792022792024, "frac_reward_zero_std": 0.453125, "grad_norm": 0.09486223757266998, "kl": 0.2824828951852396, "learning_rate": 4.301724558890381e-06, "loss": 0.0014135736273601651, "num_tokens": 97993622.0, "reward": 2.2892580032348633, "reward_std": 0.5131055116653442, "rewards/code_complexity_reward/mean": 0.9079101085662842, "rewards/code_complexity_reward/std": 0.1341455727815628, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 562, "step_time": 49.11300704814494 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 114.033203125, "completions/mean_terminated_length": 114.033203125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24824500223621726, "epoch": 0.3207977207977208, "frac_reward_zero_std": 0.421875, "grad_norm": 0.13260626792907715, "kl": 0.30997802678029984, "learning_rate": 4.298272715835329e-06, "loss": 0.0015522114699706435, "num_tokens": 98122023.0, "reward": 2.3018555641174316, "reward_std": 0.4982469379901886, "rewards/code_complexity_reward/mean": 0.914746105670929, "rewards/code_complexity_reward/std": 0.11187922209501266, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 563, "step_time": 51.64101884979755 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 120.875, "completions/mean_terminated_length": 119.3411865234375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24017807794734836, "epoch": 0.3213675213675214, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06800999492406845, "kl": 0.16743216768372804, "learning_rate": 4.29481375424081e-06, "loss": 0.0008371389703825116, "num_tokens": 98249615.0, "reward": 2.244921922683716, "reward_std": 0.4862000644207001, "rewards/code_complexity_reward/mean": 0.9022460579872131, "rewards/code_complexity_reward/std": 0.13580891489982605, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 564, "step_time": 58.22518529649824 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 333.0, "completions/max_terminated_length": 333.0, "completions/mean_length": 116.53125, "completions/mean_terminated_length": 116.53125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2319574763532728, "epoch": 0.32193732193732194, "frac_reward_zero_std": 0.34375, "grad_norm": 0.05962454155087471, "kl": 0.1546325811650604, "learning_rate": 4.291347687799274e-06, "loss": 0.0007732115918770432, "num_tokens": 98381303.0, "reward": 2.338134765625, "reward_std": 0.516154944896698, "rewards/code_complexity_reward/mean": 0.9172852039337158, "rewards/code_complexity_reward/std": 0.11640846729278564, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 565, "step_time": 37.109416044317186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 431.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 116.021484375, "completions/mean_terminated_length": 116.021484375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24005476478487253, "epoch": 0.3225071225071225, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06096187233924866, "kl": 0.16647352406289428, "learning_rate": 4.2878745302312914e-06, "loss": 0.0008324763039126992, "num_tokens": 98508714.0, "reward": 2.369385004043579, "reward_std": 0.49558863043785095, "rewards/code_complexity_reward/mean": 0.920214831829071, "rewards/code_complexity_reward/std": 0.0738019272685051, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 566, "step_time": 44.267849860712886 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 119.275390625, "completions/mean_terminated_length": 119.275390625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2357796027790755, "epoch": 0.3230769230769231, "frac_reward_zero_std": 0.34375, "grad_norm": 0.07448653131723404, "kl": 0.1870755418203771, "learning_rate": 4.284394295285505e-06, "loss": 0.0009353517089039087, "num_tokens": 98640407.0, "reward": 2.2432618141174316, "reward_std": 0.4727697968482971, "rewards/code_complexity_reward/mean": 0.9127929210662842, "rewards/code_complexity_reward/std": 0.12298239767551422, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 567, "step_time": 39.74547299928963 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 281.0, "completions/max_terminated_length": 281.0, "completions/mean_length": 109.4140625, "completions/mean_terminated_length": 109.4140625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23699919157661498, "epoch": 0.32364672364672364, "frac_reward_zero_std": 0.453125, "grad_norm": 0.058860983699560165, "kl": 0.18238112493418157, "learning_rate": 4.280906996738575e-06, "loss": 0.00091215455904603, "num_tokens": 98762627.0, "reward": 2.232715129852295, "reward_std": 0.43468621373176575, "rewards/code_complexity_reward/mean": 0.9256836175918579, "rewards/code_complexity_reward/std": 0.0889684408903122, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 568, "step_time": 42.84486521221697 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 111.447265625, "completions/mean_terminated_length": 111.447265625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23985383613035083, "epoch": 0.3242165242165242, "frac_reward_zero_std": 0.40625, "grad_norm": 0.060395900160074234, "kl": 0.18150568462442607, "learning_rate": 4.2774126483951214e-06, "loss": 0.0009075364796444774, "num_tokens": 98888112.0, "reward": 2.314697265625, "reward_std": 0.5167072415351868, "rewards/code_complexity_reward/mean": 0.917285144329071, "rewards/code_complexity_reward/std": 0.12748144567012787, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 569, "step_time": 38.374051603488624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 119.52734375, "completions/mean_terminated_length": 119.52734375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23820462240837514, "epoch": 0.3247863247863248, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06257317960262299, "kl": 0.17271487112157047, "learning_rate": 4.273911264087671e-06, "loss": 0.0008634324185550213, "num_tokens": 99016054.0, "reward": 2.246875047683716, "reward_std": 0.4587164521217346, "rewards/code_complexity_reward/mean": 0.9134765863418579, "rewards/code_complexity_reward/std": 0.10456331074237823, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 570, "step_time": 43.94328384194523 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 413.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 110.876953125, "completions/mean_terminated_length": 110.876953125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23619534284807742, "epoch": 0.32535612535612535, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06246279552578926, "kl": 0.17154606489930302, "learning_rate": 4.270402857676605e-06, "loss": 0.0008579905843362212, "num_tokens": 99139303.0, "reward": 2.45654296875, "reward_std": 0.5226767659187317, "rewards/code_complexity_reward/mean": 0.9200195074081421, "rewards/code_complexity_reward/std": 0.0846577063202858, "rewards/code_execution_reward/mean": 0.439453125, "rewards/code_execution_reward/std": 0.49680593609809875, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 571, "step_time": 61.34165361709893 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 434.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 122.193359375, "completions/mean_terminated_length": 122.193359375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2367784772068262, "epoch": 0.32592592592592595, "frac_reward_zero_std": 0.421875, "grad_norm": 0.05375044420361519, "kl": 0.1734732681652531, "learning_rate": 4.266887443050099e-06, "loss": 0.0008671684190630913, "num_tokens": 99271650.0, "reward": 2.2752442359924316, "reward_std": 0.4927445352077484, "rewards/code_complexity_reward/mean": 0.9131835699081421, "rewards/code_complexity_reward/std": 0.11084544658660889, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 572, "step_time": 58.412838087417185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 115.890625, "completions/mean_terminated_length": 115.890625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2610049466602504, "epoch": 0.3264957264957265, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05704304203391075, "kl": 0.16525822191033512, "learning_rate": 4.2633650341240706e-06, "loss": 0.0008265033829957247, "num_tokens": 99402426.0, "reward": 2.228271484375, "reward_std": 0.4603642523288727, "rewards/code_complexity_reward/mean": 0.911914050579071, "rewards/code_complexity_reward/std": 0.11685142666101456, "rewards/code_execution_reward/mean": 0.22265625, "rewards/code_execution_reward/std": 0.41643625497817993, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 573, "step_time": 38.75883698835969 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 456.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 115.3359375, "completions/mean_terminated_length": 115.3359375, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.23874731711111963, "epoch": 0.32706552706552705, "frac_reward_zero_std": 0.390625, "grad_norm": 0.05802225321531296, "kl": 0.18494107527658343, "learning_rate": 4.259835644842128e-06, "loss": 0.0009246226400136948, "num_tokens": 99528510.0, "reward": 2.2828125953674316, "reward_std": 0.48191359639167786, "rewards/code_complexity_reward/mean": 0.9154296517372131, "rewards/code_complexity_reward/std": 0.10100381821393967, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 574, "step_time": 55.20995398983359 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 110.591796875, "completions/mean_terminated_length": 110.591796875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23581687384285033, "epoch": 0.32763532763532766, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05506492033600807, "kl": 0.19006453978363425, "learning_rate": 4.256299289175511e-06, "loss": 0.0009502573520876467, "num_tokens": 99653717.0, "reward": 2.283447265625, "reward_std": 0.48303478956222534, "rewards/code_complexity_reward/mean": 0.9221678972244263, "rewards/code_complexity_reward/std": 0.10980229079723358, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 575, "step_time": 67.40851059462875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 115.234375, "completions/mean_terminated_length": 114.45792388916016, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23251285776495934, "epoch": 0.3282051282051282, "frac_reward_zero_std": 0.40625, "grad_norm": 0.061542313545942307, "kl": 0.1623246936360374, "learning_rate": 4.252755981123031e-06, "loss": 0.0008115380769595504, "num_tokens": 99782421.0, "reward": 2.341845989227295, "reward_std": 0.48554763197898865, "rewards/code_complexity_reward/mean": 0.9256835579872131, "rewards/code_complexity_reward/std": 0.07132737338542938, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 576, "step_time": 58.78698748815805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 459.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 117.873046875, "completions/mean_terminated_length": 117.873046875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23995790490880609, "epoch": 0.32877492877492875, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06503696739673615, "kl": 0.18451284221373498, "learning_rate": 4.249205734711028e-06, "loss": 0.000923157436773181, "num_tokens": 99911404.0, "reward": 2.3246095180511475, "reward_std": 0.49128395318984985, "rewards/code_complexity_reward/mean": 0.912109375, "rewards/code_complexity_reward/std": 0.10193741321563721, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 577, "step_time": 49.3942587506026 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 115.681640625, "completions/mean_terminated_length": 115.681640625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2568234261125326, "epoch": 0.32934472934472936, "frac_reward_zero_std": 0.359375, "grad_norm": 0.07240568846464157, "kl": 0.17606895812787116, "learning_rate": 4.245648563993303e-06, "loss": 0.0008808852289803326, "num_tokens": 100041369.0, "reward": 2.2383790016174316, "reward_std": 0.4712753891944885, "rewards/code_complexity_reward/mean": 0.9088866710662842, "rewards/code_complexity_reward/std": 0.11867718398571014, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 578, "step_time": 38.80038811918348 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 421.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 127.548828125, "completions/mean_terminated_length": 127.548828125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2509223259985447, "epoch": 0.3299145299145299, "frac_reward_zero_std": 0.4375, "grad_norm": 0.0569421611726284, "kl": 0.16564507805742323, "learning_rate": 4.242084483051069e-06, "loss": 0.0008281066548079252, "num_tokens": 100177106.0, "reward": 2.2411623001098633, "reward_std": 0.48376408219337463, "rewards/code_complexity_reward/mean": 0.90625, "rewards/code_complexity_reward/std": 0.12476984411478043, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 579, "step_time": 43.103756970725954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 362.0, "completions/max_terminated_length": 362.0, "completions/mean_length": 114.166015625, "completions/mean_terminated_length": 114.166015625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23972845473326743, "epoch": 0.33048433048433046, "frac_reward_zero_std": 0.5, "grad_norm": 0.05562948063015938, "kl": 0.170942984521389, "learning_rate": 4.2385135059928915e-06, "loss": 0.0008547771722078323, "num_tokens": 100309399.0, "reward": 2.3114259243011475, "reward_std": 0.4686960279941559, "rewards/code_complexity_reward/mean": 0.9220702648162842, "rewards/code_complexity_reward/std": 0.0653247982263565, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 580, "step_time": 47.54225577786565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 326.0, "completions/max_terminated_length": 326.0, "completions/mean_length": 112.078125, "completions/mean_terminated_length": 112.078125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23047792329452932, "epoch": 0.33105413105413106, "frac_reward_zero_std": 0.375, "grad_norm": 0.06347614526748657, "kl": 0.16288887639530003, "learning_rate": 4.2349356469546364e-06, "loss": 0.000814613071270287, "num_tokens": 100435703.0, "reward": 2.3275392055511475, "reward_std": 0.5059295892715454, "rewards/code_complexity_reward/mean": 0.9188476204872131, "rewards/code_complexity_reward/std": 0.10251838713884354, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 581, "step_time": 36.480862823314965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 104.82421875, "completions/mean_terminated_length": 104.82421875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2369942197110504, "epoch": 0.3316239316239316, "frac_reward_zero_std": 0.625, "grad_norm": 0.04983570799231529, "kl": 0.17866081045940518, "learning_rate": 4.231350920099412e-06, "loss": 0.0008933095959946513, "num_tokens": 100555557.0, "reward": 2.3998048305511475, "reward_std": 0.48947829008102417, "rewards/code_complexity_reward/mean": 0.9296875, "rewards/code_complexity_reward/std": 0.060473501682281494, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 582, "step_time": 42.03645411320031 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 115.728515625, "completions/mean_terminated_length": 115.728515625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2537089795805514, "epoch": 0.3321937321937322, "frac_reward_zero_std": 0.359375, "grad_norm": 0.07410811632871628, "kl": 0.1793129532597959, "learning_rate": 4.227759339617513e-06, "loss": 0.0008966419263742864, "num_tokens": 100685850.0, "reward": 2.2025880813598633, "reward_std": 0.4800719618797302, "rewards/code_complexity_reward/mean": 0.9036133289337158, "rewards/code_complexity_reward/std": 0.14745602011680603, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 583, "step_time": 57.82520010136068 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 496.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 114.0703125, "completions/mean_terminated_length": 114.0703125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21801252523437142, "epoch": 0.33276353276353277, "frac_reward_zero_std": 0.375, "grad_norm": 0.061472613364458084, "kl": 0.18055666016880423, "learning_rate": 4.224160919726366e-06, "loss": 0.0009030074579641223, "num_tokens": 100812790.0, "reward": 2.3233399391174316, "reward_std": 0.5222235321998596, "rewards/code_complexity_reward/mean": 0.910839855670929, "rewards/code_complexity_reward/std": 0.12455273419618607, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 584, "step_time": 47.73322250600904 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 481.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 112.89453125, "completions/mean_terminated_length": 112.89453125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2413024373818189, "epoch": 0.3333333333333333, "frac_reward_zero_std": 0.5, "grad_norm": 0.06344752013683319, "kl": 0.18294395692646503, "learning_rate": 4.220555674670465e-06, "loss": 0.0009146028896793723, "num_tokens": 100939280.0, "reward": 2.2936525344848633, "reward_std": 0.4647493064403534, "rewards/code_complexity_reward/mean": 0.917285144329071, "rewards/code_complexity_reward/std": 0.08531656861305237, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 585, "step_time": 50.3287717057392 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 422.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 115.73046875, "completions/mean_terminated_length": 115.73046875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2397517366334796, "epoch": 0.3339031339031339, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06305961310863495, "kl": 0.17332450719550252, "learning_rate": 4.216943618721333e-06, "loss": 0.0008665939094498754, "num_tokens": 101067614.0, "reward": 2.326709032058716, "reward_std": 0.4993104636669159, "rewards/code_complexity_reward/mean": 0.9117187261581421, "rewards/code_complexity_reward/std": 0.10468186438083649, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 586, "step_time": 43.45751782786101 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 368.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 111.00390625, "completions/mean_terminated_length": 111.00390625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22727748355828226, "epoch": 0.33447293447293447, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05807332322001457, "kl": 0.16969803103711456, "learning_rate": 4.213324766177444e-06, "loss": 0.0008484044228680432, "num_tokens": 101188920.0, "reward": 2.2337892055511475, "reward_std": 0.42774489521980286, "rewards/code_complexity_reward/mean": 0.9238280653953552, "rewards/code_complexity_reward/std": 0.08203976601362228, "rewards/code_execution_reward/mean": 0.212890625, "rewards/code_execution_reward/std": 0.409751296043396, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 587, "step_time": 45.17120980657637 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 111.1796875, "completions/mean_terminated_length": 111.1796875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2460459906142205, "epoch": 0.335042735042735, "frac_reward_zero_std": 0.375, "grad_norm": 0.06410021334886551, "kl": 0.1876766785280779, "learning_rate": 4.209699131364182e-06, "loss": 0.0009387439349666238, "num_tokens": 101313324.0, "reward": 2.270068645477295, "reward_std": 0.5145475268363953, "rewards/code_complexity_reward/mean": 0.9097656011581421, "rewards/code_complexity_reward/std": 0.14118710160255432, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 588, "step_time": 49.03317722864449 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 122.5078125, "completions/mean_terminated_length": 122.5078125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23513277247548103, "epoch": 0.3356125356125356, "frac_reward_zero_std": 0.359375, "grad_norm": 0.05768798291683197, "kl": 0.17490677838213742, "learning_rate": 4.206066728633777e-06, "loss": 0.0008749548578634858, "num_tokens": 101445624.0, "reward": 2.325732469558716, "reward_std": 0.4880426526069641, "rewards/code_complexity_reward/mean": 0.91650390625, "rewards/code_complexity_reward/std": 0.09397757798433304, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 589, "step_time": 47.13905525393784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 107.078125, "completions/mean_terminated_length": 107.078125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23475959221832454, "epoch": 0.33618233618233617, "frac_reward_zero_std": 0.53125, "grad_norm": 0.0591200515627861, "kl": 0.18004872545134276, "learning_rate": 4.202427572365251e-06, "loss": 0.0009000752470456064, "num_tokens": 101569240.0, "reward": 2.330566644668579, "reward_std": 0.5012297034263611, "rewards/code_complexity_reward/mean": 0.9200195074081421, "rewards/code_complexity_reward/std": 0.10582172870635986, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 590, "step_time": 47.64652639999986 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 333.0, "completions/max_terminated_length": 333.0, "completions/mean_length": 116.552734375, "completions/mean_terminated_length": 116.552734375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22765809227712452, "epoch": 0.3367521367521368, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05251462012529373, "kl": 0.1878194265300408, "learning_rate": 4.19878167696436e-06, "loss": 0.0009392810170538723, "num_tokens": 101696395.0, "reward": 2.30859375, "reward_std": 0.4697757065296173, "rewards/code_complexity_reward/mean": 0.9273437261581421, "rewards/code_complexity_reward/std": 0.07332190126180649, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 591, "step_time": 56.06853820383549 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 373.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 121.390625, "completions/mean_terminated_length": 121.390625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2437070095911622, "epoch": 0.3373219373219373, "frac_reward_zero_std": 0.40625, "grad_norm": 0.056789349764585495, "kl": 0.17971131845843047, "learning_rate": 4.195129056863535e-06, "loss": 0.0008989999769255519, "num_tokens": 101831315.0, "reward": 2.23876953125, "reward_std": 0.4762416481971741, "rewards/code_complexity_reward/mean": 0.9024413824081421, "rewards/code_complexity_reward/std": 0.1280934065580368, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 592, "step_time": 40.54389488045126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 382.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 114.49609375, "completions/mean_terminated_length": 114.49609375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22875352366827428, "epoch": 0.3378917378917379, "frac_reward_zero_std": 0.375, "grad_norm": 0.062084801495075226, "kl": 0.1860606506234035, "learning_rate": 4.191469726521832e-06, "loss": 0.0009307077270932496, "num_tokens": 101960041.0, "reward": 2.2906250953674316, "reward_std": 0.5162715315818787, "rewards/code_complexity_reward/mean": 0.9044921398162842, "rewards/code_complexity_reward/std": 0.13837657868862152, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 593, "step_time": 41.08983220718801 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 112.912109375, "completions/mean_terminated_length": 112.912109375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2312198372092098, "epoch": 0.3384615384615385, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05552533641457558, "kl": 0.17722782585769892, "learning_rate": 4.1878037004248635e-06, "loss": 0.0008863817201927304, "num_tokens": 102087372.0, "reward": 2.3233888149261475, "reward_std": 0.49592873454093933, "rewards/code_complexity_reward/mean": 0.91796875, "rewards/code_complexity_reward/std": 0.10092224180698395, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 594, "step_time": 58.27945764455944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 119.93359375, "completions/mean_terminated_length": 118.39608764648438, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.22589672170579433, "epoch": 0.33903133903133903, "frac_reward_zero_std": 0.4375, "grad_norm": 0.0589112862944603, "kl": 0.18943268759176135, "learning_rate": 4.184130993084753e-06, "loss": 0.0009473467944189906, "num_tokens": 102216874.0, "reward": 2.27001953125, "reward_std": 0.5195230841636658, "rewards/code_complexity_reward/mean": 0.9029296636581421, "rewards/code_complexity_reward/std": 0.1474878042936325, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 595, "step_time": 94.79124774225056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 335.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 109.125, "completions/mean_terminated_length": 109.125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2323851096443832, "epoch": 0.3396011396011396, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06235530972480774, "kl": 0.18696449231356382, "learning_rate": 4.180451619040068e-06, "loss": 0.0009348592720925808, "num_tokens": 102342378.0, "reward": 2.3851563930511475, "reward_std": 0.5240351557731628, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.1099439188838005, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 596, "step_time": 37.41149848327041 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 122.787109375, "completions/mean_terminated_length": 122.02543640136719, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24894777103327215, "epoch": 0.3401709401709402, "frac_reward_zero_std": 0.3125, "grad_norm": 0.06117106229066849, "kl": 0.16195353434886783, "learning_rate": 4.176765592855769e-06, "loss": 0.0008097727550193667, "num_tokens": 102476509.0, "reward": 2.2799317836761475, "reward_std": 0.4954330325126648, "rewards/code_complexity_reward/mean": 0.90869140625, "rewards/code_complexity_reward/std": 0.12389490008354187, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 597, "step_time": 67.75312107615173 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 338.0, "completions/max_terminated_length": 338.0, "completions/mean_length": 109.486328125, "completions/mean_terminated_length": 109.486328125, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.23079862375743687, "epoch": 0.34074074074074073, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06593427062034607, "kl": 0.17671745317056775, "learning_rate": 4.173072929123149e-06, "loss": 0.0008833928732201457, "num_tokens": 102599838.0, "reward": 2.3479981422424316, "reward_std": 0.4942598044872284, "rewards/code_complexity_reward/mean": 0.9210937023162842, "rewards/code_complexity_reward/std": 0.0822528600692749, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 598, "step_time": 54.844114603474736 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 109.998046875, "completions/mean_terminated_length": 109.998046875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23545243008993566, "epoch": 0.34131054131054134, "frac_reward_zero_std": 0.5, "grad_norm": 0.054522205144166946, "kl": 0.1731799041153863, "learning_rate": 4.169373642459773e-06, "loss": 0.0008661657921038568, "num_tokens": 102726421.0, "reward": 2.3438966274261475, "reward_std": 0.502890944480896, "rewards/code_complexity_reward/mean": 0.914257824420929, "rewards/code_complexity_reward/std": 0.09332851320505142, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 599, "step_time": 36.50155033916235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 105.01953125, "completions/mean_terminated_length": 105.01953125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23735753470100462, "epoch": 0.3418803418803419, "frac_reward_zero_std": 0.46875, "grad_norm": 0.052443478256464005, "kl": 0.1940904415678233, "learning_rate": 4.165667747509427e-06, "loss": 0.0009706401033326983, "num_tokens": 102846879.0, "reward": 2.3178224563598633, "reward_std": 0.48669782280921936, "rewards/code_complexity_reward/mean": 0.919238269329071, "rewards/code_complexity_reward/std": 0.08812021464109421, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 600, "step_time": 34.78923406265676 }, { "epoch": 0.3418803418803419, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 157.63, "eval_completions/max_terminated_length": 157.63, "eval_completions/mean_length": 113.76625, "eval_completions/mean_terminated_length": 113.76625, "eval_completions/min_length": 82.88, "eval_completions/min_terminated_length": 82.88, "eval_entropy": 0.23725016467273236, "eval_frac_reward_zero_std": 0.41, "eval_kl": 0.18120441682636737, "eval_loss": 0.002679017838090658, "eval_num_tokens": 102846879.0, "eval_reward": 2.2707501244544983, "eval_reward_std": 0.23054587630555035, "eval_rewards/code_complexity_reward/mean": 0.9069999849796295, "eval_rewards/code_complexity_reward/std": 0.0482896145246923, "eval_rewards/code_execution_reward/mean": 0.27375, "eval_rewards/code_execution_reward/std": 0.17679941326379775, "eval_rewards/code_syntax_reward/mean": 0.49, "eval_rewards/code_syntax_reward/std": 0.020222864747047424, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.5, "eval_rewards/xmlcount_reward_func/std": 0.0, "eval_runtime": 742.9926, "eval_samples_per_second": 0.135, "eval_steps_per_second": 0.017, "step": 600 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 474.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 113.501953125, "completions/mean_terminated_length": 113.501953125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2383510114159435, "epoch": 0.34245014245014244, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06637506932020187, "kl": 0.16732569865416735, "learning_rate": 4.161955258942057e-06, "loss": 0.0008367542759515345, "num_tokens": 102973176.0, "reward": 2.305469036102295, "reward_std": 0.5207151174545288, "rewards/code_complexity_reward/mean": 0.9076171517372131, "rewards/code_complexity_reward/std": 0.1338156908750534, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 601, "step_time": 49.30654059816152 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 119.0625, "completions/mean_terminated_length": 119.0625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23921845760196447, "epoch": 0.34301994301994304, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06924556940793991, "kl": 0.17597296158783138, "learning_rate": 4.158236191453703e-06, "loss": 0.0008798960479907691, "num_tokens": 103105936.0, "reward": 2.3046388626098633, "reward_std": 0.5139961242675781, "rewards/code_complexity_reward/mean": 0.911914050579071, "rewards/code_complexity_reward/std": 0.1292937695980072, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 602, "step_time": 53.35702347289771 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 112.546875, "completions/mean_terminated_length": 112.546875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22806192934513092, "epoch": 0.3435897435897436, "frac_reward_zero_std": 0.453125, "grad_norm": 0.061635151505470276, "kl": 0.21989501442294568, "learning_rate": 4.154510559766456e-06, "loss": 0.0010995065094903111, "num_tokens": 103231432.0, "reward": 2.269775390625, "reward_std": 0.515679657459259, "rewards/code_complexity_reward/mean": 0.903613269329071, "rewards/code_complexity_reward/std": 0.145821213722229, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 603, "step_time": 37.30754106771201 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 381.0, "completions/max_terminated_length": 381.0, "completions/mean_length": 111.71484375, "completions/mean_terminated_length": 111.71484375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23888190230354667, "epoch": 0.34415954415954414, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06601164489984512, "kl": 0.17773336998652667, "learning_rate": 4.150778378628387e-06, "loss": 0.0008885924471542239, "num_tokens": 103355958.0, "reward": 2.3064942359924316, "reward_std": 0.4959997236728668, "rewards/code_complexity_reward/mean": 0.9178711175918579, "rewards/code_complexity_reward/std": 0.10762494802474976, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 604, "step_time": 39.10096556134522 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 105.623046875, "completions/mean_terminated_length": 105.623046875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23585079866461456, "epoch": 0.34472934472934474, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06436892598867416, "kl": 0.18388333334587514, "learning_rate": 4.147039662813495e-06, "loss": 0.0009195044985972345, "num_tokens": 103477853.0, "reward": 2.3924806118011475, "reward_std": 0.5212669968605042, "rewards/code_complexity_reward/mean": 0.92529296875, "rewards/code_complexity_reward/std": 0.10644546896219254, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 605, "step_time": 38.007206469774246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 366.0, "completions/max_terminated_length": 366.0, "completions/mean_length": 111.490234375, "completions/mean_terminated_length": 111.490234375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23628262127749622, "epoch": 0.3452991452991453, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05859498679637909, "kl": 0.1927013578824699, "learning_rate": 4.1432944271216475e-06, "loss": 0.0009635695023462176, "num_tokens": 103604000.0, "reward": 2.339160203933716, "reward_std": 0.4858211576938629, "rewards/code_complexity_reward/mean": 0.9188476800918579, "rewards/code_complexity_reward/std": 0.07560091465711594, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 606, "step_time": 46.76686047296971 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 112.8046875, "completions/mean_terminated_length": 112.8046875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24348281952552497, "epoch": 0.34586894586894584, "frac_reward_zero_std": 0.5, "grad_norm": 0.06917499005794525, "kl": 0.19734150753356516, "learning_rate": 4.139542686378519e-06, "loss": 0.0009869430214166641, "num_tokens": 103728116.0, "reward": 2.2669923305511475, "reward_std": 0.46248114109039307, "rewards/code_complexity_reward/mean": 0.923828125, "rewards/code_complexity_reward/std": 0.09846469759941101, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 607, "step_time": 40.65103352069855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 110.296875, "completions/mean_terminated_length": 110.296875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.23616289789788425, "epoch": 0.34643874643874645, "frac_reward_zero_std": 0.453125, "grad_norm": 1.1539963483810425, "kl": 0.18812797719147056, "learning_rate": 4.135784455435536e-06, "loss": 0.0009409255580976605, "num_tokens": 103855276.0, "reward": 2.3117189407348633, "reward_std": 0.541875958442688, "rewards/code_complexity_reward/mean": 0.9109375476837158, "rewards/code_complexity_reward/std": 0.15535925328731537, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 608, "step_time": 35.158955994062126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 437.0, "completions/max_terminated_length": 437.0, "completions/mean_length": 118.173828125, "completions/mean_terminated_length": 118.173828125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22699234844185412, "epoch": 0.347008547008547, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06490109860897064, "kl": 0.17749156488571316, "learning_rate": 4.1320197491698165e-06, "loss": 0.0008875161875039339, "num_tokens": 103984029.0, "reward": 2.3246092796325684, "reward_std": 0.5011140704154968, "rewards/code_complexity_reward/mean": 0.9108397960662842, "rewards/code_complexity_reward/std": 0.11123166233301163, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 609, "step_time": 43.195055017247796 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 114.265625, "completions/mean_terminated_length": 114.265625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2523451098240912, "epoch": 0.3475783475783476, "frac_reward_zero_std": 0.4375, "grad_norm": 0.07891048491001129, "kl": 0.1800728616071865, "learning_rate": 4.1282485824841115e-06, "loss": 0.0009002963779494166, "num_tokens": 104113653.0, "reward": 2.2804689407348633, "reward_std": 0.44639167189598083, "rewards/code_complexity_reward/mean": 0.928515613079071, "rewards/code_complexity_reward/std": 0.069716677069664, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 610, "step_time": 46.922002635896206 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 460.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 109.380859375, "completions/mean_terminated_length": 109.380859375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.23356578801758587, "epoch": 0.34814814814814815, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06531723588705063, "kl": 0.1996291344985366, "learning_rate": 4.124470970306746e-06, "loss": 0.0009982048068195581, "num_tokens": 104240512.0, "reward": 2.2672364711761475, "reward_std": 0.4790646731853485, "rewards/code_complexity_reward/mean": 0.91845703125, "rewards/code_complexity_reward/std": 0.11694145947694778, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 611, "step_time": 45.1312516303733 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 466.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 108.80078125, "completions/mean_terminated_length": 108.80078125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23634642153047025, "epoch": 0.3487179487179487, "frac_reward_zero_std": 0.625, "grad_norm": 0.048649199306964874, "kl": 0.17599830962717533, "learning_rate": 4.1206869275915575e-06, "loss": 0.0008802440715953708, "num_tokens": 104363178.0, "reward": 2.2484865188598633, "reward_std": 0.44792839884757996, "rewards/code_complexity_reward/mean": 0.924121081829071, "rewards/code_complexity_reward/std": 0.09215409308671951, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 612, "step_time": 55.55844254978001 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 383.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 113.810546875, "completions/mean_terminated_length": 113.810546875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24299076152965426, "epoch": 0.3492877492877493, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06104053184390068, "kl": 0.1784581784158945, "learning_rate": 4.1168964693178425e-06, "loss": 0.0008921066182665527, "num_tokens": 104488873.0, "reward": 2.327441453933716, "reward_std": 0.48436811566352844, "rewards/code_complexity_reward/mean": 0.9256834983825684, "rewards/code_complexity_reward/std": 0.07906945049762726, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 613, "step_time": 40.61695087980479 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 451.0, "completions/max_terminated_length": 451.0, "completions/mean_length": 109.173828125, "completions/mean_terminated_length": 109.173828125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22730028838850558, "epoch": 0.34985754985754985, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05591674894094467, "kl": 0.18611752334982157, "learning_rate": 4.113099610490292e-06, "loss": 0.0009307838627137244, "num_tokens": 104611274.0, "reward": 2.376171827316284, "reward_std": 0.5292229652404785, "rewards/code_complexity_reward/mean": 0.918749988079071, "rewards/code_complexity_reward/std": 0.1194409653544426, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 614, "step_time": 44.257872329093516 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 401.0, "completions/max_terminated_length": 401.0, "completions/mean_length": 116.5703125, "completions/mean_terminated_length": 116.5703125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24314835900440812, "epoch": 0.3504273504273504, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05522380396723747, "kl": 0.18655092944391072, "learning_rate": 4.109296366138935e-06, "loss": 0.0009330709581263363, "num_tokens": 104740862.0, "reward": 2.2887697219848633, "reward_std": 0.535789430141449, "rewards/code_complexity_reward/mean": 0.902636706829071, "rewards/code_complexity_reward/std": 0.15153448283672333, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 615, "step_time": 49.767922529019415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 289.0, "completions/max_terminated_length": 289.0, "completions/mean_length": 114.556640625, "completions/mean_terminated_length": 114.556640625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23211682285182178, "epoch": 0.350997150997151, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06680502742528915, "kl": 0.21303038054611534, "learning_rate": 4.105486751319075e-06, "loss": 0.0010651128832250834, "num_tokens": 104866067.0, "reward": 2.2667970657348633, "reward_std": 0.4933422803878784, "rewards/code_complexity_reward/mean": 0.9197266101837158, "rewards/code_complexity_reward/std": 0.12795059382915497, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 616, "step_time": 34.18826917745173 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 107.8203125, "completions/mean_terminated_length": 107.8203125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2339207655750215, "epoch": 0.35156695156695156, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05607902631163597, "kl": 0.18337667628657073, "learning_rate": 4.1016707811112364e-06, "loss": 0.0009170105331577361, "num_tokens": 104988783.0, "reward": 2.3465819358825684, "reward_std": 0.507066547870636, "rewards/code_complexity_reward/mean": 0.92431640625, "rewards/code_complexity_reward/std": 0.10360205173492432, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 617, "step_time": 47.59456484671682 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 341.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 109.375, "completions/mean_terminated_length": 109.375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2364127126056701, "epoch": 0.35213675213675216, "frac_reward_zero_std": 0.53125, "grad_norm": 0.052737683057785034, "kl": 0.1809463717509061, "learning_rate": 4.097848470621101e-06, "loss": 0.0009048818610608578, "num_tokens": 105112839.0, "reward": 2.349121332168579, "reward_std": 0.482242614030838, "rewards/code_complexity_reward/mean": 0.9249023199081421, "rewards/code_complexity_reward/std": 0.07565751671791077, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 618, "step_time": 36.91931826155633 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 113.0859375, "completions/mean_terminated_length": 113.0859375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24300266453064978, "epoch": 0.3527065527065527, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05891371890902519, "kl": 0.18540472048334777, "learning_rate": 4.094019834979447e-06, "loss": 0.0009271663147956133, "num_tokens": 105239339.0, "reward": 2.3189454078674316, "reward_std": 0.5013996958732605, "rewards/code_complexity_reward/mean": 0.916210949420929, "rewards/code_complexity_reward/std": 0.11378086358308792, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 619, "step_time": 36.23218022007495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 111.28125, "completions/mean_terminated_length": 111.28125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23081468767486513, "epoch": 0.35327635327635326, "frac_reward_zero_std": 0.515625, "grad_norm": 0.04990362003445625, "kl": 0.18885364336892962, "learning_rate": 4.090184889342093e-06, "loss": 0.0009445666801184416, "num_tokens": 105367211.0, "reward": 2.287402629852295, "reward_std": 0.5149454474449158, "rewards/code_complexity_reward/mean": 0.9149414300918579, "rewards/code_complexity_reward/std": 0.1393878012895584, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 620, "step_time": 41.47549583390355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 459.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 114.4296875, "completions/mean_terminated_length": 114.4296875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23784560267813504, "epoch": 0.35384615384615387, "frac_reward_zero_std": 0.359375, "grad_norm": 0.06972802430391312, "kl": 0.19060762785375118, "learning_rate": 4.086343648889835e-06, "loss": 0.0009529030066914856, "num_tokens": 105494783.0, "reward": 2.1788086891174316, "reward_std": 0.4433959722518921, "rewards/code_complexity_reward/mean": 0.9059569835662842, "rewards/code_complexity_reward/std": 0.13318565487861633, "rewards/code_execution_reward/mean": 0.181640625, "rewards/code_execution_reward/std": 0.38592514395713806, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 621, "step_time": 61.69360932987183 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 333.0, "completions/max_terminated_length": 333.0, "completions/mean_length": 104.048828125, "completions/mean_terminated_length": 104.048828125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22869850578717887, "epoch": 0.3544159544159544, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05009528622031212, "kl": 0.18050364102236927, "learning_rate": 4.0824961288283896e-06, "loss": 0.0009028965141624212, "num_tokens": 105615312.0, "reward": 2.326953172683716, "reward_std": 0.4840267598628998, "rewards/code_complexity_reward/mean": 0.9281249642372131, "rewards/code_complexity_reward/std": 0.08786442130804062, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 622, "step_time": 45.205739746801555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 112.87890625, "completions/mean_terminated_length": 112.87890625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24004328227601945, "epoch": 0.35498575498575496, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05715504661202431, "kl": 0.17840873077511787, "learning_rate": 4.078642344388327e-06, "loss": 0.0008920239633880556, "num_tokens": 105745074.0, "reward": 2.3449220657348633, "reward_std": 0.5059587359428406, "rewards/code_complexity_reward/mean": 0.92724609375, "rewards/code_complexity_reward/std": 0.10461387783288956, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 623, "step_time": 63.57939196098596 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 435.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 109.859375, "completions/mean_terminated_length": 109.859375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24259828147478402, "epoch": 0.35555555555555557, "frac_reward_zero_std": 0.515625, "grad_norm": 0.06679441779851913, "kl": 0.17808094224892557, "learning_rate": 4.0747823108250185e-06, "loss": 0.0008904431597329676, "num_tokens": 105870706.0, "reward": 2.2635254859924316, "reward_std": 0.46924901008605957, "rewards/code_complexity_reward/mean": 0.9237304925918579, "rewards/code_complexity_reward/std": 0.10798976570367813, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 624, "step_time": 43.263346125371754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 379.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 112.72265625, "completions/mean_terminated_length": 112.72265625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2289326151367277, "epoch": 0.3561253561253561, "frac_reward_zero_std": 0.4375, "grad_norm": 0.0650501549243927, "kl": 0.18844229099340737, "learning_rate": 4.0709160434185725e-06, "loss": 0.0009420836577191949, "num_tokens": 105997140.0, "reward": 2.2987308502197266, "reward_std": 0.48530593514442444, "rewards/code_complexity_reward/mean": 0.9116210341453552, "rewards/code_complexity_reward/std": 0.10811816900968552, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 625, "step_time": 46.62779372278601 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 115.564453125, "completions/mean_terminated_length": 115.564453125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2328420754056424, "epoch": 0.35669515669515667, "frac_reward_zero_std": 0.546875, "grad_norm": 0.06072783097624779, "kl": 0.1771392982918769, "learning_rate": 4.067043557473773e-06, "loss": 0.0008858561050146818, "num_tokens": 106125453.0, "reward": 2.320507764816284, "reward_std": 0.5388519167900085, "rewards/code_complexity_reward/mean": 0.899218738079071, "rewards/code_complexity_reward/std": 0.14618225395679474, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 626, "step_time": 40.19934383407235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 111.9921875, "completions/mean_terminated_length": 111.9921875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24195943004451692, "epoch": 0.3572649572649573, "frac_reward_zero_std": 0.640625, "grad_norm": 0.05155298486351967, "kl": 0.1796148109715432, "learning_rate": 4.063164868320023e-06, "loss": 0.0008982047438621521, "num_tokens": 106253105.0, "reward": 2.3541016578674316, "reward_std": 0.5069022178649902, "rewards/code_complexity_reward/mean": 0.9181640148162842, "rewards/code_complexity_reward/std": 0.09312022477388382, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 627, "step_time": 46.62200675718486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 252.0, "completions/max_terminated_length": 252.0, "completions/mean_length": 106.837890625, "completions/mean_terminated_length": 106.837890625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2517943773418665, "epoch": 0.3578347578347578, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05822316184639931, "kl": 0.18702706415206194, "learning_rate": 4.059279991311278e-06, "loss": 0.0009354769717901945, "num_tokens": 106379870.0, "reward": 2.2530274391174316, "reward_std": 0.4434957504272461, "rewards/code_complexity_reward/mean": 0.928417980670929, "rewards/code_complexity_reward/std": 0.08818698674440384, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 628, "step_time": 32.11759647447616 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 316.0, "completions/max_terminated_length": 316.0, "completions/mean_length": 107.740234375, "completions/mean_terminated_length": 107.740234375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2245407395530492, "epoch": 0.3584045584045584, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06059575080871582, "kl": 0.1702239253791049, "learning_rate": 4.05538894182599e-06, "loss": 0.0008511043852195144, "num_tokens": 106501969.0, "reward": 2.425049066543579, "reward_std": 0.5094303488731384, "rewards/code_complexity_reward/mean": 0.927050769329071, "rewards/code_complexity_reward/std": 0.0780491903424263, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 629, "step_time": 35.286958067677915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 414.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 120.91015625, "completions/mean_terminated_length": 120.91015625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23927058931440115, "epoch": 0.358974358974359, "frac_reward_zero_std": 0.421875, "grad_norm": 0.05610001087188721, "kl": 0.17503837903495878, "learning_rate": 4.051491735267044e-06, "loss": 0.0008751384448260069, "num_tokens": 106634011.0, "reward": 2.290722608566284, "reward_std": 0.47843602299690247, "rewards/code_complexity_reward/mean": 0.914355456829071, "rewards/code_complexity_reward/std": 0.0987890213727951, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 630, "step_time": 50.775963727384806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 480.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 109.48828125, "completions/mean_terminated_length": 109.48828125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24462459236383438, "epoch": 0.3595441595441595, "frac_reward_zero_std": 0.421875, "grad_norm": 0.062307391315698624, "kl": 0.21100214181933552, "learning_rate": 4.047588387061702e-06, "loss": 0.001054836786352098, "num_tokens": 106760573.0, "reward": 2.2710938453674316, "reward_std": 0.49160704016685486, "rewards/code_complexity_reward/mean": 0.9152343273162842, "rewards/code_complexity_reward/std": 0.12200359255075455, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 631, "step_time": 46.08856946695596 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 115.224609375, "completions/mean_terminated_length": 115.224609375, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.24776764982379973, "epoch": 0.36011396011396013, "frac_reward_zero_std": 0.34375, "grad_norm": 0.07384619116783142, "kl": 0.19625309947878122, "learning_rate": 4.0436789126615315e-06, "loss": 0.0009813782526180148, "num_tokens": 106888344.0, "reward": 2.282958984375, "reward_std": 0.48067232966423035, "rewards/code_complexity_reward/mean": 0.9236328601837158, "rewards/code_complexity_reward/std": 0.10694149136543274, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 632, "step_time": 57.677601493895054 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 345.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 112.12890625, "completions/mean_terminated_length": 112.12890625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23199384030885994, "epoch": 0.3606837606837607, "frac_reward_zero_std": 0.453125, "grad_norm": 0.052821654826402664, "kl": 0.17629973264411092, "learning_rate": 4.0397633275423555e-06, "loss": 0.0008817382622510195, "num_tokens": 107013762.0, "reward": 2.3324222564697266, "reward_std": 0.48010680079460144, "rewards/code_complexity_reward/mean": 0.9267578125, "rewards/code_complexity_reward/std": 0.07973068207502365, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 633, "step_time": 46.33662892319262 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 112.62109375, "completions/mean_terminated_length": 112.62109375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23203160613775253, "epoch": 0.36125356125356123, "frac_reward_zero_std": 0.515625, "grad_norm": 0.06253884732723236, "kl": 0.19263337121810764, "learning_rate": 4.035841647204185e-06, "loss": 0.0009632316650822759, "num_tokens": 107138400.0, "reward": 2.3299317359924316, "reward_std": 0.5315799117088318, "rewards/code_complexity_reward/mean": 0.9100586175918579, "rewards/code_complexity_reward/std": 0.135994553565979, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 634, "step_time": 49.36657038144767 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 431.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 124.564453125, "completions/mean_terminated_length": 124.564453125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23143530869856477, "epoch": 0.36182336182336183, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06224702671170235, "kl": 0.17908741044811904, "learning_rate": 4.0319138871711595e-06, "loss": 0.0008953216020017862, "num_tokens": 107272169.0, "reward": 2.2737796306610107, "reward_std": 0.525236189365387, "rewards/code_complexity_reward/mean": 0.9017578363418579, "rewards/code_complexity_reward/std": 0.14849813282489777, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 635, "step_time": 52.50675879046321 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 113.927734375, "completions/mean_terminated_length": 113.927734375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2368740935344249, "epoch": 0.3623931623931624, "frac_reward_zero_std": 0.53125, "grad_norm": 0.06681753695011139, "kl": 0.20016507618129253, "learning_rate": 4.027980062991484e-06, "loss": 0.0010009568650275469, "num_tokens": 107398316.0, "reward": 2.260791301727295, "reward_std": 0.452982097864151, "rewards/code_complexity_reward/mean": 0.9268555045127869, "rewards/code_complexity_reward/std": 0.08823377639055252, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 636, "step_time": 36.68148867134005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 112.865234375, "completions/mean_terminated_length": 112.865234375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23292569862678647, "epoch": 0.362962962962963, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05910129100084305, "kl": 0.17470720945857465, "learning_rate": 4.024040190237372e-06, "loss": 0.0008737160242162645, "num_tokens": 107526287.0, "reward": 2.2824220657348633, "reward_std": 0.5110459327697754, "rewards/code_complexity_reward/mean": 0.915820300579071, "rewards/code_complexity_reward/std": 0.13789623975753784, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 637, "step_time": 34.28967386484146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 119.556640625, "completions/mean_terminated_length": 118.78865051269531, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2347353172954172, "epoch": 0.36353276353276354, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05673488974571228, "kl": 0.17817010986618698, "learning_rate": 4.020094284504977e-06, "loss": 0.0008910587057471275, "num_tokens": 107654204.0, "reward": 2.343750238418579, "reward_std": 0.5199094414710999, "rewards/code_complexity_reward/mean": 0.9133788347244263, "rewards/code_complexity_reward/std": 0.11442747712135315, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 638, "step_time": 54.99392313696444 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 280.0, "completions/max_terminated_length": 280.0, "completions/mean_length": 110.697265625, "completions/mean_terminated_length": 110.697265625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24503993848338723, "epoch": 0.3641025641025641, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05574658140540123, "kl": 0.18571158964186907, "learning_rate": 4.016142361414335e-06, "loss": 0.0009285840205848217, "num_tokens": 107779273.0, "reward": 2.3709962368011475, "reward_std": 0.525088369846344, "rewards/code_complexity_reward/mean": 0.92333984375, "rewards/code_complexity_reward/std": 0.11856602877378464, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 639, "step_time": 33.37636773940176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 118.220703125, "completions/mean_terminated_length": 118.220703125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24304231954738498, "epoch": 0.3646723646723647, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06239503622055054, "kl": 0.16957242786884308, "learning_rate": 4.012184436609304e-06, "loss": 0.0008480445831082761, "num_tokens": 107910802.0, "reward": 2.2530274391174316, "reward_std": 0.4983295202255249, "rewards/code_complexity_reward/mean": 0.9108397960662842, "rewards/code_complexity_reward/std": 0.13789966702461243, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 640, "step_time": 38.11959115974605 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 114.58984375, "completions/mean_terminated_length": 114.58984375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24286733381450176, "epoch": 0.36524216524216524, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06454315036535263, "kl": 0.19137109292205423, "learning_rate": 4.008220525757498e-06, "loss": 0.0009567840024828911, "num_tokens": 108038536.0, "reward": 2.3143556118011475, "reward_std": 0.49911266565322876, "rewards/code_complexity_reward/mean": 0.91845703125, "rewards/code_complexity_reward/std": 0.11593305319547653, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 641, "step_time": 37.30924874357879 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 107.40625, "completions/mean_terminated_length": 107.40625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23727290821261704, "epoch": 0.3658119658119658, "frac_reward_zero_std": 0.5, "grad_norm": 0.06296800076961517, "kl": 0.18478197418153286, "learning_rate": 4.0042506445502276e-06, "loss": 0.0009238187922164798, "num_tokens": 108160752.0, "reward": 2.374072551727295, "reward_std": 0.4949924349784851, "rewards/code_complexity_reward/mean": 0.9297851324081421, "rewards/code_complexity_reward/std": 0.07780346274375916, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 642, "step_time": 44.62979501578957 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 264.0, "completions/max_terminated_length": 264.0, "completions/mean_length": 108.818359375, "completions/mean_terminated_length": 108.818359375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22974326252005994, "epoch": 0.3663817663817664, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05124944448471069, "kl": 0.17071579617913812, "learning_rate": 4.0002748087024365e-06, "loss": 0.0008540545823052526, "num_tokens": 108284235.0, "reward": 2.33056640625, "reward_std": 0.4901358187198639, "rewards/code_complexity_reward/mean": 0.9249023199081421, "rewards/code_complexity_reward/std": 0.09529022127389908, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 643, "step_time": 33.42429467570037 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 114.197265625, "completions/mean_terminated_length": 113.41878509521484, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24042661394923925, "epoch": 0.36695156695156694, "frac_reward_zero_std": 0.5625, "grad_norm": 0.04893922433257103, "kl": 0.1616590163903311, "learning_rate": 3.9962930339526425e-06, "loss": 0.0008083715802058578, "num_tokens": 108412840.0, "reward": 2.2828614711761475, "reward_std": 0.5039137601852417, "rewards/code_complexity_reward/mean": 0.91455078125, "rewards/code_complexity_reward/std": 0.12896619737148285, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 644, "step_time": 50.037367979064584 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 319.0, "completions/max_terminated_length": 319.0, "completions/mean_length": 112.541015625, "completions/mean_terminated_length": 112.541015625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24611827987246215, "epoch": 0.36752136752136755, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05609235540032387, "kl": 0.18588561133947223, "learning_rate": 3.99230533606287e-06, "loss": 0.0009296175558120012, "num_tokens": 108539253.0, "reward": 2.24658203125, "reward_std": 0.4327053129673004, "rewards/code_complexity_reward/mean": 0.9288085699081421, "rewards/code_complexity_reward/std": 0.07829610258340836, "rewards/code_execution_reward/mean": 0.220703125, "rewards/code_execution_reward/std": 0.4151262938976288, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 645, "step_time": 36.23244920466095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 494.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 118.33203125, "completions/mean_terminated_length": 118.33203125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23868477065116167, "epoch": 0.3680911680911681, "frac_reward_zero_std": 0.421875, "grad_norm": 0.05461610481142998, "kl": 0.17361726542003453, "learning_rate": 3.9883117308185925e-06, "loss": 0.0008682718616910279, "num_tokens": 108669439.0, "reward": 2.2303712368011475, "reward_std": 0.4991484582424164, "rewards/code_complexity_reward/mean": 0.90771484375, "rewards/code_complexity_reward/std": 0.14842306077480316, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 646, "step_time": 59.358145327307284 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 495.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 115.322265625, "completions/mean_terminated_length": 115.322265625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22891955357044935, "epoch": 0.36866096866096865, "frac_reward_zero_std": 0.515625, "grad_norm": 0.06296122074127197, "kl": 0.18202598730567843, "learning_rate": 3.984312234028668e-06, "loss": 0.0009102144977077842, "num_tokens": 108796692.0, "reward": 2.3023438453674316, "reward_std": 0.5258479118347168, "rewards/code_complexity_reward/mean": 0.909960925579071, "rewards/code_complexity_reward/std": 0.13950006663799286, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 647, "step_time": 56.14137730561197 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 117.734375, "completions/mean_terminated_length": 117.734375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24297252926044166, "epoch": 0.36923076923076925, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05477124825119972, "kl": 0.17790685198269784, "learning_rate": 3.980306861525273e-06, "loss": 0.0008895678911358118, "num_tokens": 108925364.0, "reward": 2.2541017532348633, "reward_std": 0.46021008491516113, "rewards/code_complexity_reward/mean": 0.917773425579071, "rewards/code_complexity_reward/std": 0.09303810447454453, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 648, "step_time": 37.42288285680115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 335.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 113.912109375, "completions/mean_terminated_length": 113.912109375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2429465684108436, "epoch": 0.3698005698005698, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05496738851070404, "kl": 0.1745616690022871, "learning_rate": 3.976295629163849e-06, "loss": 0.0008728957618586719, "num_tokens": 109052319.0, "reward": 2.3368165493011475, "reward_std": 0.4941408932209015, "rewards/code_complexity_reward/mean": 0.92431640625, "rewards/code_complexity_reward/std": 0.09017009288072586, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 649, "step_time": 36.763185404241085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 302.0, "completions/max_terminated_length": 302.0, "completions/mean_length": 110.28515625, "completions/mean_terminated_length": 110.28515625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.24246830539777875, "epoch": 0.37037037037037035, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05310526117682457, "kl": 0.1793299064738676, "learning_rate": 3.972278552823028e-06, "loss": 0.0008969479822553694, "num_tokens": 109175129.0, "reward": 2.3040528297424316, "reward_std": 0.4554620087146759, "rewards/code_complexity_reward/mean": 0.9339843392372131, "rewards/code_complexity_reward/std": 0.05175343528389931, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 650, "step_time": 34.57944609411061 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 450.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 119.478515625, "completions/mean_terminated_length": 119.478515625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24407339352183044, "epoch": 0.37094017094017095, "frac_reward_zero_std": 0.453125, "grad_norm": 0.057861726731061935, "kl": 0.2175727125722915, "learning_rate": 3.968255648404581e-06, "loss": 0.0010878713801503181, "num_tokens": 109305622.0, "reward": 2.244140625, "reward_std": 0.46613797545433044, "rewards/code_complexity_reward/mean": 0.9185546636581421, "rewards/code_complexity_reward/std": 0.11385204643011093, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 651, "step_time": 52.47399342712015 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 367.0, "completions/max_terminated_length": 367.0, "completions/mean_length": 115.3203125, "completions/mean_terminated_length": 115.3203125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23820790275931358, "epoch": 0.3715099715099715, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05665844306349754, "kl": 0.17236329207662493, "learning_rate": 3.9642269318333445e-06, "loss": 0.0008617307175882161, "num_tokens": 109432122.0, "reward": 2.303711175918579, "reward_std": 0.4748038947582245, "rewards/code_complexity_reward/mean": 0.92333984375, "rewards/code_complexity_reward/std": 0.08923011273145676, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 652, "step_time": 38.5756424497813 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 324.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 111.939453125, "completions/mean_terminated_length": 111.939453125, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.23375355103053153, "epoch": 0.37207977207977205, "frac_reward_zero_std": 0.421875, "grad_norm": 0.08248698711395264, "kl": 0.18770048499573022, "learning_rate": 3.960192419057168e-06, "loss": 0.0009387528989464045, "num_tokens": 109557587.0, "reward": 2.2698731422424316, "reward_std": 0.4901294410228729, "rewards/code_complexity_reward/mean": 0.9168945550918579, "rewards/code_complexity_reward/std": 0.12212535738945007, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 653, "step_time": 36.261675781570375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 500.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 115.078125, "completions/mean_terminated_length": 115.078125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2451459034346044, "epoch": 0.37264957264957266, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06841903924942017, "kl": 0.15829441114328802, "learning_rate": 3.956152126046843e-06, "loss": 0.0007913759909570217, "num_tokens": 109685867.0, "reward": 2.349170207977295, "reward_std": 0.5178710222244263, "rewards/code_complexity_reward/mean": 0.9134765863418579, "rewards/code_complexity_reward/std": 0.11166775971651077, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 654, "step_time": 58.25333908665925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 110.86328125, "completions/mean_terminated_length": 110.07827758789062, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23753261845558882, "epoch": 0.3732193732193732, "frac_reward_zero_std": 0.4375, "grad_norm": 0.059726860374212265, "kl": 0.19256487605161965, "learning_rate": 3.952106068796039e-06, "loss": 0.0009628646075725555, "num_tokens": 109810165.0, "reward": 2.3111817836761475, "reward_std": 0.5243397951126099, "rewards/code_complexity_reward/mean": 0.914257824420929, "rewards/code_complexity_reward/std": 0.13845610618591309, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 655, "step_time": 48.633241719566286 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 484.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 118.259765625, "completions/mean_terminated_length": 118.259765625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2433216804638505, "epoch": 0.3737891737891738, "frac_reward_zero_std": 0.4375, "grad_norm": 0.0660393163561821, "kl": 0.16294487030245364, "learning_rate": 3.948054263321249e-06, "loss": 0.0008149300701916218, "num_tokens": 109940514.0, "reward": 2.2770020961761475, "reward_std": 0.5032863020896912, "rewards/code_complexity_reward/mean": 0.910839855670929, "rewards/code_complexity_reward/std": 0.12959623336791992, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 656, "step_time": 56.48742446769029 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 115.46875, "completions/mean_terminated_length": 115.46875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24679984198883176, "epoch": 0.37435897435897436, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05606478452682495, "kl": 0.17858839873224497, "learning_rate": 3.94399672566172e-06, "loss": 0.0008930362528190017, "num_tokens": 110067634.0, "reward": 2.379931926727295, "reward_std": 0.5108283162117004, "rewards/code_complexity_reward/mean": 0.9208984375, "rewards/code_complexity_reward/std": 0.09738645702600479, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 657, "step_time": 37.20570822432637 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 114.537109375, "completions/mean_terminated_length": 114.537109375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2541255976539105, "epoch": 0.3749287749287749, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06348611414432526, "kl": 0.17786450462881476, "learning_rate": 3.939933471879384e-06, "loss": 0.0008894654456526041, "num_tokens": 110197373.0, "reward": 2.292773485183716, "reward_std": 0.499347984790802, "rewards/code_complexity_reward/mean": 0.918261706829071, "rewards/code_complexity_reward/std": 0.12059662491083145, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 658, "step_time": 37.521602891385555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 429.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 114.5390625, "completions/mean_terminated_length": 114.5390625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24936872767284513, "epoch": 0.3754985754985755, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05765112489461899, "kl": 0.17700482869986445, "learning_rate": 3.935864518058809e-06, "loss": 0.0008850643062032759, "num_tokens": 110325129.0, "reward": 2.2113771438598633, "reward_std": 0.48339220881462097, "rewards/code_complexity_reward/mean": 0.90576171875, "rewards/code_complexity_reward/std": 0.14392799139022827, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 659, "step_time": 50.66922311484814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 362.0, "completions/max_terminated_length": 362.0, "completions/mean_length": 120.09765625, "completions/mean_terminated_length": 120.09765625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24294399074278772, "epoch": 0.37606837606837606, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06374938040971756, "kl": 0.16697489493526518, "learning_rate": 3.9317898803071205e-06, "loss": 0.0008351161377504468, "num_tokens": 110455779.0, "reward": 2.293261766433716, "reward_std": 0.4931313991546631, "rewards/code_complexity_reward/mean": 0.9120116829872131, "rewards/code_complexity_reward/std": 0.11298847198486328, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 660, "step_time": 47.86786345113069 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 412.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 122.19140625, "completions/mean_terminated_length": 122.19140625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24332149815745652, "epoch": 0.3766381766381766, "frac_reward_zero_std": 0.53125, "grad_norm": 0.050441425293684006, "kl": 0.15563403512351215, "learning_rate": 3.9277095747539465e-06, "loss": 0.0007784847402945161, "num_tokens": 110588317.0, "reward": 2.320068597793579, "reward_std": 0.515089213848114, "rewards/code_complexity_reward/mean": 0.914843738079071, "rewards/code_complexity_reward/std": 0.12423690408468246, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 661, "step_time": 41.770223980769515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 435.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 119.361328125, "completions/mean_terminated_length": 119.361328125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23485787701793015, "epoch": 0.3772079772079772, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06290233135223389, "kl": 0.19011199416127056, "learning_rate": 3.923623617551352e-06, "loss": 0.0009503298788331449, "num_tokens": 110718438.0, "reward": 2.4071779251098633, "reward_std": 0.5383833646774292, "rewards/code_complexity_reward/mean": 0.91845703125, "rewards/code_complexity_reward/std": 0.11860310286283493, "rewards/code_execution_reward/mean": 0.396484375, "rewards/code_execution_reward/std": 0.4896455705165863, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 662, "step_time": 42.96699670702219 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 116.8671875, "completions/mean_terminated_length": 116.8671875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2360433132853359, "epoch": 0.37777777777777777, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05803109332919121, "kl": 0.16298837086651474, "learning_rate": 3.9195320248737725e-06, "loss": 0.0008150557987391949, "num_tokens": 110848946.0, "reward": 2.354053020477295, "reward_std": 0.5012941956520081, "rewards/code_complexity_reward/mean": 0.9214843511581421, "rewards/code_complexity_reward/std": 0.10163701325654984, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 663, "step_time": 38.4489661147818 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 399.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 120.095703125, "completions/mean_terminated_length": 120.095703125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24485201225616038, "epoch": 0.3783475783475784, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05853130295872688, "kl": 0.16437078814487904, "learning_rate": 3.915434812917953e-06, "loss": 0.000822270754724741, "num_tokens": 110979891.0, "reward": 2.224609375, "reward_std": 0.4538537561893463, "rewards/code_complexity_reward/mean": 0.9191405773162842, "rewards/code_complexity_reward/std": 0.10770167410373688, "rewards/code_execution_reward/mean": 0.212890625, "rewards/code_execution_reward/std": 0.409751296043396, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 664, "step_time": 58.921291516162455 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 117.6796875, "completions/mean_terminated_length": 116.90802001953125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24278567475266755, "epoch": 0.3789173789173789, "frac_reward_zero_std": 0.5, "grad_norm": 0.04862125590443611, "kl": 0.17058206826914102, "learning_rate": 3.911331997902882e-06, "loss": 0.0008529839105904102, "num_tokens": 111106127.0, "reward": 2.2978515625, "reward_std": 0.4919387102127075, "rewards/code_complexity_reward/mean": 0.915332019329071, "rewards/code_complexity_reward/std": 0.10582028329372406, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 665, "step_time": 48.2766019878909 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 301.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 110.771484375, "completions/mean_terminated_length": 110.771484375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24150505918078125, "epoch": 0.37948717948717947, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05489492416381836, "kl": 0.17338621255476028, "learning_rate": 3.907223596069729e-06, "loss": 0.0008671105024404824, "num_tokens": 111230738.0, "reward": 2.307910203933716, "reward_std": 0.4891892969608307, "rewards/code_complexity_reward/mean": 0.9227538704872131, "rewards/code_complexity_reward/std": 0.10526656359434128, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 666, "step_time": 40.10822342336178 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 394.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 118.896484375, "completions/mean_terminated_length": 118.896484375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23572024097666144, "epoch": 0.3800569800569801, "frac_reward_zero_std": 0.5625, "grad_norm": 0.048473846167325974, "kl": 0.1650019692024216, "learning_rate": 3.9031096236817775e-06, "loss": 0.000825306517072022, "num_tokens": 111362701.0, "reward": 2.2813477516174316, "reward_std": 0.5070044994354248, "rewards/code_complexity_reward/mean": 0.9108397960662842, "rewards/code_complexity_reward/std": 0.12876302003860474, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 667, "step_time": 41.06813834607601 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 490.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 118.638671875, "completions/mean_terminated_length": 118.638671875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24202741705812514, "epoch": 0.3806267806267806, "frac_reward_zero_std": 0.421875, "grad_norm": 0.0562620647251606, "kl": 0.16506955830845982, "learning_rate": 3.8989900970243635e-06, "loss": 0.0008253400446847081, "num_tokens": 111492220.0, "reward": 2.2938477993011475, "reward_std": 0.491817831993103, "rewards/code_complexity_reward/mean": 0.91650390625, "rewards/code_complexity_reward/std": 0.1125495433807373, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 668, "step_time": 53.49357948731631 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 309.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 113.74609375, "completions/mean_terminated_length": 113.74609375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2242802355904132, "epoch": 0.3811965811965812, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06544898450374603, "kl": 0.17415380117017776, "learning_rate": 3.89486503240481e-06, "loss": 0.000870672520250082, "num_tokens": 111618386.0, "reward": 2.341064453125, "reward_std": 0.5100453495979309, "rewards/code_complexity_reward/mean": 0.9170898199081421, "rewards/code_complexity_reward/std": 0.1049458310008049, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 669, "step_time": 43.182668005116284 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 381.0, "completions/max_terminated_length": 381.0, "completions/mean_length": 122.533203125, "completions/mean_terminated_length": 122.533203125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24100402439944446, "epoch": 0.3817663817663818, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06134599819779396, "kl": 0.174259940860793, "learning_rate": 3.89073444615236e-06, "loss": 0.000871487136464566, "num_tokens": 111752011.0, "reward": 2.3150880336761475, "reward_std": 0.5022022724151611, "rewards/code_complexity_reward/mean": 0.9090819954872131, "rewards/code_complexity_reward/std": 0.10654021799564362, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 670, "step_time": 45.093547088094056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 262.0, "completions/max_terminated_length": 262.0, "completions/mean_length": 107.986328125, "completions/mean_terminated_length": 107.986328125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24913778738118708, "epoch": 0.38233618233618233, "frac_reward_zero_std": 0.390625, "grad_norm": 0.1276504248380661, "kl": 0.15454800112638623, "learning_rate": 3.886598354618119e-06, "loss": 0.0007726794574409723, "num_tokens": 111874996.0, "reward": 2.2877931594848633, "reward_std": 0.49856439232826233, "rewards/code_complexity_reward/mean": 0.9167969226837158, "rewards/code_complexity_reward/std": 0.12001750618219376, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 671, "step_time": 39.829415320418775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 377.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 121.119140625, "completions/mean_terminated_length": 121.119140625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23869637097232044, "epoch": 0.38290598290598293, "frac_reward_zero_std": 0.375, "grad_norm": 0.06663624197244644, "kl": 0.17042225773911923, "learning_rate": 3.882456774174978e-06, "loss": 0.0008521160343661904, "num_tokens": 112011817.0, "reward": 2.285937786102295, "reward_std": 0.4809710383415222, "rewards/code_complexity_reward/mean": 0.9171874523162842, "rewards/code_complexity_reward/std": 0.1014450415968895, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 672, "step_time": 41.57492789532989 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 118.732421875, "completions/mean_terminated_length": 118.732421875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23440858954563737, "epoch": 0.3834757834757835, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06778474152088165, "kl": 0.1566198015352711, "learning_rate": 3.8783097212175644e-06, "loss": 0.0007831896655261517, "num_tokens": 112141856.0, "reward": 2.275146484375, "reward_std": 0.48817601799964905, "rewards/code_complexity_reward/mean": 0.9166015386581421, "rewards/code_complexity_reward/std": 0.117324098944664, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 673, "step_time": 48.63230613991618 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 113.0703125, "completions/mean_terminated_length": 113.0703125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24305410566739738, "epoch": 0.38404558404558403, "frac_reward_zero_std": 0.5, "grad_norm": 0.06774785369634628, "kl": 0.18152852973435074, "learning_rate": 3.874157212162163e-06, "loss": 0.0009078475995920599, "num_tokens": 112270060.0, "reward": 2.3187501430511475, "reward_std": 0.494448184967041, "rewards/code_complexity_reward/mean": 0.915234386920929, "rewards/code_complexity_reward/std": 0.0880998745560646, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 674, "step_time": 41.29955590888858 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 420.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 122.162109375, "completions/mean_terminated_length": 122.162109375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.227489004842937, "epoch": 0.38461538461538464, "frac_reward_zero_std": 0.453125, "grad_norm": 0.060258977115154266, "kl": 0.16194647655356675, "learning_rate": 3.869999263446658e-06, "loss": 0.0008097368408925831, "num_tokens": 112402431.0, "reward": 2.3561525344848633, "reward_std": 0.5027115941047668, "rewards/code_complexity_reward/mean": 0.9140625, "rewards/code_complexity_reward/std": 0.09107694774866104, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 675, "step_time": 46.25463405158371 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 411.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 113.5390625, "completions/mean_terminated_length": 113.5390625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23945276509039104, "epoch": 0.3851851851851852, "frac_reward_zero_std": 0.5, "grad_norm": 0.06349361687898636, "kl": 0.18696836999151856, "learning_rate": 3.865835891530468e-06, "loss": 0.0009345933794975281, "num_tokens": 112527059.0, "reward": 2.3441896438598633, "reward_std": 0.5140841007232666, "rewards/code_complexity_reward/mean": 0.914355456829071, "rewards/code_complexity_reward/std": 0.10796993970870972, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 676, "step_time": 44.22952814679593 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 472.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 117.92578125, "completions/mean_terminated_length": 117.92578125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22910130862146616, "epoch": 0.38575498575498574, "frac_reward_zero_std": 0.53125, "grad_norm": 0.06042501702904701, "kl": 0.1750644997227937, "learning_rate": 3.861667112894479e-06, "loss": 0.0008752202847972512, "num_tokens": 112656445.0, "reward": 2.316406488418579, "reward_std": 0.4864649176597595, "rewards/code_complexity_reward/mean": 0.9253906011581421, "rewards/code_complexity_reward/std": 0.0979728177189827, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 677, "step_time": 48.427821746096015 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 412.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 119.814453125, "completions/mean_terminated_length": 119.814453125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2284725229255855, "epoch": 0.38632478632478634, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05813857167959213, "kl": 0.16198152489960194, "learning_rate": 3.857492944040978e-06, "loss": 0.0008102008141577244, "num_tokens": 112787598.0, "reward": 2.2960939407348633, "reward_std": 0.4921242296695709, "rewards/code_complexity_reward/mean": 0.911914050579071, "rewards/code_complexity_reward/std": 0.11210775375366211, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 678, "step_time": 53.39000390376896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 115.26953125, "completions/mean_terminated_length": 115.26953125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2298676185309887, "epoch": 0.3868945868945869, "frac_reward_zero_std": 0.34375, "grad_norm": 0.06741980463266373, "kl": 0.1637957957573235, "learning_rate": 3.853313401493594e-06, "loss": 0.000819120672531426, "num_tokens": 112914176.0, "reward": 2.3681640625, "reward_std": 0.5316611528396606, "rewards/code_complexity_reward/mean": 0.9156249761581421, "rewards/code_complexity_reward/std": 0.12167294323444366, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 679, "step_time": 41.55080367345363 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 111.33984375, "completions/mean_terminated_length": 111.33984375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23201232799328864, "epoch": 0.38746438746438744, "frac_reward_zero_std": 0.46875, "grad_norm": 0.057085417211055756, "kl": 0.16548554704058915, "learning_rate": 3.8491285017972216e-06, "loss": 0.0008276647422462702, "num_tokens": 113037982.0, "reward": 2.3849611282348633, "reward_std": 0.5216880440711975, "rewards/code_complexity_reward/mean": 0.9250000715255737, "rewards/code_complexity_reward/std": 0.10745226591825485, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 680, "step_time": 40.70187960751355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 309.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 116.576171875, "completions/mean_terminated_length": 116.576171875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23821350350044668, "epoch": 0.38803418803418804, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05923786014318466, "kl": 0.16688714397605509, "learning_rate": 3.844938261517967e-06, "loss": 0.000834582606330514, "num_tokens": 113167421.0, "reward": 2.304931640625, "reward_std": 0.502436637878418, "rewards/code_complexity_reward/mean": 0.9141601324081421, "rewards/code_complexity_reward/std": 0.12033741176128387, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 681, "step_time": 44.28197694290429 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 113.0859375, "completions/mean_terminated_length": 112.30528259277344, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2304919755551964, "epoch": 0.3886039886039886, "frac_reward_zero_std": 0.4375, "grad_norm": 0.07684151083230972, "kl": 0.16202034312300384, "learning_rate": 3.840742697243075e-06, "loss": 0.0008098774706013501, "num_tokens": 113296281.0, "reward": 2.3043458461761475, "reward_std": 0.5027852058410645, "rewards/code_complexity_reward/mean": 0.9191405773162842, "rewards/code_complexity_reward/std": 0.1163487359881401, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 682, "step_time": 49.406416771933436 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 315.0, "completions/mean_length": 115.119140625, "completions/mean_terminated_length": 114.34246826171875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24463514634408057, "epoch": 0.3891737891737892, "frac_reward_zero_std": 0.375, "grad_norm": 0.06865151226520538, "kl": 0.18371348024811596, "learning_rate": 3.836541825580867e-06, "loss": 0.000918708392418921, "num_tokens": 113421238.0, "reward": 2.3165528774261475, "reward_std": 0.5359216332435608, "rewards/code_complexity_reward/mean": 0.91015625, "rewards/code_complexity_reward/std": 0.1490490883588791, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 683, "step_time": 65.18858028296381 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 362.0, "completions/max_terminated_length": 362.0, "completions/mean_length": 113.48828125, "completions/mean_terminated_length": 113.48828125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23655944387428463, "epoch": 0.38974358974358975, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06704092025756836, "kl": 0.1795344491256401, "learning_rate": 3.832335663160672e-06, "loss": 0.0008975530508905649, "num_tokens": 113548408.0, "reward": 2.3478517532348633, "reward_std": 0.49820151925086975, "rewards/code_complexity_reward/mean": 0.9207030534744263, "rewards/code_complexity_reward/std": 0.09907148778438568, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 684, "step_time": 39.638719839043915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 297.0, "completions/max_terminated_length": 297.0, "completions/mean_length": 110.19140625, "completions/mean_terminated_length": 110.19140625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23803682369180024, "epoch": 0.3903133903133903, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06851726770401001, "kl": 0.1723640572745353, "learning_rate": 3.828124226632765e-06, "loss": 0.0008618078427389264, "num_tokens": 113673578.0, "reward": 2.380664348602295, "reward_std": 0.5363747477531433, "rewards/code_complexity_reward/mean": 0.9164062738418579, "rewards/code_complexity_reward/std": 0.12757860124111176, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 685, "step_time": 45.740091861225665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 396.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 118.244140625, "completions/mean_terminated_length": 118.244140625, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.22671299171634018, "epoch": 0.3908831908831909, "frac_reward_zero_std": 0.375, "grad_norm": 0.07325198501348495, "kl": 0.17285684100352228, "learning_rate": 3.823907532668298e-06, "loss": 0.000864300993271172, "num_tokens": 113800855.0, "reward": 2.3668458461761475, "reward_std": 0.5163384079933167, "rewards/code_complexity_reward/mean": 0.915234386920929, "rewards/code_complexity_reward/std": 0.10917921364307404, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 686, "step_time": 48.7450690343976 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 116.193359375, "completions/mean_terminated_length": 116.193359375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23577139154076576, "epoch": 0.39145299145299145, "frac_reward_zero_std": 0.515625, "grad_norm": 0.060606569051742554, "kl": 0.17584743187762797, "learning_rate": 3.819685597959233e-06, "loss": 0.0008796093752607703, "num_tokens": 113928522.0, "reward": 2.2877931594848633, "reward_std": 0.4692239761352539, "rewards/code_complexity_reward/mean": 0.9248046875, "rewards/code_complexity_reward/std": 0.09498151391744614, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 687, "step_time": 41.44029765017331 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 107.412109375, "completions/mean_terminated_length": 107.412109375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2332377142738551, "epoch": 0.392022792022792, "frac_reward_zero_std": 0.4375, "grad_norm": 0.0587838850915432, "kl": 0.17514221731107682, "learning_rate": 3.815458439218279e-06, "loss": 0.0008756866445764899, "num_tokens": 114054789.0, "reward": 2.363086223602295, "reward_std": 0.5376660227775574, "rewards/code_complexity_reward/mean": 0.9134765267372131, "rewards/code_complexity_reward/std": 0.13437549769878387, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 688, "step_time": 35.55883902031928 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 113.02734375, "completions/mean_terminated_length": 113.02734375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23392724827863276, "epoch": 0.3925925925925926, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06161997839808464, "kl": 0.17407996754627675, "learning_rate": 3.8112260731788265e-06, "loss": 0.0008705899817869067, "num_tokens": 114180083.0, "reward": 2.3100099563598633, "reward_std": 0.4835439622402191, "rewards/code_complexity_reward/mean": 0.927050769329071, "rewards/code_complexity_reward/std": 0.09392346441745758, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 689, "step_time": 44.86054327059537 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 478.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 125.9765625, "completions/mean_terminated_length": 125.9765625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22897805296815932, "epoch": 0.39316239316239315, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06307430565357208, "kl": 0.15642370213754475, "learning_rate": 3.8069885165948763e-06, "loss": 0.0007821710896678269, "num_tokens": 114314071.0, "reward": 2.2894532680511475, "reward_std": 0.5158886313438416, "rewards/code_complexity_reward/mean": 0.8994140625, "rewards/code_complexity_reward/std": 0.13595227897167206, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 690, "step_time": 52.818407187238336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 111.658203125, "completions/mean_terminated_length": 111.658203125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2379490127786994, "epoch": 0.39373219373219376, "frac_reward_zero_std": 0.359375, "grad_norm": 0.058050792664289474, "kl": 0.16300404618959874, "learning_rate": 3.8027457862409765e-06, "loss": 0.0008152775117196143, "num_tokens": 114436832.0, "reward": 2.354541063308716, "reward_std": 0.5024793148040771, "rewards/code_complexity_reward/mean": 0.9222656488418579, "rewards/code_complexity_reward/std": 0.09612015634775162, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 691, "step_time": 53.73725020699203 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 265.0, "completions/max_terminated_length": 265.0, "completions/mean_length": 112.005859375, "completions/mean_terminated_length": 112.005859375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24552522995509207, "epoch": 0.3943019943019943, "frac_reward_zero_std": 0.5625, "grad_norm": 0.057175714522600174, "kl": 0.17353504581842571, "learning_rate": 3.798497898912159e-06, "loss": 0.0008676411234773695, "num_tokens": 114565955.0, "reward": 2.2940919399261475, "reward_std": 0.49386700987815857, "rewards/code_complexity_reward/mean": 0.924023449420929, "rewards/code_complexity_reward/std": 0.11230123788118362, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 692, "step_time": 35.640116839669645 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 305.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 119.5078125, "completions/mean_terminated_length": 119.5078125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2423981644678861, "epoch": 0.39487179487179486, "frac_reward_zero_std": 0.625, "grad_norm": 0.04984462633728981, "kl": 0.1688721920363605, "learning_rate": 3.7942448714238666e-06, "loss": 0.0008443054975941777, "num_tokens": 114697055.0, "reward": 2.261035442352295, "reward_std": 0.48962852358818054, "rewards/code_complexity_reward/mean": 0.9188476800918579, "rewards/code_complexity_reward/std": 0.1271440088748932, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 693, "step_time": 42.857873206958175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 471.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 114.29296875, "completions/mean_terminated_length": 114.29296875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2274797644931823, "epoch": 0.39544159544159546, "frac_reward_zero_std": 0.53125, "grad_norm": 0.060275666415691376, "kl": 0.18009670125320554, "learning_rate": 3.789986720611889e-06, "loss": 0.0009001196594908834, "num_tokens": 114821717.0, "reward": 2.3467774391174316, "reward_std": 0.48381271958351135, "rewards/code_complexity_reward/mean": 0.9245116710662842, "rewards/code_complexity_reward/std": 0.0681726485490799, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 694, "step_time": 55.836214563809335 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 125.501953125, "completions/mean_terminated_length": 123.2239761352539, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2368286473210901, "epoch": 0.396011396011396, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06782995909452438, "kl": 0.1868578534340486, "learning_rate": 3.785723463332299e-06, "loss": 0.0009340765536762774, "num_tokens": 114955286.0, "reward": 2.2870118618011475, "reward_std": 0.5294111371040344, "rewards/code_complexity_reward/mean": 0.9020507335662842, "rewards/code_complexity_reward/std": 0.14655433595180511, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 695, "step_time": 56.956250650808215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 114.220703125, "completions/mean_terminated_length": 114.220703125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2331822810228914, "epoch": 0.39658119658119656, "frac_reward_zero_std": 0.296875, "grad_norm": 0.0669047012925148, "kl": 0.17191121552605182, "learning_rate": 3.7814551164613843e-06, "loss": 0.0008593917591497302, "num_tokens": 115079359.0, "reward": 2.2999024391174316, "reward_std": 0.5185455083847046, "rewards/code_complexity_reward/mean": 0.913281261920929, "rewards/code_complexity_reward/std": 0.13435856997966766, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 696, "step_time": 44.24775052908808 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 120.86328125, "completions/mean_terminated_length": 120.86328125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23618002654984593, "epoch": 0.39715099715099716, "frac_reward_zero_std": 0.46875, "grad_norm": 0.055493276566267014, "kl": 0.16204775660298765, "learning_rate": 3.777181696895578e-06, "loss": 0.0008105994202196598, "num_tokens": 115208993.0, "reward": 2.2789063453674316, "reward_std": 0.49116894602775574, "rewards/code_complexity_reward/mean": 0.9162108898162842, "rewards/code_complexity_reward/std": 0.1163741946220398, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 697, "step_time": 52.25361196696758 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 324.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 116.388671875, "completions/mean_terminated_length": 116.388671875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23722506244666874, "epoch": 0.3977207977207977, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06827378273010254, "kl": 0.1822912273928523, "learning_rate": 3.772903221551394e-06, "loss": 0.0009117099107243121, "num_tokens": 115335560.0, "reward": 2.381884813308716, "reward_std": 0.49947646260261536, "rewards/code_complexity_reward/mean": 0.9297851324081421, "rewards/code_complexity_reward/std": 0.07685445994138718, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 698, "step_time": 43.96596196759492 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 330.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 119.150390625, "completions/mean_terminated_length": 119.150390625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2290241066366434, "epoch": 0.39829059829059826, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05694069713354111, "kl": 0.17298267292790115, "learning_rate": 3.7686197073653586e-06, "loss": 0.0008651025709696114, "num_tokens": 115464493.0, "reward": 2.263916015625, "reward_std": 0.4975408911705017, "rewards/code_complexity_reward/mean": 0.914355456829071, "rewards/code_complexity_reward/std": 0.13346201181411743, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 699, "step_time": 43.83567818719894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 428.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 118.37890625, "completions/mean_terminated_length": 118.37890625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23280910612083972, "epoch": 0.39886039886039887, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06303789466619492, "kl": 0.1621873287949711, "learning_rate": 3.7643311712939477e-06, "loss": 0.0008109075715765357, "num_tokens": 115594495.0, "reward": 2.2579591274261475, "reward_std": 0.47129586338996887, "rewards/code_complexity_reward/mean": 0.9132812023162842, "rewards/code_complexity_reward/std": 0.10749493539333344, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 700, "step_time": 51.02748606167734 }, { "epoch": 0.39886039886039887, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.00125, "eval_completions/max_length": 167.05, "eval_completions/max_terminated_length": 166.98, "eval_completions/mean_length": 118.79, "eval_completions/mean_terminated_length": 118.65017883300781, "eval_completions/min_length": 86.3, "eval_completions/min_terminated_length": 86.3, "eval_entropy": 0.2335691875219345, "eval_frac_reward_zero_std": 0.48, "eval_kl": 0.15798569057136774, "eval_loss": 0.0007887644460424781, "eval_num_tokens": 115594495.0, "eval_reward": 2.2676251232624054, "eval_reward_std": 0.22255863316357136, "eval_rewards/code_complexity_reward/mean": 0.9061874839663505, "eval_rewards/code_complexity_reward/std": 0.060024201069027186, "eval_rewards/code_execution_reward/mean": 0.275, "eval_rewards/code_execution_reward/std": 0.1482235845923424, "eval_rewards/code_syntax_reward/mean": 0.486875, "eval_rewards/code_syntax_reward/std": 0.029061699360609053, "eval_rewards/reasoning_present_reward_func/mean": 0.0998750015348196, "eval_rewards/reasoning_present_reward_func/std": 0.000353553406894207, "eval_rewards/xmlcount_reward_func/mean": 0.4996875, "eval_rewards/xmlcount_reward_func/std": 0.000883883461356163, "eval_runtime": 766.0967, "eval_samples_per_second": 0.131, "eval_steps_per_second": 0.017, "step": 700 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 279.0, "completions/max_terminated_length": 279.0, "completions/mean_length": 110.611328125, "completions/mean_terminated_length": 110.611328125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23640405223704875, "epoch": 0.3994301994301994, "frac_reward_zero_std": 0.46875, "grad_norm": 0.0632546916604042, "kl": 0.1701169062871486, "learning_rate": 3.760037630313515e-06, "loss": 0.0008505244622938335, "num_tokens": 115721456.0, "reward": 2.392334222793579, "reward_std": 0.5090434551239014, "rewards/code_complexity_reward/mean": 0.9253906011581421, "rewards/code_complexity_reward/std": 0.08724890649318695, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 701, "step_time": 36.61891868058592 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 115.62890625, "completions/mean_terminated_length": 115.62890625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24270047014579177, "epoch": 0.4, "frac_reward_zero_std": 0.4375, "grad_norm": 0.0582696907222271, "kl": 0.16208100400399417, "learning_rate": 3.755739101420225e-06, "loss": 0.0008104760781861842, "num_tokens": 115849122.0, "reward": 2.2579593658447266, "reward_std": 0.46244093775749207, "rewards/code_complexity_reward/mean": 0.9240233898162842, "rewards/code_complexity_reward/std": 0.10542535036802292, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 702, "step_time": 35.435103121213615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 324.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 111.982421875, "completions/mean_terminated_length": 111.982421875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23700660024769604, "epoch": 0.40056980056980057, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05044201761484146, "kl": 0.17311206413432956, "learning_rate": 3.751435601629988e-06, "loss": 0.0008656866266392171, "num_tokens": 115971873.0, "reward": 2.303955078125, "reward_std": 0.4814872741699219, "rewards/code_complexity_reward/mean": 0.922167956829071, "rewards/code_complexity_reward/std": 0.09808233380317688, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 703, "step_time": 35.88959005754441 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 494.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 117.62109375, "completions/mean_terminated_length": 117.62109375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2383694564923644, "epoch": 0.4011396011396011, "frac_reward_zero_std": 0.546875, "grad_norm": 0.0515897199511528, "kl": 0.1599585881922394, "learning_rate": 3.7471271479783933e-06, "loss": 0.000799915986135602, "num_tokens": 116102375.0, "reward": 2.298633098602295, "reward_std": 0.4697379171848297, "rewards/code_complexity_reward/mean": 0.9281250238418579, "rewards/code_complexity_reward/std": 0.08108848333358765, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 704, "step_time": 75.6774734063074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 230.0, "completions/max_terminated_length": 230.0, "completions/mean_length": 107.67578125, "completions/mean_terminated_length": 107.67578125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2357315954286605, "epoch": 0.4017094017094017, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06834089756011963, "kl": 0.16589025291614234, "learning_rate": 3.7428137575206375e-06, "loss": 0.0008296574233099818, "num_tokens": 116225313.0, "reward": 2.356445550918579, "reward_std": 0.49475473165512085, "rewards/code_complexity_reward/mean": 0.9322265982627869, "rewards/code_complexity_reward/std": 0.08627156168222427, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 705, "step_time": 30.588569400832057 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 382.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 113.818359375, "completions/mean_terminated_length": 113.818359375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23017433565109968, "epoch": 0.4022792022792023, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05445219948887825, "kl": 0.18434874701779336, "learning_rate": 3.7384954473314626e-06, "loss": 0.0009216598700731993, "num_tokens": 116349980.0, "reward": 2.3768556118011475, "reward_std": 0.5196401476860046, "rewards/code_complexity_reward/mean": 0.91943359375, "rewards/code_complexity_reward/std": 0.10657572746276855, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 706, "step_time": 39.214836897328496 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 114.640625, "completions/mean_terminated_length": 114.640625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2264320496469736, "epoch": 0.4028490028490028, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06418747454881668, "kl": 0.16625475697219372, "learning_rate": 3.7341722345050837e-06, "loss": 0.00083155557513237, "num_tokens": 116476420.0, "reward": 2.3956055641174316, "reward_std": 0.5140636563301086, "rewards/code_complexity_reward/mean": 0.922558605670929, "rewards/code_complexity_reward/std": 0.09073463827371597, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 707, "step_time": 47.83978483360261 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 416.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 113.97265625, "completions/mean_terminated_length": 113.97265625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23769161547534168, "epoch": 0.40341880341880343, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05738550052046776, "kl": 0.1686255328822881, "learning_rate": 3.729844136155123e-06, "loss": 0.0008430968737229705, "num_tokens": 116603158.0, "reward": 2.3392090797424316, "reward_std": 0.48067042231559753, "rewards/code_complexity_reward/mean": 0.928906261920929, "rewards/code_complexity_reward/std": 0.06755714118480682, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 708, "step_time": 49.21909289248288 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 123.994140625, "completions/mean_terminated_length": 123.994140625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2438400185201317, "epoch": 0.403988603988604, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05228610709309578, "kl": 0.1670793378725648, "learning_rate": 3.7255111694145433e-06, "loss": 0.0008353212615475059, "num_tokens": 116736603.0, "reward": 2.2716798782348633, "reward_std": 0.46624431014060974, "rewards/code_complexity_reward/mean": 0.923535168170929, "rewards/code_complexity_reward/std": 0.09974430501461029, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 709, "step_time": 39.601037150248885 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 117.087890625, "completions/mean_terminated_length": 117.087890625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23455155896954238, "epoch": 0.4045584045584046, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06947251409292221, "kl": 0.17743457341566682, "learning_rate": 3.7211733514355795e-06, "loss": 0.0008874499471858144, "num_tokens": 116860840.0, "reward": 2.320605754852295, "reward_std": 0.5016023516654968, "rewards/code_complexity_reward/mean": 0.9178711175918579, "rewards/code_complexity_reward/std": 0.10776123404502869, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 710, "step_time": 50.766481664031744 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 295.0, "completions/max_terminated_length": 295.0, "completions/mean_length": 114.89453125, "completions/mean_terminated_length": 114.89453125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24546696431934834, "epoch": 0.40512820512820513, "frac_reward_zero_std": 0.4375, "grad_norm": 0.062289733439683914, "kl": 0.1692392585100606, "learning_rate": 3.7168306993896697e-06, "loss": 0.0008462097030133009, "num_tokens": 116987634.0, "reward": 2.4311037063598633, "reward_std": 0.5261048674583435, "rewards/code_complexity_reward/mean": 0.924121081829071, "rewards/code_complexity_reward/std": 0.09620595723390579, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 711, "step_time": 43.432843551039696 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 373.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 122.04296875, "completions/mean_terminated_length": 122.04296875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2464156267233193, "epoch": 0.4056980056980057, "frac_reward_zero_std": 0.421875, "grad_norm": 0.059002432972192764, "kl": 0.1900366669287905, "learning_rate": 3.7124832304673864e-06, "loss": 0.0009502447210252285, "num_tokens": 117117688.0, "reward": 2.2850587368011475, "reward_std": 0.5337405800819397, "rewards/code_complexity_reward/mean": 0.90478515625, "rewards/code_complexity_reward/std": 0.1539802998304367, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 712, "step_time": 41.29827772453427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 117.0703125, "completions/mean_terminated_length": 117.0703125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24875785107724369, "epoch": 0.4062678062678063, "frac_reward_zero_std": 0.375, "grad_norm": 0.06947971880435944, "kl": 0.1688207247061655, "learning_rate": 3.7081309618783724e-06, "loss": 0.0008443070109933615, "num_tokens": 117247916.0, "reward": 2.274951457977295, "reward_std": 0.4872656762599945, "rewards/code_complexity_reward/mean": 0.91650390625, "rewards/code_complexity_reward/std": 0.11132577806711197, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 713, "step_time": 42.410633904859424 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 309.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 122.953125, "completions/mean_terminated_length": 122.953125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22826417488977313, "epoch": 0.40683760683760684, "frac_reward_zero_std": 0.515625, "grad_norm": 0.054096437990665436, "kl": 0.1569164857501164, "learning_rate": 3.703773910851269e-06, "loss": 0.000784490373916924, "num_tokens": 117384436.0, "reward": 2.3388185501098633, "reward_std": 0.5002232789993286, "rewards/code_complexity_reward/mean": 0.91796875, "rewards/code_complexity_reward/std": 0.10722151398658752, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 714, "step_time": 43.83051173016429 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 123.0703125, "completions/mean_terminated_length": 123.0703125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23558189254254103, "epoch": 0.4074074074074074, "frac_reward_zero_std": 0.375, "grad_norm": 0.06496562063694, "kl": 0.1749324476113543, "learning_rate": 3.699412094633648e-06, "loss": 0.000874960795044899, "num_tokens": 117515168.0, "reward": 2.2818849086761475, "reward_std": 0.498417466878891, "rewards/code_complexity_reward/mean": 0.915722668170929, "rewards/code_complexity_reward/std": 0.12422702461481094, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 715, "step_time": 48.572880355641246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 116.119140625, "completions/mean_terminated_length": 116.119140625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22934174980036914, "epoch": 0.407977207977208, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06974727660417557, "kl": 0.15546481939963996, "learning_rate": 3.6950455304919465e-06, "loss": 0.0007774234982207417, "num_tokens": 117643845.0, "reward": 2.3290040493011475, "reward_std": 0.4675304591655731, "rewards/code_complexity_reward/mean": 0.93505859375, "rewards/code_complexity_reward/std": 0.0548650398850441, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 716, "step_time": 38.42190285585821 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 120.716796875, "completions/mean_terminated_length": 120.716796875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.239375727949664, "epoch": 0.40854700854700854, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05564763396978378, "kl": 0.16564976563677192, "learning_rate": 3.6906742357113958e-06, "loss": 0.0008283763309009373, "num_tokens": 117775532.0, "reward": 2.2196290493011475, "reward_std": 0.44854673743247986, "rewards/code_complexity_reward/mean": 0.91455078125, "rewards/code_complexity_reward/std": 0.11449793726205826, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 717, "step_time": 50.692036469466984 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 115.22265625, "completions/mean_terminated_length": 113.66667175292969, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23858179431408644, "epoch": 0.40911680911680914, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06824681907892227, "kl": 0.1691910884110257, "learning_rate": 3.6862982275959523e-06, "loss": 0.000845966802444309, "num_tokens": 117906910.0, "reward": 2.283008098602295, "reward_std": 0.5161523818969727, "rewards/code_complexity_reward/mean": 0.9154297113418579, "rewards/code_complexity_reward/std": 0.1397026926279068, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 718, "step_time": 55.80089107248932 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 419.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 118.455078125, "completions/mean_terminated_length": 118.455078125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23488675523549318, "epoch": 0.4096866096866097, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05550217628479004, "kl": 0.15227277716621757, "learning_rate": 3.6819175234682318e-06, "loss": 0.0007613583584316075, "num_tokens": 118038903.0, "reward": 2.3778321743011475, "reward_std": 0.514840304851532, "rewards/code_complexity_reward/mean": 0.92333984375, "rewards/code_complexity_reward/std": 0.10192462056875229, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 719, "step_time": 61.19685006979853 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 311.0, "completions/mean_length": 120.841796875, "completions/mean_terminated_length": 120.0763168334961, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23373299511149526, "epoch": 0.41025641025641024, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05622277781367302, "kl": 0.16350006533320993, "learning_rate": 3.6775321406694386e-06, "loss": 0.0008180409204214811, "num_tokens": 118167334.0, "reward": 2.3263185024261475, "reward_std": 0.5526947975158691, "rewards/code_complexity_reward/mean": 0.9082030653953552, "rewards/code_complexity_reward/std": 0.15587356686592102, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 720, "step_time": 49.219649977982044 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 428.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 129.19140625, "completions/mean_terminated_length": 129.19140625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.22546732751652598, "epoch": 0.41082621082621085, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06382560729980469, "kl": 0.15674344799481332, "learning_rate": 3.673142096559299e-06, "loss": 0.0007838350720703602, "num_tokens": 118302472.0, "reward": 2.312549114227295, "reward_std": 0.49982383847236633, "rewards/code_complexity_reward/mean": 0.9173828363418579, "rewards/code_complexity_reward/std": 0.10355985164642334, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 721, "step_time": 52.497939709573984 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 115.873046875, "completions/mean_terminated_length": 115.873046875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.22952988068573177, "epoch": 0.4113960113960114, "frac_reward_zero_std": 0.375, "grad_norm": 0.06342123448848724, "kl": 0.1729612039634958, "learning_rate": 3.668747408515989e-06, "loss": 0.0008649643277749419, "num_tokens": 118430351.0, "reward": 2.260205030441284, "reward_std": 0.45192521810531616, "rewards/code_complexity_reward/mean": 0.92626953125, "rewards/code_complexity_reward/std": 0.08846566081047058, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 722, "step_time": 36.95393682271242 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 381.0, "completions/max_terminated_length": 381.0, "completions/mean_length": 118.271484375, "completions/mean_terminated_length": 118.271484375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22973206685855985, "epoch": 0.41196581196581195, "frac_reward_zero_std": 0.53125, "grad_norm": 0.049340762197971344, "kl": 0.17255137977190316, "learning_rate": 3.6643480939360706e-06, "loss": 0.0008630963275209069, "num_tokens": 118559106.0, "reward": 2.2243165969848633, "reward_std": 0.4894878566265106, "rewards/code_complexity_reward/mean": 0.9114258289337158, "rewards/code_complexity_reward/std": 0.14625568687915802, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 723, "step_time": 48.29201743006706 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 428.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 119.375, "completions/mean_terminated_length": 119.375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23073260090313852, "epoch": 0.41253561253561255, "frac_reward_zero_std": 0.578125, "grad_norm": 0.04803076386451721, "kl": 0.15940360794775188, "learning_rate": 3.659944170234418e-06, "loss": 0.0007971449522301555, "num_tokens": 118690690.0, "reward": 2.2439942359924316, "reward_std": 0.46572789549827576, "rewards/code_complexity_reward/mean": 0.920605480670929, "rewards/code_complexity_reward/std": 0.11649741232395172, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 724, "step_time": 49.59635037742555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 259.0, "completions/max_terminated_length": 259.0, "completions/mean_length": 115.130859375, "completions/mean_terminated_length": 115.130859375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23091626190580428, "epoch": 0.4131054131054131, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06392322480678558, "kl": 0.18389429268427193, "learning_rate": 3.6555356548441513e-06, "loss": 0.0009197190520353615, "num_tokens": 118819357.0, "reward": 2.385302782058716, "reward_std": 0.5132324695587158, "rewards/code_complexity_reward/mean": 0.9208007454872131, "rewards/code_complexity_reward/std": 0.088264100253582, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 725, "step_time": 32.71997018624097 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 488.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 117.1328125, "completions/mean_terminated_length": 117.1328125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2417547171935439, "epoch": 0.41367521367521365, "frac_reward_zero_std": 0.5, "grad_norm": 0.06344756484031677, "kl": 0.17715268989559263, "learning_rate": 3.6511225652165674e-06, "loss": 0.0008860492962412536, "num_tokens": 118949209.0, "reward": 2.211474895477295, "reward_std": 0.44190526008605957, "rewards/code_complexity_reward/mean": 0.9166015982627869, "rewards/code_complexity_reward/std": 0.10875461250543594, "rewards/code_execution_reward/mean": 0.201171875, "rewards/code_execution_reward/std": 0.4012683033943176, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 726, "step_time": 49.841635377146304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 399.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 118.447265625, "completions/mean_terminated_length": 118.447265625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23642052034847438, "epoch": 0.41424501424501425, "frac_reward_zero_std": 0.46875, "grad_norm": 0.0606612004339695, "kl": 0.16554132814053446, "learning_rate": 3.6467049188210714e-06, "loss": 0.0008277344750240445, "num_tokens": 119082318.0, "reward": 2.3031251430511475, "reward_std": 0.5006062984466553, "rewards/code_complexity_reward/mean": 0.9130859375, "rewards/code_complexity_reward/std": 0.11507951468229294, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 727, "step_time": 57.779678410850465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 329.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 113.9375, "completions/mean_terminated_length": 113.9375, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.23115502623841166, "epoch": 0.4148148148148148, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06444830447435379, "kl": 0.16650866239797324, "learning_rate": 3.6422827331451037e-06, "loss": 0.0008324606460519135, "num_tokens": 119207982.0, "reward": 2.350390911102295, "reward_std": 0.48354512453079224, "rewards/code_complexity_reward/mean": 0.9288086295127869, "rewards/code_complexity_reward/std": 0.06892483681440353, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 728, "step_time": 36.50343661010265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 112.931640625, "completions/mean_terminated_length": 112.931640625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2333939024247229, "epoch": 0.4153846153846154, "frac_reward_zero_std": 0.484375, "grad_norm": 0.060546617954969406, "kl": 0.17077685263939202, "learning_rate": 3.6378560256940766e-06, "loss": 0.0008539748378098011, "num_tokens": 119333699.0, "reward": 2.3414554595947266, "reward_std": 0.5450735688209534, "rewards/code_complexity_reward/mean": 0.91455078125, "rewards/code_complexity_reward/std": 0.1497166007757187, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 729, "step_time": 47.88554634060711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 124.138671875, "completions/mean_terminated_length": 123.37964630126953, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2382366070523858, "epoch": 0.41595441595441596, "frac_reward_zero_std": 0.359375, "grad_norm": 0.06643696874380112, "kl": 0.16090020828414708, "learning_rate": 3.6334248139913012e-06, "loss": 0.0008048413437791169, "num_tokens": 119465170.0, "reward": 2.339648485183716, "reward_std": 0.5509445071220398, "rewards/code_complexity_reward/mean": 0.9073241949081421, "rewards/code_complexity_reward/std": 0.15183182060718536, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 730, "step_time": 66.88211152330041 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 330.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 119.537109375, "completions/mean_terminated_length": 119.537109375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2314699124544859, "epoch": 0.4165242165242165, "frac_reward_zero_std": 0.5, "grad_norm": 0.05670640617609024, "kl": 0.17380612902343273, "learning_rate": 3.628989115577918e-06, "loss": 0.0008693113340996206, "num_tokens": 119594029.0, "reward": 2.307910203933716, "reward_std": 0.47321707010269165, "rewards/code_complexity_reward/mean": 0.9208007454872131, "rewards/code_complexity_reward/std": 0.07469404488801956, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 731, "step_time": 36.20815826486796 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 122.1640625, "completions/mean_terminated_length": 122.1640625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23617875133641064, "epoch": 0.4170940170940171, "frac_reward_zero_std": 0.484375, "grad_norm": 0.061462193727493286, "kl": 0.16773155727423728, "learning_rate": 3.6245489480128295e-06, "loss": 0.0008387069683521986, "num_tokens": 119727641.0, "reward": 2.299853801727295, "reward_std": 0.48203158378601074, "rewards/code_complexity_reward/mean": 0.9208008050918579, "rewards/code_complexity_reward/std": 0.09364306926727295, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 732, "step_time": 42.16632583923638 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 113.47265625, "completions/mean_terminated_length": 113.47265625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23693749867379665, "epoch": 0.41766381766381766, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06342452764511108, "kl": 0.173035234445706, "learning_rate": 3.6201043288726272e-06, "loss": 0.0008651434327475727, "num_tokens": 119854867.0, "reward": 2.3118653297424316, "reward_std": 0.4831172227859497, "rewards/code_complexity_reward/mean": 0.9212890863418579, "rewards/code_complexity_reward/std": 0.09229567646980286, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 733, "step_time": 45.01651695556939 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 117.3359375, "completions/mean_terminated_length": 115.78823852539062, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2275390352588147, "epoch": 0.4182336182336182, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05729887634515762, "kl": 0.16104419762268662, "learning_rate": 3.615655275751528e-06, "loss": 0.0008053093333728611, "num_tokens": 119983383.0, "reward": 2.3769044876098633, "reward_std": 0.5283198356628418, "rewards/code_complexity_reward/mean": 0.92236328125, "rewards/code_complexity_reward/std": 0.11025180667638779, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 734, "step_time": 57.84892935678363 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 315.0, "completions/max_terminated_length": 315.0, "completions/mean_length": 118.931640625, "completions/mean_terminated_length": 118.931640625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2413768474943936, "epoch": 0.4188034188034188, "frac_reward_zero_std": 0.453125, "grad_norm": 0.0734134092926979, "kl": 0.1986909422557801, "learning_rate": 3.611201806261298e-06, "loss": 0.000993190100416541, "num_tokens": 120114708.0, "reward": 2.3026368618011475, "reward_std": 0.4894362688064575, "rewards/code_complexity_reward/mean": 0.92041015625, "rewards/code_complexity_reward/std": 0.10317099094390869, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 735, "step_time": 36.446593481116 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 413.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 116.326171875, "completions/mean_terminated_length": 116.326171875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2388411294668913, "epoch": 0.41937321937321936, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06495597213506699, "kl": 0.1840423293178901, "learning_rate": 3.6067439380311874e-06, "loss": 0.00092079839669168, "num_tokens": 120242283.0, "reward": 2.2716798782348633, "reward_std": 0.5027408003807068, "rewards/code_complexity_reward/mean": 0.915820300579071, "rewards/code_complexity_reward/std": 0.13502810895442963, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 736, "step_time": 42.137955982238054 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 333.0, "completions/mean_length": 120.880859375, "completions/mean_terminated_length": 120.1154556274414, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22861385764554143, "epoch": 0.41994301994301997, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05387306585907936, "kl": 0.16233495250344276, "learning_rate": 3.6022816887078573e-06, "loss": 0.000811712583526969, "num_tokens": 120374230.0, "reward": 2.2479004859924316, "reward_std": 0.4595334827899933, "rewards/code_complexity_reward/mean": 0.9215819835662842, "rewards/code_complexity_reward/std": 0.10716976225376129, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 737, "step_time": 48.87461355980486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 396.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 116.984375, "completions/mean_terminated_length": 116.984375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23471211222931743, "epoch": 0.4205128205128205, "frac_reward_zero_std": 0.375, "grad_norm": 0.07195467501878738, "kl": 0.17701137077528983, "learning_rate": 3.5978150759553132e-06, "loss": 0.000885074376128614, "num_tokens": 120501830.0, "reward": 2.2584962844848633, "reward_std": 0.47050073742866516, "rewards/code_complexity_reward/mean": 0.922167956829071, "rewards/code_complexity_reward/std": 0.10908706486225128, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 738, "step_time": 47.647420645691454 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 115.833984375, "completions/mean_terminated_length": 115.833984375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2511227340437472, "epoch": 0.42108262108262107, "frac_reward_zero_std": 0.5, "grad_norm": 0.061328113079071045, "kl": 0.1839169841259718, "learning_rate": 3.5933441174548328e-06, "loss": 0.0009198770276270807, "num_tokens": 120628409.0, "reward": 2.3004884719848633, "reward_std": 0.4694427251815796, "rewards/code_complexity_reward/mean": 0.927050769329071, "rewards/code_complexity_reward/std": 0.08811847865581512, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 739, "step_time": 36.71731202956289 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 315.0, "completions/max_terminated_length": 315.0, "completions/mean_length": 112.43359375, "completions/mean_terminated_length": 112.43359375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23712399927899241, "epoch": 0.42165242165242167, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06554807722568512, "kl": 0.17283915635198355, "learning_rate": 3.5888688309048957e-06, "loss": 0.0008644360932521522, "num_tokens": 120754887.0, "reward": 2.35888671875, "reward_std": 0.5166353583335876, "rewards/code_complexity_reward/mean": 0.9239257574081421, "rewards/code_complexity_reward/std": 0.11186385154724121, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 740, "step_time": 35.65244088508189 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 393.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 118.01171875, "completions/mean_terminated_length": 118.01171875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23598321876488626, "epoch": 0.4222222222222222, "frac_reward_zero_std": 0.46875, "grad_norm": 0.053581077605485916, "kl": 0.16717013716697693, "learning_rate": 3.5843892340211166e-06, "loss": 0.0008360545616596937, "num_tokens": 120882445.0, "reward": 2.296142578125, "reward_std": 0.4697982370853424, "rewards/code_complexity_reward/mean": 0.927050769329071, "rewards/code_complexity_reward/std": 0.08828487992286682, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 741, "step_time": 42.118050614371896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 121.3828125, "completions/mean_terminated_length": 120.61839294433594, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23507681884802878, "epoch": 0.42279202279202277, "frac_reward_zero_std": 0.5625, "grad_norm": 0.04904717952013016, "kl": 0.16130402614362538, "learning_rate": 3.5799053445361703e-06, "loss": 0.0008066456648521125, "num_tokens": 121013777.0, "reward": 2.2369141578674316, "reward_std": 0.48175331950187683, "rewards/code_complexity_reward/mean": 0.9110351800918579, "rewards/code_complexity_reward/std": 0.13530534505844116, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 742, "step_time": 47.92729223985225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 121.373046875, "completions/mean_terminated_length": 121.373046875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2252089532557875, "epoch": 0.4233618233618234, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05494917184114456, "kl": 0.17027407372370362, "learning_rate": 3.5754171801997255e-06, "loss": 0.0008514552609995008, "num_tokens": 121147600.0, "reward": 2.2549805641174316, "reward_std": 0.4538400173187256, "rewards/code_complexity_reward/mean": 0.9193359017372131, "rewards/code_complexity_reward/std": 0.09021147340536118, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 743, "step_time": 43.347369517199695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 118.0078125, "completions/mean_terminated_length": 118.0078125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23454862646758556, "epoch": 0.4239316239316239, "frac_reward_zero_std": 0.59375, "grad_norm": 0.050252754241228104, "kl": 0.16555820929352194, "learning_rate": 3.5709247587783722e-06, "loss": 0.0008277681190520525, "num_tokens": 121274364.0, "reward": 2.2772462368011475, "reward_std": 0.43662428855895996, "rewards/code_complexity_reward/mean": 0.92919921875, "rewards/code_complexity_reward/std": 0.037950728088617325, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 744, "step_time": 46.32749632745981 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 296.0, "completions/mean_length": 115.880859375, "completions/mean_terminated_length": 115.10567474365234, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2374174369033426, "epoch": 0.42450142450142453, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06512759625911713, "kl": 0.16693422081880271, "learning_rate": 3.566428098055554e-06, "loss": 0.0008346964023075998, "num_tokens": 121400663.0, "reward": 2.3272461891174316, "reward_std": 0.5230474472045898, "rewards/code_complexity_reward/mean": 0.9193359613418579, "rewards/code_complexity_reward/std": 0.13229550421237946, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 745, "step_time": 47.60357339680195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 116.658203125, "completions/mean_terminated_length": 116.658203125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2338986920658499, "epoch": 0.4250712250712251, "frac_reward_zero_std": 0.5, "grad_norm": 0.06031543016433716, "kl": 0.18843265215400606, "learning_rate": 3.5619272158314923e-06, "loss": 0.0009418780682608485, "num_tokens": 121530760.0, "reward": 2.342090129852295, "reward_std": 0.5276419520378113, "rewards/code_complexity_reward/mean": 0.9168944954872131, "rewards/code_complexity_reward/std": 0.12579579651355743, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 746, "step_time": 40.8469273019582 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 441.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 120.455078125, "completions/mean_terminated_length": 120.455078125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2319676843471825, "epoch": 0.4256410256410256, "frac_reward_zero_std": 0.40625, "grad_norm": 0.0630476176738739, "kl": 0.16976116818841547, "learning_rate": 3.557422129923124e-06, "loss": 0.0008486152510158718, "num_tokens": 121660137.0, "reward": 2.343799114227295, "reward_std": 0.5291751623153687, "rewards/code_complexity_reward/mean": 0.9112304449081421, "rewards/code_complexity_reward/std": 0.12948735058307648, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 747, "step_time": 49.81838066224009 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 117.798828125, "completions/mean_terminated_length": 117.798828125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22853952157311141, "epoch": 0.42621082621082623, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06220731884241104, "kl": 0.16837141872383654, "learning_rate": 3.5529128581640247e-06, "loss": 0.0008419090881943703, "num_tokens": 121785858.0, "reward": 2.342578411102295, "reward_std": 0.49728578329086304, "rewards/code_complexity_reward/mean": 0.9271484613418579, "rewards/code_complexity_reward/std": 0.09516323357820511, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 748, "step_time": 66.55954880453646 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 404.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 115.626953125, "completions/mean_terminated_length": 115.626953125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2382134289946407, "epoch": 0.4267806267806268, "frac_reward_zero_std": 0.515625, "grad_norm": 0.07325766980648041, "kl": 0.1652254048967734, "learning_rate": 3.5483994184043384e-06, "loss": 0.0008264250354841352, "num_tokens": 121914627.0, "reward": 2.3405275344848633, "reward_std": 0.5182393789291382, "rewards/code_complexity_reward/mean": 0.9201171398162842, "rewards/code_complexity_reward/std": 0.1153382882475853, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 749, "step_time": 41.60983489919454 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 490.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 123.5546875, "completions/mean_terminated_length": 123.5546875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22947077779099345, "epoch": 0.42735042735042733, "frac_reward_zero_std": 0.375, "grad_norm": 0.06573255360126495, "kl": 0.15632009517867118, "learning_rate": 3.5438818285107094e-06, "loss": 0.000781706185080111, "num_tokens": 122045423.0, "reward": 2.3265628814697266, "reward_std": 0.48661890625953674, "rewards/code_complexity_reward/mean": 0.9156249761581421, "rewards/code_complexity_reward/std": 0.0781388208270073, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 750, "step_time": 47.96175064891577 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 335.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 116.310546875, "completions/mean_terminated_length": 116.310546875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2311637920793146, "epoch": 0.42792022792022794, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05433512479066849, "kl": 0.1610881123924628, "learning_rate": 3.5393601063662107e-06, "loss": 0.0008054728386923671, "num_tokens": 122174310.0, "reward": 2.2774415016174316, "reward_std": 0.5048627853393555, "rewards/code_complexity_reward/mean": 0.9157227277755737, "rewards/code_complexity_reward/std": 0.1344040483236313, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 751, "step_time": 38.836250892840326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 418.0, "completions/max_terminated_length": 418.0, "completions/mean_length": 120.189453125, "completions/mean_terminated_length": 120.189453125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23875019396655262, "epoch": 0.4284900284900285, "frac_reward_zero_std": 0.59375, "grad_norm": 0.07605237513780594, "kl": 0.17400352284312248, "learning_rate": 3.5348342698702735e-06, "loss": 0.0008699499303475022, "num_tokens": 122303863.0, "reward": 2.3323731422424316, "reward_std": 0.483113169670105, "rewards/code_complexity_reward/mean": 0.927929699420929, "rewards/code_complexity_reward/std": 0.09200532734394073, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 752, "step_time": 78.38965049572289 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 113.810546875, "completions/mean_terminated_length": 113.03131103515625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22629180806688964, "epoch": 0.42905982905982903, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05184781923890114, "kl": 0.15675761259626597, "learning_rate": 3.530304336938614e-06, "loss": 0.0007836504373699427, "num_tokens": 122428582.0, "reward": 2.3287599086761475, "reward_std": 0.4865000247955322, "rewards/code_complexity_reward/mean": 0.92626953125, "rewards/code_complexity_reward/std": 0.08918169140815735, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 753, "step_time": 56.04995222389698 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 311.0, "completions/max_terminated_length": 311.0, "completions/mean_length": 118.94921875, "completions/mean_terminated_length": 118.94921875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23360458225943148, "epoch": 0.42962962962962964, "frac_reward_zero_std": 0.5, "grad_norm": 0.058570705354213715, "kl": 0.17828213586471975, "learning_rate": 3.525770325503167e-06, "loss": 0.0008912894991226494, "num_tokens": 122561628.0, "reward": 2.27294921875, "reward_std": 0.46312737464904785, "rewards/code_complexity_reward/mean": 0.9258788824081421, "rewards/code_complexity_reward/std": 0.09746899455785751, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 754, "step_time": 64.64471150375903 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 414.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 114.935546875, "completions/mean_terminated_length": 114.935546875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22027688706293702, "epoch": 0.4301994301994302, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06462779641151428, "kl": 0.19996636384166777, "learning_rate": 3.5212322535120092e-06, "loss": 0.0009988414822146297, "num_tokens": 122691699.0, "reward": 2.3703126907348633, "reward_std": 0.48964571952819824, "rewards/code_complexity_reward/mean": 0.930468738079071, "rewards/code_complexity_reward/std": 0.06923805922269821, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 755, "step_time": 53.647534179501235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 373.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 114.02734375, "completions/mean_terminated_length": 114.02734375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2285783274564892, "epoch": 0.4307692307692308, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05307735875248909, "kl": 0.17595203628297895, "learning_rate": 3.5166901389292927e-06, "loss": 0.0008797491900622845, "num_tokens": 122817257.0, "reward": 2.3949708938598633, "reward_std": 0.49052202701568604, "rewards/code_complexity_reward/mean": 0.93212890625, "rewards/code_complexity_reward/std": 0.0636308342218399, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 756, "step_time": 39.599124173633754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 119.056640625, "completions/mean_terminated_length": 118.28767395019531, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2224506677594036, "epoch": 0.43133903133903134, "frac_reward_zero_std": 0.484375, "grad_norm": 0.07866758853197098, "kl": 0.1672589291119948, "learning_rate": 3.512143999735174e-06, "loss": 0.0008361582877114415, "num_tokens": 122945382.0, "reward": 2.324951171875, "reward_std": 0.49605461955070496, "rewards/code_complexity_reward/mean": 0.9244140386581421, "rewards/code_complexity_reward/std": 0.10407381504774094, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 757, "step_time": 49.4339636284858 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 116.412109375, "completions/mean_terminated_length": 116.412109375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23363584652543068, "epoch": 0.4319088319088319, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06502310186624527, "kl": 0.15981316973920912, "learning_rate": 3.507593853925738e-06, "loss": 0.0007991956081241369, "num_tokens": 123072889.0, "reward": 2.2823243141174316, "reward_std": 0.5251269936561584, "rewards/code_complexity_reward/mean": 0.909863293170929, "rewards/code_complexity_reward/std": 0.15016423165798187, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 758, "step_time": 37.44995155464858 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 113.060546875, "completions/mean_terminated_length": 113.060546875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.21783102303743362, "epoch": 0.4324786324786325, "frac_reward_zero_std": 0.484375, "grad_norm": 0.053542360663414, "kl": 0.17670935136266053, "learning_rate": 3.503039719512932e-06, "loss": 0.0008838879293762147, "num_tokens": 123197696.0, "reward": 2.436523675918579, "reward_std": 0.5537225604057312, "rewards/code_complexity_reward/mean": 0.9087890386581421, "rewards/code_complexity_reward/std": 0.13563251495361328, "rewards/code_execution_reward/mean": 0.4375, "rewards/code_execution_reward/std": 0.49656352400779724, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 759, "step_time": 44.46603425126523 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 300.0, "completions/max_terminated_length": 300.0, "completions/mean_length": 116.5234375, "completions/mean_terminated_length": 116.5234375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23742236732505262, "epoch": 0.43304843304843305, "frac_reward_zero_std": 0.46875, "grad_norm": 0.07006987929344177, "kl": 0.17131425440311432, "learning_rate": 3.4984816145244926e-06, "loss": 0.0008570621721446514, "num_tokens": 123327828.0, "reward": 2.2184083461761475, "reward_std": 0.46399128437042236, "rewards/code_complexity_reward/mean": 0.9166991710662842, "rewards/code_complexity_reward/std": 0.12679028511047363, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 760, "step_time": 44.68533128499985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 284.0, "completions/max_terminated_length": 284.0, "completions/mean_length": 116.947265625, "completions/mean_terminated_length": 116.947265625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22335535916499794, "epoch": 0.4336182336182336, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06129064783453941, "kl": 0.17208216222934425, "learning_rate": 3.493919557003872e-06, "loss": 0.0008606433984823525, "num_tokens": 123454969.0, "reward": 2.3433594703674316, "reward_std": 0.5290437340736389, "rewards/code_complexity_reward/mean": 0.9113280773162842, "rewards/code_complexity_reward/std": 0.13211582601070404, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 761, "step_time": 60.691434379667044 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 443.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 117.021484375, "completions/mean_terminated_length": 117.021484375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24510022974573076, "epoch": 0.4341880341880342, "frac_reward_zero_std": 0.5625, "grad_norm": 0.056437741965055466, "kl": 0.16105268220417202, "learning_rate": 3.4893535650101716e-06, "loss": 0.0008053563069552183, "num_tokens": 123583556.0, "reward": 2.2826662063598633, "reward_std": 0.45482149720191956, "rewards/code_complexity_reward/mean": 0.931933581829071, "rewards/code_complexity_reward/std": 0.07585631310939789, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 762, "step_time": 57.762171250768006 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 280.0, "completions/max_terminated_length": 280.0, "completions/mean_length": 113.638671875, "completions/mean_terminated_length": 113.638671875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23606777214445174, "epoch": 0.43475783475783475, "frac_reward_zero_std": 0.65625, "grad_norm": 0.04642355069518089, "kl": 0.16168227279558778, "learning_rate": 3.4847836566180644e-06, "loss": 0.0008084739674814045, "num_tokens": 123710163.0, "reward": 2.3150391578674316, "reward_std": 0.4887365400791168, "rewards/code_complexity_reward/mean": 0.9240233898162842, "rewards/code_complexity_reward/std": 0.10355247557163239, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 763, "step_time": 34.28743564616889 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 389.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 121.193359375, "completions/mean_terminated_length": 121.193359375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24413298605941236, "epoch": 0.43532763532763535, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06094493344426155, "kl": 0.1688083534827456, "learning_rate": 3.480209849917729e-06, "loss": 0.0008441952522844076, "num_tokens": 123840622.0, "reward": 2.262500047683716, "reward_std": 0.4894183874130249, "rewards/code_complexity_reward/mean": 0.9173828363418579, "rewards/code_complexity_reward/std": 0.1272568702697754, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 764, "step_time": 40.300917063839734 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 256.0, "completions/max_terminated_length": 256.0, "completions/mean_length": 117.689453125, "completions/mean_terminated_length": 117.689453125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2227300051599741, "epoch": 0.4358974358974359, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06397540122270584, "kl": 0.17084820743184537, "learning_rate": 3.475632163014775e-06, "loss": 0.0008539290865883231, "num_tokens": 123968839.0, "reward": 2.432910442352295, "reward_std": 0.5037685036659241, "rewards/code_complexity_reward/mean": 0.9286132454872131, "rewards/code_complexity_reward/std": 0.0658872202038765, "rewards/code_execution_reward/mean": 0.40625, "rewards/code_execution_reward/std": 0.49161264300346375, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 765, "step_time": 34.4692477863282 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 281.0, "completions/max_terminated_length": 281.0, "completions/mean_length": 118.123046875, "completions/mean_terminated_length": 118.123046875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23029652633704245, "epoch": 0.43646723646723645, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05537508428096771, "kl": 0.16147614165674895, "learning_rate": 3.4710506140301708e-06, "loss": 0.0008070652838796377, "num_tokens": 124097398.0, "reward": 2.301074504852295, "reward_std": 0.4917432963848114, "rewards/code_complexity_reward/mean": 0.9253906011581421, "rewards/code_complexity_reward/std": 0.11247892677783966, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 766, "step_time": 58.875720732845366 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 336.0, "completions/max_terminated_length": 336.0, "completions/mean_length": 117.48828125, "completions/mean_terminated_length": 117.48828125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22696087672375143, "epoch": 0.43703703703703706, "frac_reward_zero_std": 0.484375, "grad_norm": 0.0609293095767498, "kl": 0.17395233025308698, "learning_rate": 3.4664652211001753e-06, "loss": 0.0008698339806869626, "num_tokens": 124227240.0, "reward": 2.259570598602295, "reward_std": 0.42221367359161377, "rewards/code_complexity_reward/mean": 0.9359374642372131, "rewards/code_complexity_reward/std": 0.05002445727586746, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 767, "step_time": 36.19404869712889 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 391.0, "completions/max_terminated_length": 391.0, "completions/mean_length": 116.404296875, "completions/mean_terminated_length": 116.404296875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22330038459040225, "epoch": 0.4376068376068376, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05373692512512207, "kl": 0.1638362391386181, "learning_rate": 3.4618760023762633e-06, "loss": 0.0008192109526135027, "num_tokens": 124357311.0, "reward": 2.3597657680511475, "reward_std": 0.48528480529785156, "rewards/code_complexity_reward/mean": 0.931640625, "rewards/code_complexity_reward/std": 0.06406670808792114, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 768, "step_time": 40.31347389891744 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 126.115234375, "completions/mean_terminated_length": 125.36007690429688, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22946163616143167, "epoch": 0.43817663817663816, "frac_reward_zero_std": 0.5, "grad_norm": 0.057667601853609085, "kl": 0.1812891431618482, "learning_rate": 3.457282976025051e-06, "loss": 0.0009065725025720894, "num_tokens": 124488938.0, "reward": 2.2933592796325684, "reward_std": 0.4598320722579956, "rewards/code_complexity_reward/mean": 0.9225585460662842, "rewards/code_complexity_reward/std": 0.07496833056211472, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 769, "step_time": 48.340581270866096 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 118.474609375, "completions/mean_terminated_length": 118.474609375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22855843976140022, "epoch": 0.43874643874643876, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05607304349541664, "kl": 0.15561355522368103, "learning_rate": 3.4526861602282315e-06, "loss": 0.0007782523753121495, "num_tokens": 124621213.0, "reward": 2.3197267055511475, "reward_std": 0.5306012630462646, "rewards/code_complexity_reward/mean": 0.9130859375, "rewards/code_complexity_reward/std": 0.13647307455539703, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 770, "step_time": 47.50873712450266 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 119.068359375, "completions/mean_terminated_length": 118.2994155883789, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23190352134406567, "epoch": 0.4393162393162393, "frac_reward_zero_std": 0.46875, "grad_norm": 0.058494359254837036, "kl": 0.16301431320607662, "learning_rate": 3.448085573182496e-06, "loss": 0.000814817612990737, "num_tokens": 124751128.0, "reward": 2.322510004043579, "reward_std": 0.4936261475086212, "rewards/code_complexity_reward/mean": 0.9268554449081421, "rewards/code_complexity_reward/std": 0.09674989432096481, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 771, "step_time": 48.035845559090376 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 305.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 115.775390625, "completions/mean_terminated_length": 115.775390625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23362068249844015, "epoch": 0.43988603988603986, "frac_reward_zero_std": 0.453125, "grad_norm": 0.062323082238435745, "kl": 0.15879006835166365, "learning_rate": 3.443481233099465e-06, "loss": 0.0007938719354569912, "num_tokens": 124879749.0, "reward": 2.380176067352295, "reward_std": 0.5632608532905579, "rewards/code_complexity_reward/mean": 0.9081054925918579, "rewards/code_complexity_reward/std": 0.149617001414299, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 772, "step_time": 39.69265708886087 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 129.15234375, "completions/mean_terminated_length": 128.40313720703125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24026219570077956, "epoch": 0.44045584045584046, "frac_reward_zero_std": 0.4375, "grad_norm": 0.061001554131507874, "kl": 0.15949718828778714, "learning_rate": 3.438873158205616e-06, "loss": 0.0007973855244927108, "num_tokens": 125017203.0, "reward": 2.2762696743011475, "reward_std": 0.4876065254211426, "rewards/code_complexity_reward/mean": 0.917187511920929, "rewards/code_complexity_reward/std": 0.12097734212875366, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 773, "step_time": 56.585362154990435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 116.92578125, "completions/mean_terminated_length": 116.92578125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22854342381469905, "epoch": 0.441025641025641, "frac_reward_zero_std": 0.53125, "grad_norm": 0.1216990128159523, "kl": 0.1640041135251522, "learning_rate": 3.434261366742211e-06, "loss": 0.0008198553696274757, "num_tokens": 125145189.0, "reward": 2.3419435024261475, "reward_std": 0.4956165850162506, "rewards/code_complexity_reward/mean": 0.9220702648162842, "rewards/code_complexity_reward/std": 0.09421220421791077, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 774, "step_time": 36.46484115533531 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 376.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 121.72265625, "completions/mean_terminated_length": 121.72265625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.218161687720567, "epoch": 0.4415954415954416, "frac_reward_zero_std": 0.53125, "grad_norm": 0.060818471014499664, "kl": 0.16014953679405153, "learning_rate": 3.4296458769652234e-06, "loss": 0.0008006957941688597, "num_tokens": 125274647.0, "reward": 2.293701171875, "reward_std": 0.48798245191574097, "rewards/code_complexity_reward/mean": 0.9226561784744263, "rewards/code_complexity_reward/std": 0.10706169158220291, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 775, "step_time": 58.14223639201373 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 483.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 124.474609375, "completions/mean_terminated_length": 124.474609375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23794913245365024, "epoch": 0.44216524216524217, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05728629603981972, "kl": 0.16091739921830595, "learning_rate": 3.425026707145266e-06, "loss": 0.0008047357550822198, "num_tokens": 125405322.0, "reward": 2.3885743618011475, "reward_std": 0.5178812146186829, "rewards/code_complexity_reward/mean": 0.91943359375, "rewards/code_complexity_reward/std": 0.09553777426481247, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 776, "step_time": 45.59545933175832 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 117.138671875, "completions/mean_terminated_length": 117.138671875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23287952551618218, "epoch": 0.4427350427350427, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06404656171798706, "kl": 0.16331858397461474, "learning_rate": 3.420403875567522e-06, "loss": 0.0008172281668521464, "num_tokens": 125534961.0, "reward": 2.248974561691284, "reward_std": 0.47319045662879944, "rewards/code_complexity_reward/mean": 0.9072265625, "rewards/code_complexity_reward/std": 0.1095111146569252, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 777, "step_time": 38.79372720606625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 117.669921875, "completions/mean_terminated_length": 117.669921875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22431817837059498, "epoch": 0.4433048433048433, "frac_reward_zero_std": 0.4375, "grad_norm": 0.059006575495004654, "kl": 0.15562130510807037, "learning_rate": 3.415777400531666e-06, "loss": 0.0007781411986798048, "num_tokens": 125664624.0, "reward": 2.4256837368011475, "reward_std": 0.5134368538856506, "rewards/code_complexity_reward/mean": 0.928906261920929, "rewards/code_complexity_reward/std": 0.08249043673276901, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 778, "step_time": 38.03335941955447 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 401.0, "completions/mean_length": 119.677734375, "completions/mean_terminated_length": 118.90998077392578, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.22667801869101822, "epoch": 0.44387464387464387, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05255373567342758, "kl": 0.16032835084479302, "learning_rate": 3.411147300351798e-06, "loss": 0.0008016512147150934, "num_tokens": 125793939.0, "reward": 2.3133301734924316, "reward_std": 0.5294870138168335, "rewards/code_complexity_reward/mean": 0.9147460460662842, "rewards/code_complexity_reward/std": 0.14369438588619232, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 779, "step_time": 64.28034288156778 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 433.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 123.841796875, "completions/mean_terminated_length": 123.841796875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22274881531484425, "epoch": 0.4444444444444444, "frac_reward_zero_std": 0.46875, "grad_norm": 0.059179313480854034, "kl": 0.15714357001706958, "learning_rate": 3.4065135933563688e-06, "loss": 0.0007857111049816012, "num_tokens": 125927338.0, "reward": 2.2928712368011475, "reward_std": 0.5459537506103516, "rewards/code_complexity_reward/mean": 0.8979491591453552, "rewards/code_complexity_reward/std": 0.16269107162952423, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 780, "step_time": 57.6019414588809 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 388.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 117.646484375, "completions/mean_terminated_length": 117.646484375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2322069350630045, "epoch": 0.445014245014245, "frac_reward_zero_std": 0.5, "grad_norm": 0.06085285544395447, "kl": 0.16464961774181575, "learning_rate": 3.4018762978881046e-06, "loss": 0.0008231712272390723, "num_tokens": 126058725.0, "reward": 2.2674317359924316, "reward_std": 0.4624978005886078, "rewards/code_complexity_reward/mean": 0.9247070550918579, "rewards/code_complexity_reward/std": 0.09901625663042068, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 781, "step_time": 39.55774739198387 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 127.123046875, "completions/mean_terminated_length": 127.123046875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22390033956617117, "epoch": 0.4455840455840456, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05323977395892143, "kl": 0.16521181527059525, "learning_rate": 3.3972354323039375e-06, "loss": 0.0008261142647825181, "num_tokens": 126194908.0, "reward": 2.340136766433716, "reward_std": 0.5268387794494629, "rewards/code_complexity_reward/mean": 0.9120116829872131, "rewards/code_complexity_reward/std": 0.1234978437423706, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 782, "step_time": 42.597741285339 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 329.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 121.451171875, "completions/mean_terminated_length": 121.451171875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.24166656704619527, "epoch": 0.4461538461538462, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06049978360533714, "kl": 0.15916110284160823, "learning_rate": 3.392591014974933e-06, "loss": 0.0007958052447065711, "num_tokens": 126325139.0, "reward": 2.377490520477295, "reward_std": 0.5668452382087708, "rewards/code_complexity_reward/mean": 0.9046875238418579, "rewards/code_complexity_reward/std": 0.15623849630355835, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 783, "step_time": 52.653485177084804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 286.0, "completions/max_terminated_length": 286.0, "completions/mean_length": 117.26953125, "completions/mean_terminated_length": 117.26953125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.227381547447294, "epoch": 0.44672364672364673, "frac_reward_zero_std": 0.4375, "grad_norm": 0.059365544468164444, "kl": 0.1690845617558807, "learning_rate": 3.387943064286216e-06, "loss": 0.0008452270994894207, "num_tokens": 126452781.0, "reward": 2.3018555641174316, "reward_std": 0.49611157178878784, "rewards/code_complexity_reward/mean": 0.9196288585662842, "rewards/code_complexity_reward/std": 0.11256856471300125, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 784, "step_time": 34.314411708153784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 117.76953125, "completions/mean_terminated_length": 117.76953125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22680861596018076, "epoch": 0.4472934472934473, "frac_reward_zero_std": 0.546875, "grad_norm": 0.06034993380308151, "kl": 0.16832640604116023, "learning_rate": 3.3832915986368973e-06, "loss": 0.0008416088530793786, "num_tokens": 126583079.0, "reward": 2.3702149391174316, "reward_std": 0.49620285630226135, "rewards/code_complexity_reward/mean": 0.9300781488418579, "rewards/code_complexity_reward/std": 0.07721670717000961, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 785, "step_time": 36.50326559506357 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 328.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 120.125, "completions/mean_terminated_length": 120.125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2225158002693206, "epoch": 0.4478632478632479, "frac_reward_zero_std": 0.484375, "grad_norm": 0.054697707295417786, "kl": 0.17334443237632513, "learning_rate": 3.3786366364400036e-06, "loss": 0.0008667020010761917, "num_tokens": 126713623.0, "reward": 2.2232422828674316, "reward_std": 0.4572816491127014, "rewards/code_complexity_reward/mean": 0.9220702648162842, "rewards/code_complexity_reward/std": 0.11887111514806747, "rewards/code_execution_reward/mean": 0.208984375, "rewards/code_execution_reward/std": 0.40698084235191345, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 786, "step_time": 45.89770476985723 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 121.267578125, "completions/mean_terminated_length": 121.267578125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2387210528831929, "epoch": 0.44843304843304843, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06923855096101761, "kl": 0.1941139594418928, "learning_rate": 3.3739781961224012e-06, "loss": 0.0009701287490315735, "num_tokens": 126847280.0, "reward": 2.281982660293579, "reward_std": 0.4981362819671631, "rewards/code_complexity_reward/mean": 0.916796863079071, "rewards/code_complexity_reward/std": 0.12127459794282913, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 787, "step_time": 38.82190101314336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 297.0, "completions/max_terminated_length": 297.0, "completions/mean_length": 117.58984375, "completions/mean_terminated_length": 117.58984375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22463393886573613, "epoch": 0.449002849002849, "frac_reward_zero_std": 0.5, "grad_norm": 0.05526304617524147, "kl": 0.1541865123435855, "learning_rate": 3.3693162961247255e-06, "loss": 0.0007713286904618144, "num_tokens": 126973526.0, "reward": 2.337207078933716, "reward_std": 0.49883171916007996, "rewards/code_complexity_reward/mean": 0.9256836175918579, "rewards/code_complexity_reward/std": 0.09876696020364761, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 788, "step_time": 44.313495754264295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 324.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 121.33203125, "completions/mean_terminated_length": 121.33203125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2288185900542885, "epoch": 0.4495726495726496, "frac_reward_zero_std": 0.375, "grad_norm": 0.06410795450210571, "kl": 0.18248380243312567, "learning_rate": 3.3646509549013073e-06, "loss": 0.0009124355856329203, "num_tokens": 127102960.0, "reward": 2.335693359375, "reward_std": 0.48358237743377686, "rewards/code_complexity_reward/mean": 0.921679675579071, "rewards/code_complexity_reward/std": 0.07980351895093918, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 789, "step_time": 44.28021269943565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 282.0, "completions/max_terminated_length": 282.0, "completions/mean_length": 119.8671875, "completions/mean_terminated_length": 119.8671875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23234545369632542, "epoch": 0.45014245014245013, "frac_reward_zero_std": 0.53125, "grad_norm": 0.06326513737440109, "kl": 0.16010900668334216, "learning_rate": 3.3599821909200993e-06, "loss": 0.000800329027697444, "num_tokens": 127235028.0, "reward": 2.305957317352295, "reward_std": 0.5217515826225281, "rewards/code_complexity_reward/mean": 0.915332019329071, "rewards/code_complexity_reward/std": 0.13814640045166016, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 790, "step_time": 35.79242431744933 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 382.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 120.404296875, "completions/mean_terminated_length": 120.404296875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22669726330786943, "epoch": 0.45071225071225074, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05791401118040085, "kl": 0.15934417012613267, "learning_rate": 3.3553100226626034e-06, "loss": 0.0007965745171532035, "num_tokens": 127364195.0, "reward": 2.381347894668579, "reward_std": 0.49582281708717346, "rewards/code_complexity_reward/mean": 0.9258789420127869, "rewards/code_complexity_reward/std": 0.0676644816994667, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 791, "step_time": 39.94576172251254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 433.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 119.548828125, "completions/mean_terminated_length": 119.548828125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2450017447117716, "epoch": 0.4512820512820513, "frac_reward_zero_std": 0.390625, "grad_norm": 0.07473686337471008, "kl": 0.16770800366066396, "learning_rate": 3.3506344686237972e-06, "loss": 0.0008386748959310353, "num_tokens": 127492988.0, "reward": 2.2804200649261475, "reward_std": 0.5349558591842651, "rewards/code_complexity_reward/mean": 0.9072265625, "rewards/code_complexity_reward/std": 0.16128921508789062, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.019900046288967133, "step": 792, "step_time": 43.38589058816433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 389.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 121.888671875, "completions/mean_terminated_length": 121.888671875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2272208398208022, "epoch": 0.45185185185185184, "frac_reward_zero_std": 0.53125, "grad_norm": 0.062191903591156006, "kl": 0.1611053777160123, "learning_rate": 3.3459555473120618e-06, "loss": 0.0008055316284298897, "num_tokens": 127624403.0, "reward": 2.3389649391174316, "reward_std": 0.5103721618652344, "rewards/code_complexity_reward/mean": 0.9176757335662842, "rewards/code_complexity_reward/std": 0.11210958659648895, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 793, "step_time": 45.6553747151047 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 372.0, "completions/mean_length": 124.072265625, "completions/mean_terminated_length": 123.3131103515625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.22847000998444855, "epoch": 0.45242165242165244, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05092606693506241, "kl": 0.1545516699552536, "learning_rate": 3.3412732772491073e-06, "loss": 0.0007729153148829937, "num_tokens": 127756032.0, "reward": 2.3187012672424316, "reward_std": 0.4992588460445404, "rewards/code_complexity_reward/mean": 0.9201171398162842, "rewards/code_complexity_reward/std": 0.10377222299575806, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 794, "step_time": 50.59249220136553 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 345.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 120.83203125, "completions/mean_terminated_length": 120.83203125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2313174547161907, "epoch": 0.452991452991453, "frac_reward_zero_std": 0.4375, "grad_norm": 0.0625949278473854, "kl": 0.18336113600526005, "learning_rate": 3.3365876769698995e-06, "loss": 0.0009169366676360369, "num_tokens": 127888090.0, "reward": 2.3277344703674316, "reward_std": 0.5259274244308472, "rewards/code_complexity_reward/mean": 0.9112304449081421, "rewards/code_complexity_reward/std": 0.12777583301067352, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 795, "step_time": 56.50946811493486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 387.0, "completions/max_terminated_length": 387.0, "completions/mean_length": 119.9140625, "completions/mean_terminated_length": 119.9140625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22505039745010436, "epoch": 0.45356125356125354, "frac_reward_zero_std": 0.609375, "grad_norm": 0.05771629512310028, "kl": 0.16844933188986033, "learning_rate": 3.3318987650225877e-06, "loss": 0.0008421003585681319, "num_tokens": 128018702.0, "reward": 2.348876953125, "reward_std": 0.5222636461257935, "rewards/code_complexity_reward/mean": 0.919238269329071, "rewards/code_complexity_reward/std": 0.12040377408266068, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 796, "step_time": 39.88818695396185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 425.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 125.021484375, "completions/mean_terminated_length": 125.021484375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.24157603876665235, "epoch": 0.45413105413105415, "frac_reward_zero_std": 0.40625, "grad_norm": 0.05886080116033554, "kl": 0.15582681610248983, "learning_rate": 3.3272065599684313e-06, "loss": 0.0007792251417413354, "num_tokens": 128151273.0, "reward": 2.313965082168579, "reward_std": 0.4917870759963989, "rewards/code_complexity_reward/mean": 0.9239257574081421, "rewards/code_complexity_reward/std": 0.09906025230884552, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 797, "step_time": 42.717103911563754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 127.5625, "completions/mean_terminated_length": 126.81017303466797, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2289711006451398, "epoch": 0.4547008547008547, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05290907993912697, "kl": 0.16813420818652958, "learning_rate": 3.3225110803817234e-06, "loss": 0.0008405308471992612, "num_tokens": 128284689.0, "reward": 2.283203125, "reward_std": 0.4882209300994873, "rewards/code_complexity_reward/mean": 0.9211914539337158, "rewards/code_complexity_reward/std": 0.11444217711687088, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 798, "step_time": 58.071246665902436 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 123.607421875, "completions/mean_terminated_length": 123.607421875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23265232285484672, "epoch": 0.45527065527065524, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05575861781835556, "kl": 0.16748860594816506, "learning_rate": 3.3178123448497224e-06, "loss": 0.0008376085315831006, "num_tokens": 128417592.0, "reward": 2.354785442352295, "reward_std": 0.4923103451728821, "rewards/code_complexity_reward/mean": 0.9315429925918579, "rewards/code_complexity_reward/std": 0.07710618525743484, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 799, "step_time": 55.97983641549945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 125.107421875, "completions/mean_terminated_length": 125.107421875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24387559667229652, "epoch": 0.45584045584045585, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06209540367126465, "kl": 0.15200256346724927, "learning_rate": 3.3131103719725723e-06, "loss": 0.0007600557873956859, "num_tokens": 128550823.0, "reward": 2.3046388626098633, "reward_std": 0.5250773429870605, "rewards/code_complexity_reward/mean": 0.910839855670929, "rewards/code_complexity_reward/std": 0.1381123661994934, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 800, "step_time": 64.85941811650991 }, { "epoch": 0.45584045584045585, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.00125, "eval_completions/max_length": 176.71, "eval_completions/max_terminated_length": 176.0, "eval_completions/mean_length": 124.46125, "eval_completions/mean_terminated_length": 124.28857147216797, "eval_completions/min_length": 90.99, "eval_completions/min_terminated_length": 90.99, "eval_entropy": 0.23269843325018882, "eval_frac_reward_zero_std": 0.52, "eval_kl": 0.15108369618654252, "eval_loss": 0.0007550517912022769, "eval_num_tokens": 128550823.0, "eval_reward": 2.283468871116638, "eval_reward_std": 0.1889465882629156, "eval_rewards/code_complexity_reward/mean": 0.9148749848455191, "eval_rewards/code_complexity_reward/std": 0.02947531620040536, "eval_rewards/code_execution_reward/mean": 0.2775, "eval_rewards/code_execution_reward/std": 0.1520494231581688, "eval_rewards/code_syntax_reward/mean": 0.49125, "eval_rewards/code_syntax_reward/std": 0.01292115181684494, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.49984375, "eval_rewards/xmlcount_reward_func/std": 0.0004419417306780815, "eval_runtime": 800.1937, "eval_samples_per_second": 0.125, "eval_steps_per_second": 0.016, "step": 800 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 121.875, "completions/mean_terminated_length": 121.11154174804688, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24877615738660097, "epoch": 0.4564102564102564, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06875193864107132, "kl": 0.16095571767073125, "learning_rate": 3.3084051803632337e-06, "loss": 0.0008051874465309083, "num_tokens": 128683703.0, "reward": 2.2709474563598633, "reward_std": 0.4649368226528168, "rewards/code_complexity_reward/mean": 0.9221680164337158, "rewards/code_complexity_reward/std": 0.110866479575634, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 801, "step_time": 48.909491423517466 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 321.0, "completions/max_terminated_length": 321.0, "completions/mean_length": 120.08984375, "completions/mean_terminated_length": 120.08984375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23351145582273602, "epoch": 0.456980056980057, "frac_reward_zero_std": 0.484375, "grad_norm": 0.053332604467868805, "kl": 0.16516477731056511, "learning_rate": 3.30369678864741e-06, "loss": 0.0008259965688921511, "num_tokens": 128815245.0, "reward": 2.3751466274261475, "reward_std": 0.5473371744155884, "rewards/code_complexity_reward/mean": 0.9142577648162842, "rewards/code_complexity_reward/std": 0.1330135315656662, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 802, "step_time": 35.852472252212465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 121.791015625, "completions/mean_terminated_length": 121.791015625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23941811244003475, "epoch": 0.45754985754985755, "frac_reward_zero_std": 0.5, "grad_norm": 0.059386737644672394, "kl": 0.16974048118572682, "learning_rate": 3.298985215463472e-06, "loss": 0.0008487935410812497, "num_tokens": 128947594.0, "reward": 2.3346190452575684, "reward_std": 0.512107789516449, "rewards/code_complexity_reward/mean": 0.9186522960662842, "rewards/code_complexity_reward/std": 0.11980389803647995, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 803, "step_time": 46.322570422664285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 425.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 121.724609375, "completions/mean_terminated_length": 121.724609375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2342389349360019, "epoch": 0.4581196581196581, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06334265321493149, "kl": 0.15536297613289207, "learning_rate": 3.294270479462381e-06, "loss": 0.0007766330381855369, "num_tokens": 129079413.0, "reward": 2.2865238189697266, "reward_std": 0.4853772222995758, "rewards/code_complexity_reward/mean": 0.9248046875, "rewards/code_complexity_reward/std": 0.1084040030837059, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 804, "step_time": 52.33699384704232 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 125.892578125, "completions/mean_terminated_length": 125.892578125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24131361464969814, "epoch": 0.4586894586894587, "frac_reward_zero_std": 0.609375, "grad_norm": 0.05182366073131561, "kl": 0.16215716558508575, "learning_rate": 3.2895525993076232e-06, "loss": 0.0008109534974209964, "num_tokens": 129210550.0, "reward": 2.2975099086761475, "reward_std": 0.4883424937725067, "rewards/code_complexity_reward/mean": 0.9206054210662842, "rewards/code_complexity_reward/std": 0.10867564380168915, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 805, "step_time": 37.92645088583231 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 476.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 118.05859375, "completions/mean_terminated_length": 118.05859375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2446885174140334, "epoch": 0.45925925925925926, "frac_reward_zero_std": 0.546875, "grad_norm": 0.06229613721370697, "kl": 0.1655432073166594, "learning_rate": 3.2848315936751285e-06, "loss": 0.0008278699824586511, "num_tokens": 129343980.0, "reward": 2.2323732376098633, "reward_std": 0.4589782655239105, "rewards/code_complexity_reward/mean": 0.92138671875, "rewards/code_complexity_reward/std": 0.11250844597816467, "rewards/code_execution_reward/mean": 0.21875, "rewards/code_execution_reward/std": 0.41380295157432556, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 806, "step_time": 47.281650255434215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 124.658203125, "completions/mean_terminated_length": 124.658203125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22992215119302273, "epoch": 0.4598290598290598, "frac_reward_zero_std": 0.546875, "grad_norm": 0.054169196635484695, "kl": 0.16049430705606937, "learning_rate": 3.280107481253199e-06, "loss": 0.000802607974037528, "num_tokens": 129475917.0, "reward": 2.287353515625, "reward_std": 0.4671576917171478, "rewards/code_complexity_reward/mean": 0.925097644329071, "rewards/code_complexity_reward/std": 0.08503475785255432, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 807, "step_time": 39.45241067465395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 309.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 123.244140625, "completions/mean_terminated_length": 123.244140625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23191237985156476, "epoch": 0.4603988603988604, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05407465621829033, "kl": 0.15893248154316097, "learning_rate": 3.275380280742437e-06, "loss": 0.0007950245635583997, "num_tokens": 129606746.0, "reward": 2.296435594558716, "reward_std": 0.5401861667633057, "rewards/code_complexity_reward/mean": 0.9036132097244263, "rewards/code_complexity_reward/std": 0.1490400731563568, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 808, "step_time": 34.929995723068714 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 126.669921875, "completions/mean_terminated_length": 126.669921875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24430173682048917, "epoch": 0.46096866096866096, "frac_reward_zero_std": 0.59375, "grad_norm": 0.04634905606508255, "kl": 0.1684416759526357, "learning_rate": 3.270650010855667e-06, "loss": 0.000842415785882622, "num_tokens": 129741209.0, "reward": 2.2586426734924316, "reward_std": 0.48588672280311584, "rewards/code_complexity_reward/mean": 0.9178711175918579, "rewards/code_complexity_reward/std": 0.12628191709518433, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 809, "step_time": 35.552818179130554 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 117.671875, "completions/mean_terminated_length": 117.671875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2275457768701017, "epoch": 0.46153846153846156, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05473889037966728, "kl": 0.16165940021164715, "learning_rate": 3.2659166903178653e-06, "loss": 0.0008082098793238401, "num_tokens": 129870265.0, "reward": 2.3182129859924316, "reward_std": 0.48057326674461365, "rewards/code_complexity_reward/mean": 0.9215819835662842, "rewards/code_complexity_reward/std": 0.09086471050977707, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 810, "step_time": 36.52925143111497 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 272.0, "completions/max_terminated_length": 272.0, "completions/mean_length": 118.509765625, "completions/mean_terminated_length": 118.509765625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2415964431129396, "epoch": 0.4621082621082621, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06190142035484314, "kl": 0.17454838147386909, "learning_rate": 3.2611803378660827e-06, "loss": 0.000872763863299042, "num_tokens": 130000110.0, "reward": 2.2440919876098633, "reward_std": 0.4742338955402374, "rewards/code_complexity_reward/mean": 0.91943359375, "rewards/code_complexity_reward/std": 0.12720951437950134, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 811, "step_time": 55.02431126125157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 421.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 126.814453125, "completions/mean_terminated_length": 126.814453125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23090545600280166, "epoch": 0.46267806267806266, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05680426210165024, "kl": 0.15675365866627544, "learning_rate": 3.2564409722493752e-06, "loss": 0.0007838882738724351, "num_tokens": 130133527.0, "reward": 2.3292481899261475, "reward_std": 0.5112771987915039, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.11417873948812485, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 812, "step_time": 53.47978405188769 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 126.865234375, "completions/mean_terminated_length": 126.11154174804688, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22921849112026393, "epoch": 0.46324786324786327, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05567697063088417, "kl": 0.16336123016662896, "learning_rate": 3.2516986122287215e-06, "loss": 0.0008171017398126423, "num_tokens": 130269410.0, "reward": 2.242920160293579, "reward_std": 0.5118029117584229, "rewards/code_complexity_reward/mean": 0.9029296636581421, "rewards/code_complexity_reward/std": 0.15193213522434235, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 813, "step_time": 49.96121471375227 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 117.177734375, "completions/mean_terminated_length": 117.177734375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23461221251636744, "epoch": 0.4638176638176638, "frac_reward_zero_std": 0.515625, "grad_norm": 0.0540800616145134, "kl": 0.15843329310882837, "learning_rate": 3.2469532765769573e-06, "loss": 0.0007924389210529625, "num_tokens": 130396957.0, "reward": 2.33544921875, "reward_std": 0.5192250609397888, "rewards/code_complexity_reward/mean": 0.921679675579071, "rewards/code_complexity_reward/std": 0.12118426710367203, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 814, "step_time": 39.214739493094385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 229.0, "completions/max_terminated_length": 229.0, "completions/mean_length": 114.470703125, "completions/mean_terminated_length": 114.470703125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23656264296732843, "epoch": 0.46438746438746437, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06593368202447891, "kl": 0.1605719376821071, "learning_rate": 3.2422049840786986e-06, "loss": 0.0008028195588849485, "num_tokens": 130522070.0, "reward": 2.303515911102295, "reward_std": 0.48633718490600586, "rewards/code_complexity_reward/mean": 0.9261718392372131, "rewards/code_complexity_reward/std": 0.10355044156312943, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 815, "step_time": 30.749315267428756 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 121.603515625, "completions/mean_terminated_length": 121.603515625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2513493949081749, "epoch": 0.46495726495726497, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06093893572688103, "kl": 0.1627770719351247, "learning_rate": 3.237453753530262e-06, "loss": 0.0008138656849041581, "num_tokens": 130653035.0, "reward": 2.263232469558716, "reward_std": 0.47984012961387634, "rewards/code_complexity_reward/mean": 0.9214843511581421, "rewards/code_complexity_reward/std": 0.12008372694253922, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 816, "step_time": 37.0054689059034 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 303.0, "completions/max_terminated_length": 303.0, "completions/mean_length": 117.712890625, "completions/mean_terminated_length": 117.712890625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.24456154252402484, "epoch": 0.4655270655270655, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06421583145856857, "kl": 0.1639903699979186, "learning_rate": 3.232699603739598e-06, "loss": 0.0008202822064049542, "num_tokens": 130781784.0, "reward": 2.3328614234924316, "reward_std": 0.5109758973121643, "rewards/code_complexity_reward/mean": 0.9227539300918579, "rewards/code_complexity_reward/std": 0.12035837769508362, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 817, "step_time": 38.738131360150874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 127.912109375, "completions/mean_terminated_length": 127.16046905517578, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23471385333687067, "epoch": 0.46609686609686607, "frac_reward_zero_std": 0.4375, "grad_norm": 0.061310578137636185, "kl": 0.16118370182812214, "learning_rate": 3.2279425535262126e-06, "loss": 0.0008058095118030906, "num_tokens": 130916019.0, "reward": 2.3095703125, "reward_std": 0.4911234676837921, "rewards/code_complexity_reward/mean": 0.9133789539337158, "rewards/code_complexity_reward/std": 0.10328133404254913, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 818, "step_time": 49.96456774417311 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 275.0, "completions/max_terminated_length": 275.0, "completions/mean_length": 123.26171875, "completions/mean_terminated_length": 123.26171875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23644412122666836, "epoch": 0.4666666666666667, "frac_reward_zero_std": 0.484375, "grad_norm": 0.057328369468450546, "kl": 0.16482829733286053, "learning_rate": 3.2231826217210917e-06, "loss": 0.0008244050550274551, "num_tokens": 131047977.0, "reward": 2.327392578125, "reward_std": 0.5054138898849487, "rewards/code_complexity_reward/mean": 0.922167956829071, "rewards/code_complexity_reward/std": 0.11261778324842453, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 819, "step_time": 33.736144395545125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 123.86328125, "completions/mean_terminated_length": 122.3411865234375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24190334370359778, "epoch": 0.4672364672364672, "frac_reward_zero_std": 0.5, "grad_norm": 0.06024705991148949, "kl": 0.15500522567890584, "learning_rate": 3.2184198271666287e-06, "loss": 0.0007750588119961321, "num_tokens": 131181035.0, "reward": 2.30419921875, "reward_std": 0.5150684118270874, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.1264972686767578, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.019099153578281403, "step": 820, "step_time": 56.26218077261001 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 463.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 127.06640625, "completions/mean_terminated_length": 127.06640625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2276306222192943, "epoch": 0.46780626780626783, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05791917070746422, "kl": 0.1640171945327893, "learning_rate": 3.2136541887165497e-06, "loss": 0.000820106128230691, "num_tokens": 131315541.0, "reward": 2.312304735183716, "reward_std": 0.4876006245613098, "rewards/code_complexity_reward/mean": 0.9164062142372131, "rewards/code_complexity_reward/std": 0.10128515958786011, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 821, "step_time": 46.57104045525193 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 118.244140625, "completions/mean_terminated_length": 118.244140625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24995919363573194, "epoch": 0.4683760683760684, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06424742192029953, "kl": 0.16524368908721954, "learning_rate": 3.208885725235839e-06, "loss": 0.0008262796909548342, "num_tokens": 131448778.0, "reward": 2.4129395484924316, "reward_std": 0.5072370767593384, "rewards/code_complexity_reward/mean": 0.9295898675918579, "rewards/code_complexity_reward/std": 0.08101863414049149, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 822, "step_time": 64.28928547352552 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 329.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 119.291015625, "completions/mean_terminated_length": 119.291015625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2434918114449829, "epoch": 0.4689458689458689, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05993318930268288, "kl": 0.15696828439831734, "learning_rate": 3.2041144556006624e-06, "loss": 0.000784918898716569, "num_tokens": 131576639.0, "reward": 2.32080078125, "reward_std": 0.4939773678779602, "rewards/code_complexity_reward/mean": 0.9190429449081421, "rewards/code_complexity_reward/std": 0.09839048236608505, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 823, "step_time": 36.570344569161534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 367.0, "completions/max_terminated_length": 367.0, "completions/mean_length": 124.16796875, "completions/mean_terminated_length": 124.16796875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.24165827664546669, "epoch": 0.46951566951566953, "frac_reward_zero_std": 0.59375, "grad_norm": 0.048621468245983124, "kl": 0.17772278981283307, "learning_rate": 3.199340398698296e-06, "loss": 0.0008889199234545231, "num_tokens": 131710277.0, "reward": 2.253955364227295, "reward_std": 0.48750555515289307, "rewards/code_complexity_reward/mean": 0.9149413704872131, "rewards/code_complexity_reward/std": 0.1266627013683319, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 824, "step_time": 39.516601101495326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 338.0, "completions/mean_length": 122.76953125, "completions/mean_terminated_length": 122.00782775878906, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23824721644632518, "epoch": 0.4700854700854701, "frac_reward_zero_std": 0.578125, "grad_norm": 0.06041388958692551, "kl": 0.15961024328134954, "learning_rate": 3.194563573427047e-06, "loss": 0.0007980511290952563, "num_tokens": 131843311.0, "reward": 2.3197267055511475, "reward_std": 0.5055794715881348, "rewards/code_complexity_reward/mean": 0.9225585460662842, "rewards/code_complexity_reward/std": 0.11280056834220886, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 825, "step_time": 58.99342115782201 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 267.0, "completions/max_terminated_length": 267.0, "completions/mean_length": 119.19140625, "completions/mean_terminated_length": 119.19140625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23879164224490523, "epoch": 0.47065527065527063, "frac_reward_zero_std": 0.5, "grad_norm": 0.06493835896253586, "kl": 0.16496789443772286, "learning_rate": 3.1897839986961824e-06, "loss": 0.0008245189092122018, "num_tokens": 131973553.0, "reward": 2.3771486282348633, "reward_std": 0.5110475420951843, "rewards/code_complexity_reward/mean": 0.9207030534744263, "rewards/code_complexity_reward/std": 0.09606287628412247, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 826, "step_time": 33.30933533515781 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 122.580078125, "completions/mean_terminated_length": 122.580078125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23248945153318346, "epoch": 0.47122507122507123, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06282210350036621, "kl": 0.1673032563412562, "learning_rate": 3.185001693425855e-06, "loss": 0.0008363948436453938, "num_tokens": 132103642.0, "reward": 2.397510051727295, "reward_std": 0.5012751817703247, "rewards/code_complexity_reward/mean": 0.9336913824081421, "rewards/code_complexity_reward/std": 0.07593285292387009, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 827, "step_time": 35.84446067921817 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 484.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 118.31640625, "completions/mean_terminated_length": 118.31640625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2221780400723219, "epoch": 0.4717948717948718, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05183728411793709, "kl": 0.15936278633307666, "learning_rate": 3.1802166765470228e-06, "loss": 0.0007969246944412589, "num_tokens": 132233188.0, "reward": 2.4688477516174316, "reward_std": 0.5156752467155457, "rewards/code_complexity_reward/mean": 0.9225585460662842, "rewards/code_complexity_reward/std": 0.08760733157396317, "rewards/code_execution_reward/mean": 0.44921875, "rewards/code_execution_reward/std": 0.497901052236557, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 828, "step_time": 53.00308920163661 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 128.01171875, "completions/mean_terminated_length": 128.01171875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24355331505648792, "epoch": 0.4723646723646724, "frac_reward_zero_std": 0.421875, "grad_norm": 0.05672354996204376, "kl": 0.16878616076428443, "learning_rate": 3.175428967001381e-06, "loss": 0.0008438542718067765, "num_tokens": 132367066.0, "reward": 2.3213868141174316, "reward_std": 0.5126521587371826, "rewards/code_complexity_reward/mean": 0.9195312261581421, "rewards/code_complexity_reward/std": 0.12138862907886505, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 829, "step_time": 57.81430136691779 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 121.0078125, "completions/mean_terminated_length": 121.0078125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23369985539466143, "epoch": 0.47293447293447294, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06046457588672638, "kl": 0.16302936524152756, "learning_rate": 3.1706385837412822e-06, "loss": 0.0008150379871949553, "num_tokens": 132494582.0, "reward": 2.3265137672424316, "reward_std": 0.4952256679534912, "rewards/code_complexity_reward/mean": 0.9196288585662842, "rewards/code_complexity_reward/std": 0.09717365354299545, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 830, "step_time": 48.95263022463769 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 120.125, "completions/mean_terminated_length": 120.125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2302632317878306, "epoch": 0.4735042735042735, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05663071200251579, "kl": 0.15602274239063263, "learning_rate": 3.1658455457296644e-06, "loss": 0.0007800939492881298, "num_tokens": 132622214.0, "reward": 2.363330125808716, "reward_std": 0.5166212916374207, "rewards/code_complexity_reward/mean": 0.9139648675918579, "rewards/code_complexity_reward/std": 0.10706593841314316, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 831, "step_time": 54.281549440696836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 378.0, "completions/max_terminated_length": 378.0, "completions/mean_length": 116.240234375, "completions/mean_terminated_length": 116.240234375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23593853344209492, "epoch": 0.4740740740740741, "frac_reward_zero_std": 0.5, "grad_norm": 0.0621720515191555, "kl": 0.15896421717479825, "learning_rate": 3.161049871939972e-06, "loss": 0.000794709543697536, "num_tokens": 132752097.0, "reward": 2.3622560501098633, "reward_std": 0.4999241232872009, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.08389720320701599, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 832, "step_time": 40.74536345433444 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 120.81640625, "completions/mean_terminated_length": 120.81640625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2419892898760736, "epoch": 0.47464387464387464, "frac_reward_zero_std": 0.5625, "grad_norm": 0.054381050169467926, "kl": 0.15341022319626063, "learning_rate": 3.1562515813560852e-06, "loss": 0.0007669542683288455, "num_tokens": 132882547.0, "reward": 2.3565919399261475, "reward_std": 0.5074944496154785, "rewards/code_complexity_reward/mean": 0.929882824420929, "rewards/code_complexity_reward/std": 0.09567755460739136, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 833, "step_time": 36.866725945845246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 331.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 118.90625, "completions/mean_terminated_length": 118.90625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23636774788610637, "epoch": 0.4752136752136752, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05829969793558121, "kl": 0.18094918562565, "learning_rate": 3.151450692972243e-06, "loss": 0.0009051993256434798, "num_tokens": 133012411.0, "reward": 2.363769769668579, "reward_std": 0.5265465974807739, "rewards/code_complexity_reward/mean": 0.9219726324081421, "rewards/code_complexity_reward/std": 0.12259671837091446, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 834, "step_time": 45.8738777814433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 278.0, "completions/max_terminated_length": 278.0, "completions/mean_length": 120.728515625, "completions/mean_terminated_length": 120.728515625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23625498125329614, "epoch": 0.4757834757834758, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05062362924218178, "kl": 0.17314654786605388, "learning_rate": 3.1466472257929674e-06, "loss": 0.0008657848229631782, "num_tokens": 133143056.0, "reward": 2.30712890625, "reward_std": 0.48263686895370483, "rewards/code_complexity_reward/mean": 0.9209960699081421, "rewards/code_complexity_reward/std": 0.0954112708568573, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 835, "step_time": 40.970171798951924 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 121.4765625, "completions/mean_terminated_length": 121.4765625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23825355130247772, "epoch": 0.47635327635327634, "frac_reward_zero_std": 0.5, "grad_norm": 0.08221376687288284, "kl": 0.15841286233626306, "learning_rate": 3.141841198832988e-06, "loss": 0.0007918050396256149, "num_tokens": 133273988.0, "reward": 2.357470989227295, "reward_std": 0.5039463639259338, "rewards/code_complexity_reward/mean": 0.9288085699081421, "rewards/code_complexity_reward/std": 0.09536800533533096, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 836, "step_time": 36.154169355519116 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 333.0, "completions/max_terminated_length": 333.0, "completions/mean_length": 121.271484375, "completions/mean_terminated_length": 121.271484375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2389631257392466, "epoch": 0.47692307692307695, "frac_reward_zero_std": 0.5, "grad_norm": 0.052273914217948914, "kl": 0.1759709760081023, "learning_rate": 3.13703263111717e-06, "loss": 0.0008797477348707616, "num_tokens": 133406247.0, "reward": 2.288867473602295, "reward_std": 0.5195522308349609, "rewards/code_complexity_reward/mean": 0.9105468988418579, "rewards/code_complexity_reward/std": 0.13956212997436523, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 837, "step_time": 38.19533581286669 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 125.94140625, "completions/mean_terminated_length": 125.94140625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24405123991891742, "epoch": 0.4774928774928775, "frac_reward_zero_std": 0.46875, "grad_norm": 0.057351987808942795, "kl": 0.15761076426133513, "learning_rate": 3.1322215416804325e-06, "loss": 0.0007879084441810846, "num_tokens": 133540241.0, "reward": 2.259570598602295, "reward_std": 0.46195387840270996, "rewards/code_complexity_reward/mean": 0.9251953363418579, "rewards/code_complexity_reward/std": 0.1039341613650322, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 838, "step_time": 35.08511689119041 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 486.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 123.59375, "completions/mean_terminated_length": 123.59375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22802720358595252, "epoch": 0.47806267806267805, "frac_reward_zero_std": 0.578125, "grad_norm": 0.049226753413677216, "kl": 0.167504349257797, "learning_rate": 3.127407949567679e-06, "loss": 0.000837442115880549, "num_tokens": 133669841.0, "reward": 2.3008790016174316, "reward_std": 0.5120888948440552, "rewards/code_complexity_reward/mean": 0.912792980670929, "rewards/code_complexity_reward/std": 0.12911489605903625, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 839, "step_time": 60.11478852853179 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 429.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 123.6328125, "completions/mean_terminated_length": 123.6328125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.23462721146643162, "epoch": 0.47863247863247865, "frac_reward_zero_std": 0.53125, "grad_norm": 0.061296723783016205, "kl": 0.16301685490179807, "learning_rate": 3.12259187383372e-06, "loss": 0.0008152241352945566, "num_tokens": 133802013.0, "reward": 2.3722169399261475, "reward_std": 0.5333091020584106, "rewards/code_complexity_reward/mean": 0.9201172590255737, "rewards/code_complexity_reward/std": 0.12773652374744415, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 840, "step_time": 42.70382742583752 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 124.10546875, "completions/mean_terminated_length": 124.10546875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23160757450386882, "epoch": 0.4792022792022792, "frac_reward_zero_std": 0.375, "grad_norm": 0.06319032609462738, "kl": 0.14794572500977665, "learning_rate": 3.117773333543198e-06, "loss": 0.0007396883447654545, "num_tokens": 133934363.0, "reward": 2.3348634243011475, "reward_std": 0.5224044919013977, "rewards/code_complexity_reward/mean": 0.90966796875, "rewards/code_complexity_reward/std": 0.11594623327255249, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 841, "step_time": 39.666801234707236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 121.08203125, "completions/mean_terminated_length": 121.08203125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2271541114896536, "epoch": 0.47977207977207975, "frac_reward_zero_std": 0.5, "grad_norm": 0.054991986602544785, "kl": 0.17182912258431315, "learning_rate": 3.11295234777051e-06, "loss": 0.0008592131780460477, "num_tokens": 134064829.0, "reward": 2.3058595657348633, "reward_std": 0.47065016627311707, "rewards/code_complexity_reward/mean": 0.9324219226837158, "rewards/code_complexity_reward/std": 0.08682028949260712, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 842, "step_time": 41.18213636428118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 307.0, "completions/max_terminated_length": 307.0, "completions/mean_length": 121.44140625, "completions/mean_terminated_length": 121.44140625, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.23373173736035824, "epoch": 0.48034188034188036, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05667448416352272, "kl": 0.16176313592586666, "learning_rate": 3.1081289355997345e-06, "loss": 0.000808571232482791, "num_tokens": 134194271.0, "reward": 2.338379144668579, "reward_std": 0.5154716968536377, "rewards/code_complexity_reward/mean": 0.9170898199081421, "rewards/code_complexity_reward/std": 0.1170659139752388, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 843, "step_time": 38.37075015716255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 128.369140625, "completions/mean_terminated_length": 127.61839294433594, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22946361964568496, "epoch": 0.4809116809116809, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06776446849107742, "kl": 0.16314086550846696, "learning_rate": 3.1033031161245565e-06, "loss": 0.0008157033007591963, "num_tokens": 134331404.0, "reward": 2.3868653774261475, "reward_std": 0.5204827189445496, "rewards/code_complexity_reward/mean": 0.9205077886581421, "rewards/code_complexity_reward/std": 0.10082504898309708, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 844, "step_time": 52.17608444672078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 294.0, "completions/max_terminated_length": 294.0, "completions/mean_length": 118.419921875, "completions/mean_terminated_length": 118.419921875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23414282244630158, "epoch": 0.48148148148148145, "frac_reward_zero_std": 0.515625, "grad_norm": 0.06139364466071129, "kl": 0.16021072550211102, "learning_rate": 3.0984749084481856e-06, "loss": 0.0008009649463929236, "num_tokens": 134459563.0, "reward": 2.341601848602295, "reward_std": 0.5071796774864197, "rewards/code_complexity_reward/mean": 0.9239257574081421, "rewards/code_complexity_reward/std": 0.10430470108985901, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 845, "step_time": 37.723433080129325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 435.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 126.724609375, "completions/mean_terminated_length": 126.724609375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23554762732237577, "epoch": 0.48205128205128206, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06219238042831421, "kl": 0.1664989033015445, "learning_rate": 3.0936443316832904e-06, "loss": 0.0008324635564349592, "num_tokens": 134592310.0, "reward": 2.2857909202575684, "reward_std": 0.48085108399391174, "rewards/code_complexity_reward/mean": 0.92138671875, "rewards/code_complexity_reward/std": 0.10071481764316559, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 846, "step_time": 53.14124486129731 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 429.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 129.650390625, "completions/mean_terminated_length": 129.650390625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23338242410682142, "epoch": 0.4826210826210826, "frac_reward_zero_std": 0.5, "grad_norm": 0.05486099049448967, "kl": 0.16379877249710262, "learning_rate": 3.088811404951916e-06, "loss": 0.0008190415683202446, "num_tokens": 134728019.0, "reward": 2.392627000808716, "reward_std": 0.5040909647941589, "rewards/code_complexity_reward/mean": 0.9198242425918579, "rewards/code_complexity_reward/std": 0.0817648321390152, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 847, "step_time": 43.02325729560107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 123.466796875, "completions/mean_terminated_length": 123.466796875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23903533979319036, "epoch": 0.4831908831908832, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05445999279618263, "kl": 0.15786757902242243, "learning_rate": 3.083976147385409e-06, "loss": 0.0007893494330346584, "num_tokens": 134864162.0, "reward": 2.35498046875, "reward_std": 0.5279210805892944, "rewards/code_complexity_reward/mean": 0.917773425579071, "rewards/code_complexity_reward/std": 0.12016183882951736, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49365234375, "rewards/xmlcount_reward_func/std": 0.054370980709791183, "step": 848, "step_time": 46.385288843885064 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 377.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 129.521484375, "completions/mean_terminated_length": 129.521484375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22968919738195837, "epoch": 0.48376068376068376, "frac_reward_zero_std": 0.359375, "grad_norm": 0.06353406608104706, "kl": 0.1563997466582805, "learning_rate": 3.0791385781243434e-06, "loss": 0.0007824560161679983, "num_tokens": 134995757.0, "reward": 2.3515138626098633, "reward_std": 0.4973057508468628, "rewards/code_complexity_reward/mean": 0.9236327409744263, "rewards/code_complexity_reward/std": 0.0926775261759758, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 849, "step_time": 45.30305808130652 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 302.0, "completions/mean_length": 124.994140625, "completions/mean_terminated_length": 123.47647857666016, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2426522746682167, "epoch": 0.4843304843304843, "frac_reward_zero_std": 0.578125, "grad_norm": 0.050249531865119934, "kl": 0.163931856979616, "learning_rate": 3.0742987163184445e-06, "loss": 0.0008197090355679393, "num_tokens": 135129650.0, "reward": 2.2945799827575684, "reward_std": 0.517108142375946, "rewards/code_complexity_reward/mean": 0.9181640148162842, "rewards/code_complexity_reward/std": 0.1328674703836441, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 850, "step_time": 52.76682075392455 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 122.91796875, "completions/mean_terminated_length": 122.15655517578125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.22612838889472187, "epoch": 0.4849002849002849, "frac_reward_zero_std": 0.421875, "grad_norm": 0.061633992940187454, "kl": 0.1532340762205422, "learning_rate": 3.0694565811265115e-06, "loss": 0.0007663713186047971, "num_tokens": 135260224.0, "reward": 2.36865234375, "reward_std": 0.5258915424346924, "rewards/code_complexity_reward/mean": 0.919726550579071, "rewards/code_complexity_reward/std": 0.11989690363407135, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 851, "step_time": 49.81168276909739 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 376.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 124.16796875, "completions/mean_terminated_length": 124.16796875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2267418501432985, "epoch": 0.48547008547008547, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05859503895044327, "kl": 0.16126981121487916, "learning_rate": 3.0646121917163435e-06, "loss": 0.0008062483393587172, "num_tokens": 135393438.0, "reward": 2.39794921875, "reward_std": 0.5029081106185913, "rewards/code_complexity_reward/mean": 0.9249023199081421, "rewards/code_complexity_reward/std": 0.07139914482831955, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 852, "step_time": 39.83631788752973 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 288.0, "completions/max_terminated_length": 288.0, "completions/mean_length": 123.451171875, "completions/mean_terminated_length": 123.451171875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.24157170858234167, "epoch": 0.486039886039886, "frac_reward_zero_std": 0.5625, "grad_norm": 0.051377732306718826, "kl": 0.15704422153066844, "learning_rate": 3.0597655672646635e-06, "loss": 0.0007850005058571696, "num_tokens": 135522741.0, "reward": 2.268798828125, "reward_std": 0.475564181804657, "rewards/code_complexity_reward/mean": 0.919140636920929, "rewards/code_complexity_reward/std": 0.11388377845287323, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 853, "step_time": 43.574511219747365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 442.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 126.59765625, "completions/mean_terminated_length": 126.59765625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23421777365729213, "epoch": 0.4866096866096866, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05634952709078789, "kl": 0.16179845051374286, "learning_rate": 3.0549167269570427e-06, "loss": 0.0008090406190603971, "num_tokens": 135652799.0, "reward": 2.335449457168579, "reward_std": 0.5315839052200317, "rewards/code_complexity_reward/mean": 0.9161132574081421, "rewards/code_complexity_reward/std": 0.14079424738883972, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 854, "step_time": 44.32468921225518 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 486.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 126.412109375, "completions/mean_terminated_length": 126.412109375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24252650002017617, "epoch": 0.48717948717948717, "frac_reward_zero_std": 0.453125, "grad_norm": 0.0557769276201725, "kl": 0.166861932259053, "learning_rate": 3.050065689987822e-06, "loss": 0.0008344381349161267, "num_tokens": 135786154.0, "reward": 2.3568849563598633, "reward_std": 0.5151403546333313, "rewards/code_complexity_reward/mean": 0.91748046875, "rewards/code_complexity_reward/std": 0.10773463547229767, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 855, "step_time": 56.90680915489793 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 416.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 126.517578125, "completions/mean_terminated_length": 126.517578125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2298741308040917, "epoch": 0.4877492877492878, "frac_reward_zero_std": 0.59375, "grad_norm": 0.04984140768647194, "kl": 0.15690729580819607, "learning_rate": 3.0452124755600376e-06, "loss": 0.0007844107458367944, "num_tokens": 135921371.0, "reward": 2.3600099086761475, "reward_std": 0.4813586175441742, "rewards/code_complexity_reward/mean": 0.930371105670929, "rewards/code_complexity_reward/std": 0.050636447966098785, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 856, "step_time": 42.66845160257071 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 396.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 125.9921875, "completions/mean_terminated_length": 125.9921875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23362051392905414, "epoch": 0.4883190883190883, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05536293983459473, "kl": 0.1594585239654407, "learning_rate": 3.0403571028853487e-06, "loss": 0.0007973090978339314, "num_tokens": 136055223.0, "reward": 2.291259765625, "reward_std": 0.4974622428417206, "rewards/code_complexity_reward/mean": 0.9182616472244263, "rewards/code_complexity_reward/std": 0.12108246982097626, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 857, "step_time": 48.926323905587196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 118.28515625, "completions/mean_terminated_length": 118.28515625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2352384424302727, "epoch": 0.4888888888888889, "frac_reward_zero_std": 0.53125, "grad_norm": 0.056877098977565765, "kl": 0.18672167707700282, "learning_rate": 3.0354995911839543e-06, "loss": 0.0009336061775684357, "num_tokens": 136183009.0, "reward": 2.325000286102295, "reward_std": 0.5171141624450684, "rewards/code_complexity_reward/mean": 0.9125000238418579, "rewards/code_complexity_reward/std": 0.12469139695167542, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 858, "step_time": 46.77554509602487 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 326.0, "completions/max_terminated_length": 326.0, "completions/mean_length": 123.005859375, "completions/mean_terminated_length": 123.005859375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24652943946421146, "epoch": 0.4894586894586895, "frac_reward_zero_std": 0.5, "grad_norm": 0.05700213834643364, "kl": 0.15527623193338513, "learning_rate": 3.030639959684522e-06, "loss": 0.0007765267509967089, "num_tokens": 136319164.0, "reward": 2.3182129859924316, "reward_std": 0.4723019003868103, "rewards/code_complexity_reward/mean": 0.925488293170929, "rewards/code_complexity_reward/std": 0.06759610027074814, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 859, "step_time": 48.113984196446836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 117.072265625, "completions/mean_terminated_length": 117.072265625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2291941181756556, "epoch": 0.49002849002849, "frac_reward_zero_std": 0.46875, "grad_norm": 0.058606233447790146, "kl": 0.15173486666753888, "learning_rate": 3.0257782276241123e-06, "loss": 0.0007586788269691169, "num_tokens": 136447009.0, "reward": 2.3934082984924316, "reward_std": 0.4991644322872162, "rewards/code_complexity_reward/mean": 0.9295898675918579, "rewards/code_complexity_reward/std": 0.07837904989719391, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 860, "step_time": 36.337849175557494 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 320.0, "completions/max_terminated_length": 320.0, "completions/mean_length": 124.138671875, "completions/mean_terminated_length": 124.138671875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2243297309614718, "epoch": 0.4905982905982906, "frac_reward_zero_std": 0.53125, "grad_norm": 0.054403018206357956, "kl": 0.15745543607044965, "learning_rate": 3.0209144142480982e-06, "loss": 0.0007872540736570954, "num_tokens": 136579448.0, "reward": 2.3498048782348633, "reward_std": 0.5204055309295654, "rewards/code_complexity_reward/mean": 0.915820300579071, "rewards/code_complexity_reward/std": 0.11879906803369522, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 861, "step_time": 36.131402785889804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 122.76171875, "completions/mean_terminated_length": 122.76171875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23088604700751603, "epoch": 0.4911680911680912, "frac_reward_zero_std": 0.484375, "grad_norm": 0.1087946742773056, "kl": 0.1543750544078648, "learning_rate": 3.0160485388100935e-06, "loss": 0.0007720161229372025, "num_tokens": 136709622.0, "reward": 2.277587890625, "reward_std": 0.5103402137756348, "rewards/code_complexity_reward/mean": 0.9192383289337158, "rewards/code_complexity_reward/std": 0.1397019773721695, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 862, "step_time": 36.44544576201588 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 118.34765625, "completions/mean_terminated_length": 118.34765625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23440989409573376, "epoch": 0.49173789173789173, "frac_reward_zero_std": 0.46875, "grad_norm": 0.055307336151599884, "kl": 0.1544677393976599, "learning_rate": 3.0111806205718752e-06, "loss": 0.0007722678128629923, "num_tokens": 136837632.0, "reward": 2.3561525344848633, "reward_std": 0.4844007194042206, "rewards/code_complexity_reward/mean": 0.928027331829071, "rewards/code_complexity_reward/std": 0.06774690002202988, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 863, "step_time": 74.1840892424807 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 444.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 120.400390625, "completions/mean_terminated_length": 120.400390625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2399555032607168, "epoch": 0.49230769230769234, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06051776558160782, "kl": 0.15131939854472876, "learning_rate": 3.0063106788033043e-06, "loss": 0.0007565978448837996, "num_tokens": 136967069.0, "reward": 2.3517091274261475, "reward_std": 0.5326688289642334, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.1320973038673401, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 864, "step_time": 52.28357925731689 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 471.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 124.357421875, "completions/mean_terminated_length": 124.357421875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.24355984572321177, "epoch": 0.4928774928774929, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05467887222766876, "kl": 0.15514207258820534, "learning_rate": 3.0014387327822536e-06, "loss": 0.0007757723797112703, "num_tokens": 137097820.0, "reward": 2.2432618141174316, "reward_std": 0.4937956929206848, "rewards/code_complexity_reward/mean": 0.9137694835662842, "rewards/code_complexity_reward/std": 0.13884076476097107, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 865, "step_time": 45.9951028265059 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 116.720703125, "completions/mean_terminated_length": 116.720703125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24013491440564394, "epoch": 0.49344729344729343, "frac_reward_zero_std": 0.546875, "grad_norm": 0.056835636496543884, "kl": 0.1624889657832682, "learning_rate": 2.996564801794531e-06, "loss": 0.0008125060121528804, "num_tokens": 137224821.0, "reward": 2.3689942359924316, "reward_std": 0.5147119760513306, "rewards/code_complexity_reward/mean": 0.9217773675918579, "rewards/code_complexity_reward/std": 0.1053340807557106, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 866, "step_time": 35.33932889159769 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 125.720703125, "completions/mean_terminated_length": 124.96477508544922, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23610746092163026, "epoch": 0.49401709401709404, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05746598169207573, "kl": 0.15565543517004699, "learning_rate": 2.9916889051338e-06, "loss": 0.0007782797329127789, "num_tokens": 137356446.0, "reward": 2.376025676727295, "reward_std": 0.5218678116798401, "rewards/code_complexity_reward/mean": 0.9237304925918579, "rewards/code_complexity_reward/std": 0.11155522614717484, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 867, "step_time": 48.161194579675794 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 353.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 118.666015625, "completions/mean_terminated_length": 118.666015625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23754809773527086, "epoch": 0.4945868945868946, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06052958965301514, "kl": 0.16756768303457648, "learning_rate": 2.9868110621015068e-06, "loss": 0.000837769650388509, "num_tokens": 137482979.0, "reward": 2.367969036102295, "reward_std": 0.5399041175842285, "rewards/code_complexity_reward/mean": 0.9144531488418579, "rewards/code_complexity_reward/std": 0.13251331448554993, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 868, "step_time": 37.601195628754795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 435.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 123.73046875, "completions/mean_terminated_length": 123.73046875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23037662939168513, "epoch": 0.49515669515669514, "frac_reward_zero_std": 0.484375, "grad_norm": 0.057185254991054535, "kl": 0.14960128802340478, "learning_rate": 2.981931292006801e-06, "loss": 0.0007478746701963246, "num_tokens": 137613417.0, "reward": 2.4408204555511475, "reward_std": 0.5211887359619141, "rewards/code_complexity_reward/mean": 0.9267578125, "rewards/code_complexity_reward/std": 0.09208837896585464, "rewards/code_execution_reward/mean": 0.41796875, "rewards/code_execution_reward/std": 0.4937073290348053, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 869, "step_time": 46.8903083903715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 126.474609375, "completions/mean_terminated_length": 126.474609375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23521424131467938, "epoch": 0.49572649572649574, "frac_reward_zero_std": 0.484375, "grad_norm": 0.056714195758104324, "kl": 0.1567023084498942, "learning_rate": 2.9770496141664617e-06, "loss": 0.0007834451971575618, "num_tokens": 137746532.0, "reward": 2.303515672683716, "reward_std": 0.48335033655166626, "rewards/code_complexity_reward/mean": 0.9203125238418579, "rewards/code_complexity_reward/std": 0.09214501827955246, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 870, "step_time": 56.92607970815152 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 121.958984375, "completions/mean_terminated_length": 121.958984375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23130494984798133, "epoch": 0.4962962962962963, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05754624679684639, "kl": 0.15190447645727545, "learning_rate": 2.972166047904819e-06, "loss": 0.000759373651817441, "num_tokens": 137878951.0, "reward": 2.3758301734924316, "reward_std": 0.5326640009880066, "rewards/code_complexity_reward/mean": 0.9176757335662842, "rewards/code_complexity_reward/std": 0.12584379315376282, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 871, "step_time": 42.2237835759297 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 130.4375, "completions/mean_terminated_length": 130.4375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23592909472063184, "epoch": 0.49686609686609684, "frac_reward_zero_std": 0.40625, "grad_norm": 0.08451095223426819, "kl": 0.16750196472276002, "learning_rate": 2.967280612553679e-06, "loss": 0.0008375007892027497, "num_tokens": 138015023.0, "reward": 2.273193597793579, "reward_std": 0.48680275678634644, "rewards/code_complexity_reward/mean": 0.9146484136581421, "rewards/code_complexity_reward/std": 0.11230668425559998, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 872, "step_time": 58.62397459987551 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 440.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 119.677734375, "completions/mean_terminated_length": 119.677734375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23488785070367157, "epoch": 0.49743589743589745, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05557798594236374, "kl": 0.16399833629839122, "learning_rate": 2.9623933274522464e-06, "loss": 0.0008201766759157181, "num_tokens": 138148882.0, "reward": 2.3563477993011475, "reward_std": 0.4897119104862213, "rewards/code_complexity_reward/mean": 0.93310546875, "rewards/code_complexity_reward/std": 0.07599776238203049, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 873, "step_time": 45.2359454119578 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 126.361328125, "completions/mean_terminated_length": 126.361328125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.23547398881055415, "epoch": 0.498005698005698, "frac_reward_zero_std": 0.546875, "grad_norm": 0.09473158419132233, "kl": 0.16381482291035354, "learning_rate": 2.9575042119470487e-06, "loss": 0.0008191816741600633, "num_tokens": 138280859.0, "reward": 2.2032227516174316, "reward_std": 0.4510418474674225, "rewards/code_complexity_reward/mean": 0.9137694835662842, "rewards/code_complexity_reward/std": 0.12359127402305603, "rewards/code_execution_reward/mean": 0.197265625, "rewards/code_execution_reward/std": 0.3983237147331238, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 874, "step_time": 69.30759833287448 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 116.12109375, "completions/mean_terminated_length": 116.12109375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22437715763226151, "epoch": 0.4985754985754986, "frac_reward_zero_std": 0.5, "grad_norm": 0.061338864266872406, "kl": 0.1581438009161502, "learning_rate": 2.9526132853918587e-06, "loss": 0.0007905212696641684, "num_tokens": 138407473.0, "reward": 2.454882860183716, "reward_std": 0.5151193737983704, "rewards/code_complexity_reward/mean": 0.9281250238418579, "rewards/code_complexity_reward/std": 0.07636558264493942, "rewards/code_execution_reward/mean": 0.4296875, "rewards/code_execution_reward/std": 0.4955156147480011, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 875, "step_time": 35.64139434788376 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 441.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 119.4296875, "completions/mean_terminated_length": 119.4296875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2315079914405942, "epoch": 0.49914529914529915, "frac_reward_zero_std": 0.46875, "grad_norm": 0.0555909164249897, "kl": 0.16251009504776448, "learning_rate": 2.9477205671476183e-06, "loss": 0.0008126345928758383, "num_tokens": 138537453.0, "reward": 2.319092035293579, "reward_std": 0.47397324442863464, "rewards/code_complexity_reward/mean": 0.9292968511581421, "rewards/code_complexity_reward/std": 0.07707402110099792, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 876, "step_time": 51.56293382868171 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 123.966796875, "completions/mean_terminated_length": 123.966796875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.24476633151061833, "epoch": 0.4997150997150997, "frac_reward_zero_std": 0.453125, "grad_norm": 0.062038447707891464, "kl": 0.15725392382591963, "learning_rate": 2.942826076582362e-06, "loss": 0.0007863215869292617, "num_tokens": 138669396.0, "reward": 2.28125, "reward_std": 0.4970264136791229, "rewards/code_complexity_reward/mean": 0.9195312261581421, "rewards/code_complexity_reward/std": 0.12066099047660828, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 877, "step_time": 45.80029018595815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 368.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 126.40625, "completions/mean_terminated_length": 126.40625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23315021628513932, "epoch": 0.5002849002849002, "frac_reward_zero_std": 0.484375, "grad_norm": 0.054015662521123886, "kl": 0.16556875582318753, "learning_rate": 2.9379298330711393e-06, "loss": 0.0008280999027192593, "num_tokens": 138801492.0, "reward": 2.2735841274261475, "reward_std": 0.520134449005127, "rewards/code_complexity_reward/mean": 0.9083983898162842, "rewards/code_complexity_reward/std": 0.14873166382312775, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 878, "step_time": 48.64850107580423 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 468.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 124.11328125, "completions/mean_terminated_length": 124.11328125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24772832170128822, "epoch": 0.5008547008547009, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05441240593791008, "kl": 0.16726462228689343, "learning_rate": 2.9330318559959402e-06, "loss": 0.0008367159171029925, "num_tokens": 138934014.0, "reward": 2.3023438453674316, "reward_std": 0.4948135018348694, "rewards/code_complexity_reward/mean": 0.9237304329872131, "rewards/code_complexity_reward/std": 0.10992058366537094, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 879, "step_time": 45.39573998935521 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 372.0, "completions/max_terminated_length": 372.0, "completions/mean_length": 124.322265625, "completions/mean_terminated_length": 124.322265625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22162195155397058, "epoch": 0.5014245014245015, "frac_reward_zero_std": 0.53125, "grad_norm": 0.0682344138622284, "kl": 0.15668290492612869, "learning_rate": 2.9281321647456174e-06, "loss": 0.0007834680145606399, "num_tokens": 139062651.0, "reward": 2.3287110328674316, "reward_std": 0.512322187423706, "rewards/code_complexity_reward/mean": 0.9178711175918579, "rewards/code_complexity_reward/std": 0.11225303262472153, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 880, "step_time": 50.147632125765085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 310.0, "completions/max_terminated_length": 310.0, "completions/mean_length": 114.466796875, "completions/mean_terminated_length": 114.466796875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.22606166056357324, "epoch": 0.501994301994302, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05540899187326431, "kl": 0.14552191039547324, "learning_rate": 2.9232307787158067e-06, "loss": 0.000727676204405725, "num_tokens": 139187626.0, "reward": 2.400439739227295, "reward_std": 0.5127542614936829, "rewards/code_complexity_reward/mean": 0.9249023199081421, "rewards/code_complexity_reward/std": 0.09544411301612854, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 881, "step_time": 35.49229808431119 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 125.236328125, "completions/mean_terminated_length": 125.236328125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23022629180923104, "epoch": 0.5025641025641026, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06434732675552368, "kl": 0.16125723428558558, "learning_rate": 2.9183277173088555e-06, "loss": 0.0008064790745265782, "num_tokens": 139321115.0, "reward": 2.3133790493011475, "reward_std": 0.4815598726272583, "rewards/code_complexity_reward/mean": 0.92138671875, "rewards/code_complexity_reward/std": 0.09171464294195175, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 882, "step_time": 53.15292826667428 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 126.814453125, "completions/mean_terminated_length": 126.814453125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23273616307415068, "epoch": 0.5031339031339032, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06025048717856407, "kl": 0.17072443396318704, "learning_rate": 2.913422999933741e-06, "loss": 0.0008538829279132187, "num_tokens": 139454460.0, "reward": 2.2542970180511475, "reward_std": 0.4769732654094696, "rewards/code_complexity_reward/mean": 0.9142577648162842, "rewards/code_complexity_reward/std": 0.11862650513648987, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 883, "step_time": 37.09280993510038 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 423.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 126.791015625, "completions/mean_terminated_length": 126.791015625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2476087745744735, "epoch": 0.5037037037037037, "frac_reward_zero_std": 0.546875, "grad_norm": 0.04961567744612694, "kl": 0.1701256714295596, "learning_rate": 2.9085166460059976e-06, "loss": 0.0008510378538630903, "num_tokens": 139590073.0, "reward": 2.2460451126098633, "reward_std": 0.47199997305870056, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.11957848817110062, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 884, "step_time": 43.34682826232165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 341.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 123.498046875, "completions/mean_terminated_length": 123.498046875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23914738604798913, "epoch": 0.5042735042735043, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05775998532772064, "kl": 0.1567950325552374, "learning_rate": 2.903608674947637e-06, "loss": 0.0007839404279366136, "num_tokens": 139722424.0, "reward": 2.3543946743011475, "reward_std": 0.5074062347412109, "rewards/code_complexity_reward/mean": 0.92138671875, "rewards/code_complexity_reward/std": 0.09675068408250809, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 885, "step_time": 38.932203833945096 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 127.177734375, "completions/mean_terminated_length": 126.4246597290039, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2410367166157812, "epoch": 0.5048433048433049, "frac_reward_zero_std": 0.5, "grad_norm": 0.050995469093322754, "kl": 0.1638046910520643, "learning_rate": 2.898699106187073e-06, "loss": 0.0008189224172383547, "num_tokens": 139857467.0, "reward": 2.3343262672424316, "reward_std": 0.5194595456123352, "rewards/code_complexity_reward/mean": 0.9142577648162842, "rewards/code_complexity_reward/std": 0.11989817768335342, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 886, "step_time": 50.040653553791344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 300.0, "completions/max_terminated_length": 300.0, "completions/mean_length": 115.98046875, "completions/mean_terminated_length": 115.98046875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.25372263439930975, "epoch": 0.5054131054131055, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05855691060423851, "kl": 0.16221393446903676, "learning_rate": 2.8937879591590416e-06, "loss": 0.000811481149867177, "num_tokens": 139986577.0, "reward": 2.3553223609924316, "reward_std": 0.537804901599884, "rewards/code_complexity_reward/mean": 0.9188476800918579, "rewards/code_complexity_reward/std": 0.13873639702796936, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 887, "step_time": 35.596938233822584 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 124.650390625, "completions/mean_terminated_length": 124.650390625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2255871829111129, "epoch": 0.505982905982906, "frac_reward_zero_std": 0.453125, "grad_norm": 0.0626867339015007, "kl": 0.1573458316270262, "learning_rate": 2.888875253304531e-06, "loss": 0.000787031720392406, "num_tokens": 140117086.0, "reward": 2.3541016578674316, "reward_std": 0.4952961504459381, "rewards/code_complexity_reward/mean": 0.9219726324081421, "rewards/code_complexity_reward/std": 0.08252743631601334, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 888, "step_time": 44.676345543935895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 381.0, "completions/max_terminated_length": 381.0, "completions/mean_length": 125.072265625, "completions/mean_terminated_length": 125.072265625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23894690838642418, "epoch": 0.5065527065527066, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05793754756450653, "kl": 0.15744627872481942, "learning_rate": 2.8839610080706966e-06, "loss": 0.0007874093716964126, "num_tokens": 140249307.0, "reward": 2.3231444358825684, "reward_std": 0.5369028449058533, "rewards/code_complexity_reward/mean": 0.91259765625, "rewards/code_complexity_reward/std": 0.14595822989940643, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 889, "step_time": 41.59524639137089 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 468.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 128.732421875, "completions/mean_terminated_length": 128.732421875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.23367923218756914, "epoch": 0.5071225071225072, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06319049745798111, "kl": 0.14332594175357372, "learning_rate": 2.8790452429107873e-06, "loss": 0.0007168445736169815, "num_tokens": 140382242.0, "reward": 2.2543458938598633, "reward_std": 0.45994868874549866, "rewards/code_complexity_reward/mean": 0.92138671875, "rewards/code_complexity_reward/std": 0.10196998715400696, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 890, "step_time": 46.75133796501905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 441.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 124.201171875, "completions/mean_terminated_length": 124.201171875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23423447157256305, "epoch": 0.5076923076923077, "frac_reward_zero_std": 0.515625, "grad_norm": 0.060652460902929306, "kl": 0.16324298083782196, "learning_rate": 2.8741279772840706e-06, "loss": 0.0008163702441379428, "num_tokens": 140515017.0, "reward": 2.291259765625, "reward_std": 0.48365962505340576, "rewards/code_complexity_reward/mean": 0.9180663824081421, "rewards/code_complexity_reward/std": 0.10926175862550735, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 891, "step_time": 46.94731149636209 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 353.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 125.701171875, "completions/mean_terminated_length": 125.701171875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2395166226197034, "epoch": 0.5082621082621083, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05499669536948204, "kl": 0.16481849120464176, "learning_rate": 2.869209230655753e-06, "loss": 0.0008240279275923967, "num_tokens": 140648328.0, "reward": 2.239062786102295, "reward_std": 0.4369492828845978, "rewards/code_complexity_reward/mean": 0.9288085699081421, "rewards/code_complexity_reward/std": 0.0877259224653244, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 892, "step_time": 45.37575795035809 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 351.0, "completions/max_terminated_length": 351.0, "completions/mean_length": 120.0546875, "completions/mean_terminated_length": 120.0546875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22345745097845793, "epoch": 0.5088319088319089, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05798332393169403, "kl": 0.15362578281201422, "learning_rate": 2.8642890224969026e-06, "loss": 0.0007683713338337839, "num_tokens": 140779196.0, "reward": 2.4224610328674316, "reward_std": 0.5170700550079346, "rewards/code_complexity_reward/mean": 0.9279296398162842, "rewards/code_complexity_reward/std": 0.08958051353693008, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 893, "step_time": 45.58651671838015 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 125.259765625, "completions/mean_terminated_length": 125.259765625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23312891460955143, "epoch": 0.5094017094017094, "frac_reward_zero_std": 0.609375, "grad_norm": 0.05297663435339928, "kl": 0.1609012526459992, "learning_rate": 2.859367372284375e-06, "loss": 0.0008047735318541527, "num_tokens": 140912113.0, "reward": 2.3553223609924316, "reward_std": 0.5019797086715698, "rewards/code_complexity_reward/mean": 0.9266601800918579, "rewards/code_complexity_reward/std": 0.09553217142820358, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 894, "step_time": 39.66415669117123 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 117.3359375, "completions/mean_terminated_length": 117.3359375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23281597322784364, "epoch": 0.50997150997151, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05560997128486633, "kl": 0.15701821632683277, "learning_rate": 2.854444299500733e-06, "loss": 0.0007850364781916142, "num_tokens": 141038581.0, "reward": 2.3866701126098633, "reward_std": 0.5372288227081299, "rewards/code_complexity_reward/mean": 0.9208984375, "rewards/code_complexity_reward/std": 0.125757098197937, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 895, "step_time": 56.39173480682075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 134.416015625, "completions/mean_terminated_length": 133.67710876464844, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2420894994866103, "epoch": 0.5105413105413106, "frac_reward_zero_std": 0.53125, "grad_norm": 0.052813947200775146, "kl": 0.1557091197464615, "learning_rate": 2.8495198236341693e-06, "loss": 0.0007788059883750975, "num_tokens": 141176994.0, "reward": 2.242969036102295, "reward_std": 0.49150505661964417, "rewards/code_complexity_reward/mean": 0.9122070074081421, "rewards/code_complexity_reward/std": 0.1388462781906128, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 896, "step_time": 49.16557374317199 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 345.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 120.138671875, "completions/mean_terminated_length": 120.138671875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24256004998460412, "epoch": 0.5111111111111111, "frac_reward_zero_std": 0.5625, "grad_norm": 0.06415161490440369, "kl": 0.16860516439191997, "learning_rate": 2.844593964178433e-06, "loss": 0.0008432421600446105, "num_tokens": 141307993.0, "reward": 2.278564453125, "reward_std": 0.48340436816215515, "rewards/code_complexity_reward/mean": 0.9219726324081421, "rewards/code_complexity_reward/std": 0.11399443447589874, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 897, "step_time": 38.097504490986466 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 275.0, "completions/mean_length": 123.849609375, "completions/mean_terminated_length": 123.09001922607422, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23552294657565653, "epoch": 0.5116809116809117, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05937613546848297, "kl": 0.15869734878651798, "learning_rate": 2.8396667406327507e-06, "loss": 0.0007933324086479843, "num_tokens": 141438780.0, "reward": 2.298388719558716, "reward_std": 0.508257269859314, "rewards/code_complexity_reward/mean": 0.9144531488418579, "rewards/code_complexity_reward/std": 0.12642957270145416, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 898, "step_time": 48.34727467224002 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 296.0, "completions/max_terminated_length": 296.0, "completions/mean_length": 122.541015625, "completions/mean_terminated_length": 122.541015625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23773981537669897, "epoch": 0.5122507122507123, "frac_reward_zero_std": 0.546875, "grad_norm": 0.057068243622779846, "kl": 0.16381880291737616, "learning_rate": 2.834738172501746e-06, "loss": 0.0008191669476218522, "num_tokens": 141571729.0, "reward": 2.394287109375, "reward_std": 0.5090212225914001, "rewards/code_complexity_reward/mean": 0.9195312261581421, "rewards/code_complexity_reward/std": 0.08847133070230484, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 899, "step_time": 38.400874150916934 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 120.92578125, "completions/mean_terminated_length": 120.92578125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23702305858023465, "epoch": 0.5128205128205128, "frac_reward_zero_std": 0.5, "grad_norm": 0.06130313500761986, "kl": 0.15483945445157588, "learning_rate": 2.8298082792953672e-06, "loss": 0.0007744452450424433, "num_tokens": 141704803.0, "reward": 2.352099895477295, "reward_std": 0.5043419599533081, "rewards/code_complexity_reward/mean": 0.9214843511581421, "rewards/code_complexity_reward/std": 0.09830930083990097, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 900, "step_time": 46.82710315659642 }, { "epoch": 0.5128205128205128, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 176.16, "eval_completions/max_terminated_length": 176.16, "eval_completions/mean_length": 125.8775, "eval_completions/mean_terminated_length": 125.8775, "eval_completions/min_length": 90.3, "eval_completions/min_terminated_length": 90.3, "eval_entropy": 0.23226853989064694, "eval_frac_reward_zero_std": 0.44, "eval_kl": 0.15559350028634072, "eval_loss": 0.0007787066861055791, "eval_num_tokens": 141704803.0, "eval_reward": 2.3140938758850096, "eval_reward_std": 0.23401953654363752, "eval_rewards/code_complexity_reward/mean": 0.9154999846220017, "eval_rewards/code_complexity_reward/std": 0.04711962735280394, "eval_rewards/code_execution_reward/mean": 0.3075, "eval_rewards/code_execution_reward/std": 0.17662842750549315, "eval_rewards/code_syntax_reward/mean": 0.49125, "eval_rewards/code_syntax_reward/std": 0.022033182233572007, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.49984375, "eval_rewards/xmlcount_reward_func/std": 0.0004419417306780815, "eval_runtime": 795.5128, "eval_samples_per_second": 0.126, "eval_steps_per_second": 0.016, "step": 900 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 387.0, "completions/max_terminated_length": 387.0, "completions/mean_length": 121.69921875, "completions/mean_terminated_length": 121.69921875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2326793409883976, "epoch": 0.5133903133903134, "frac_reward_zero_std": 0.390625, "grad_norm": 0.0666738748550415, "kl": 0.15413477027323097, "learning_rate": 2.8248770805288083e-06, "loss": 0.0007705667521804571, "num_tokens": 141835697.0, "reward": 2.3055665493011475, "reward_std": 0.4847208261489868, "rewards/code_complexity_reward/mean": 0.9194335341453552, "rewards/code_complexity_reward/std": 0.09816452115774155, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 901, "step_time": 39.72260526288301 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 277.0, "completions/max_terminated_length": 277.0, "completions/mean_length": 123.6015625, "completions/mean_terminated_length": 123.6015625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.241148752393201, "epoch": 0.513960113960114, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06134183704853058, "kl": 0.17030273156706244, "learning_rate": 2.8199445957224293e-06, "loss": 0.0008515514782629907, "num_tokens": 141965365.0, "reward": 2.333789348602295, "reward_std": 0.5608569979667664, "rewards/code_complexity_reward/mean": 0.9046875238418579, "rewards/code_complexity_reward/std": 0.16025756299495697, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 902, "step_time": 32.58681991044432 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 129.390625, "completions/mean_terminated_length": 128.64187622070312, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2420972571708262, "epoch": 0.5145299145299145, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05408443138003349, "kl": 0.15696435037534684, "learning_rate": 2.815010844401682e-06, "loss": 0.0007852421258576214, "num_tokens": 142099789.0, "reward": 2.3056154251098633, "reward_std": 0.4917409420013428, "rewards/code_complexity_reward/mean": 0.9148437976837158, "rewards/code_complexity_reward/std": 0.11021405458450317, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 903, "step_time": 57.924823007546365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 122.669921875, "completions/mean_terminated_length": 122.669921875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2394076983910054, "epoch": 0.5150997150997151, "frac_reward_zero_std": 0.5, "grad_norm": 0.13606178760528564, "kl": 0.15028440079186112, "learning_rate": 2.8100758460970334e-06, "loss": 0.0007512631127610803, "num_tokens": 142228316.0, "reward": 2.36572265625, "reward_std": 0.5216185450553894, "rewards/code_complexity_reward/mean": 0.9190429449081421, "rewards/code_complexity_reward/std": 0.11262592673301697, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 904, "step_time": 52.3289746530354 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 125.521484375, "completions/mean_terminated_length": 125.521484375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23926722072064877, "epoch": 0.5156695156695157, "frac_reward_zero_std": 0.515625, "grad_norm": 0.0617661215364933, "kl": 0.150087837013416, "learning_rate": 2.8051396203438846e-06, "loss": 0.0007502379012294114, "num_tokens": 142363831.0, "reward": 2.324462890625, "reward_std": 0.5150271058082581, "rewards/code_complexity_reward/mean": 0.914843738079071, "rewards/code_complexity_reward/std": 0.11917185038328171, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 905, "step_time": 55.31032776273787 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 446.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 127.369140625, "completions/mean_terminated_length": 127.369140625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24490493908524513, "epoch": 0.5162393162393163, "frac_reward_zero_std": 0.5, "grad_norm": 0.06121162325143814, "kl": 0.1961834612302482, "learning_rate": 2.8002021866824964e-06, "loss": 0.0009813314536586404, "num_tokens": 142500860.0, "reward": 2.3761231899261475, "reward_std": 0.5153822898864746, "rewards/code_complexity_reward/mean": 0.9191405773162842, "rewards/code_complexity_reward/std": 0.11145209521055222, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 906, "step_time": 43.62790545541793 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 427.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 123.58203125, "completions/mean_terminated_length": 123.58203125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2250993822235614, "epoch": 0.5168091168091168, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05739313364028931, "kl": 0.15371958550531417, "learning_rate": 2.7952635646579114e-06, "loss": 0.0007685550954192877, "num_tokens": 142632230.0, "reward": 2.427783489227295, "reward_std": 0.5439947843551636, "rewards/code_complexity_reward/mean": 0.9180663824081421, "rewards/code_complexity_reward/std": 0.12131363153457642, "rewards/code_execution_reward/mean": 0.41796875, "rewards/code_execution_reward/std": 0.4937073290348053, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 907, "step_time": 43.27433938253671 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 359.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 121.634765625, "completions/mean_terminated_length": 121.634765625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23587231896817684, "epoch": 0.5173789173789174, "frac_reward_zero_std": 0.609375, "grad_norm": 0.054459650069475174, "kl": 0.16384848242159933, "learning_rate": 2.7903237738198757e-06, "loss": 0.0008194809197448194, "num_tokens": 142766283.0, "reward": 2.2781739234924316, "reward_std": 0.45989179611206055, "rewards/code_complexity_reward/mean": 0.9286133050918579, "rewards/code_complexity_reward/std": 0.08678104728460312, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 908, "step_time": 45.12384353391826 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 120.361328125, "completions/mean_terminated_length": 120.361328125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24299318552948534, "epoch": 0.517948717948718, "frac_reward_zero_std": 0.4375, "grad_norm": 0.062347158789634705, "kl": 0.15853715722914785, "learning_rate": 2.7853828337227634e-06, "loss": 0.0007926627877168357, "num_tokens": 142897108.0, "reward": 2.346484661102295, "reward_std": 0.5579415559768677, "rewards/code_complexity_reward/mean": 0.9046875238418579, "rewards/code_complexity_reward/std": 0.154696524143219, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 909, "step_time": 39.20590987429023 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 283.0, "completions/max_terminated_length": 283.0, "completions/mean_length": 121.509765625, "completions/mean_terminated_length": 121.509765625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22192594758234918, "epoch": 0.5185185185185185, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05425991117954254, "kl": 0.15326021541841328, "learning_rate": 2.780440763925496e-06, "loss": 0.0007665685843676329, "num_tokens": 143027561.0, "reward": 2.380420207977295, "reward_std": 0.5234133005142212, "rewards/code_complexity_reward/mean": 0.9166015386581421, "rewards/code_complexity_reward/std": 0.11190300434827805, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 910, "step_time": 52.98761723656207 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 378.0, "completions/max_terminated_length": 378.0, "completions/mean_length": 127.21484375, "completions/mean_terminated_length": 127.21484375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.24317420809529722, "epoch": 0.5190883190883191, "frac_reward_zero_std": 0.484375, "grad_norm": 0.055354367941617966, "kl": 0.1669982058228925, "learning_rate": 2.7754975839914696e-06, "loss": 0.000835240411106497, "num_tokens": 143159879.0, "reward": 2.274414300918579, "reward_std": 0.4667382538318634, "rewards/code_complexity_reward/mean": 0.926074206829071, "rewards/code_complexity_reward/std": 0.09533552825450897, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 911, "step_time": 39.412008627317846 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 478.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 122.427734375, "completions/mean_terminated_length": 122.427734375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.24207019759342074, "epoch": 0.5196581196581197, "frac_reward_zero_std": 0.609375, "grad_norm": 0.05692822486162186, "kl": 0.1678520357236266, "learning_rate": 2.7705533134884726e-06, "loss": 0.0008393704192712903, "num_tokens": 143288714.0, "reward": 2.2706055641174316, "reward_std": 0.46083420515060425, "rewards/code_complexity_reward/mean": 0.9186522960662842, "rewards/code_complexity_reward/std": 0.08885408937931061, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 912, "step_time": 46.964586914516985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 125.140625, "completions/mean_terminated_length": 125.140625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24295478034764528, "epoch": 0.5202279202279202, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05467695742845535, "kl": 0.15890052379108965, "learning_rate": 2.7656079719886116e-06, "loss": 0.000794316059909761, "num_tokens": 143422186.0, "reward": 2.325439453125, "reward_std": 0.5186195969581604, "rewards/code_complexity_reward/mean": 0.9185546636581421, "rewards/code_complexity_reward/std": 0.12146149575710297, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.04008939489722252, "step": 913, "step_time": 47.818261618725955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 130.5078125, "completions/mean_terminated_length": 129.01177978515625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.25234773522242904, "epoch": 0.5207977207977208, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05915921553969383, "kl": 0.16239717393182218, "learning_rate": 2.7606615790682325e-06, "loss": 0.0008119660778902471, "num_tokens": 143557310.0, "reward": 2.2333006858825684, "reward_std": 0.5164485573768616, "rewards/code_complexity_reward/mean": 0.904296875, "rewards/code_complexity_reward/std": 0.15993238985538483, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 914, "step_time": 75.49653916154057 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 128.185546875, "completions/mean_terminated_length": 128.185546875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2385648563504219, "epoch": 0.5213675213675214, "frac_reward_zero_std": 0.53125, "grad_norm": 0.060731880366802216, "kl": 0.15905313531402498, "learning_rate": 2.755714154307842e-06, "loss": 0.0007953314343467355, "num_tokens": 143693141.0, "reward": 2.2809083461761475, "reward_std": 0.49388930201530457, "rewards/code_complexity_reward/mean": 0.9215819835662842, "rewards/code_complexity_reward/std": 0.1191868931055069, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 915, "step_time": 38.88141783326864 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 287.0, "completions/max_terminated_length": 287.0, "completions/mean_length": 120.869140625, "completions/mean_terminated_length": 120.869140625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.24009809270501137, "epoch": 0.5219373219373219, "frac_reward_zero_std": 0.59375, "grad_norm": 0.051036760210990906, "kl": 0.15534095268230885, "learning_rate": 2.7507657172920345e-06, "loss": 0.0007766250055283308, "num_tokens": 143825930.0, "reward": 2.335010051727295, "reward_std": 0.4800686538219452, "rewards/code_complexity_reward/mean": 0.9276367425918579, "rewards/code_complexity_reward/std": 0.07644806802272797, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 916, "step_time": 35.115571291185915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 126.16796875, "completions/mean_terminated_length": 126.16796875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23599006794393063, "epoch": 0.5225071225071225, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06743097305297852, "kl": 0.1590923797339201, "learning_rate": 2.7458162876094075e-06, "loss": 0.000795305531937629, "num_tokens": 143959248.0, "reward": 2.3585939407348633, "reward_std": 0.5089361667633057, "rewards/code_complexity_reward/mean": 0.923632800579071, "rewards/code_complexity_reward/std": 0.09650489687919617, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 917, "step_time": 36.485746058635414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 301.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 120.37890625, "completions/mean_terminated_length": 120.37890625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23062844504602253, "epoch": 0.5230769230769231, "frac_reward_zero_std": 0.53125, "grad_norm": 0.06298386305570602, "kl": 0.15354954497888684, "learning_rate": 2.7408658848524923e-06, "loss": 0.0007679228438064456, "num_tokens": 144087282.0, "reward": 2.4337892532348633, "reward_std": 0.5514266490936279, "rewards/code_complexity_reward/mean": 0.917773425579071, "rewards/code_complexity_reward/std": 0.11922172456979752, "rewards/code_execution_reward/mean": 0.423828125, "rewards/code_execution_reward/std": 0.4946470856666565, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 918, "step_time": 45.44145175348967 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 119.224609375, "completions/mean_terminated_length": 119.224609375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24051502090878785, "epoch": 0.5236467236467236, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05622071400284767, "kl": 0.17260406620334834, "learning_rate": 2.7359145286176697e-06, "loss": 0.0008628762443549931, "num_tokens": 144217053.0, "reward": 2.3612306118011475, "reward_std": 0.49708133935928345, "rewards/code_complexity_reward/mean": 0.92333984375, "rewards/code_complexity_reward/std": 0.08729002624750137, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 919, "step_time": 52.73740313760936 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 122.75390625, "completions/mean_terminated_length": 121.99217224121094, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2391504063270986, "epoch": 0.5242165242165242, "frac_reward_zero_std": 0.5, "grad_norm": 0.06300543993711472, "kl": 0.14882029965519905, "learning_rate": 2.7309622385050936e-06, "loss": 0.000744168646633625, "num_tokens": 144350839.0, "reward": 2.299072265625, "reward_std": 0.4955234229564667, "rewards/code_complexity_reward/mean": 0.9209960699081421, "rewards/code_complexity_reward/std": 0.11301586776971817, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 920, "step_time": 56.473961509764194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 123.1875, "completions/mean_terminated_length": 123.1875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2316111766267568, "epoch": 0.5247863247863248, "frac_reward_zero_std": 0.5, "grad_norm": 0.0562516525387764, "kl": 0.16299362515565008, "learning_rate": 2.7260090341186174e-06, "loss": 0.0008152095833793283, "num_tokens": 144481319.0, "reward": 2.3648438453674316, "reward_std": 0.47847694158554077, "rewards/code_complexity_reward/mean": 0.9327148199081421, "rewards/code_complexity_reward/std": 0.047738004475831985, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 921, "step_time": 57.74849547818303 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 128.224609375, "completions/mean_terminated_length": 128.224609375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22624896024353802, "epoch": 0.5253561253561253, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05896074324846268, "kl": 0.15968782070558518, "learning_rate": 2.721054935065712e-06, "loss": 0.0007984080584719777, "num_tokens": 144613194.0, "reward": 2.3112306594848633, "reward_std": 0.5161822438240051, "rewards/code_complexity_reward/mean": 0.909472644329071, "rewards/code_complexity_reward/std": 0.1270742416381836, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 922, "step_time": 41.98868646938354 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 120.89453125, "completions/mean_terminated_length": 120.89453125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24240667931735516, "epoch": 0.5259259259259259, "frac_reward_zero_std": 0.625, "grad_norm": 0.048175740987062454, "kl": 0.16013657744042575, "learning_rate": 2.7160999609573907e-06, "loss": 0.0008005741983652115, "num_tokens": 144743996.0, "reward": 2.3937501907348633, "reward_std": 0.48779627680778503, "rewards/code_complexity_reward/mean": 0.928515613079071, "rewards/code_complexity_reward/std": 0.04751499742269516, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 923, "step_time": 35.467219163663685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 125.79296875, "completions/mean_terminated_length": 125.79296875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.239512977655977, "epoch": 0.5264957264957265, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05677700787782669, "kl": 0.16085303691215813, "learning_rate": 2.7111441314081304e-06, "loss": 0.0008041923283599317, "num_tokens": 144877074.0, "reward": 2.2806642055511475, "reward_std": 0.493705153465271, "rewards/code_complexity_reward/mean": 0.920703113079071, "rewards/code_complexity_reward/std": 0.11940447986125946, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812851272523403, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 924, "step_time": 37.723523365333676 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 131.833984375, "completions/mean_terminated_length": 131.833984375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24423302803188562, "epoch": 0.5270655270655271, "frac_reward_zero_std": 0.578125, "grad_norm": 0.07049722224473953, "kl": 0.15082939830608666, "learning_rate": 2.7061874660357944e-06, "loss": 0.0007541475351899862, "num_tokens": 145014045.0, "reward": 2.317431926727295, "reward_std": 0.5134255290031433, "rewards/code_complexity_reward/mean": 0.9129882454872131, "rewards/code_complexity_reward/std": 0.12308118492364883, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 925, "step_time": 50.103341861627996 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 261.0, "completions/max_terminated_length": 261.0, "completions/mean_length": 115.1328125, "completions/mean_terminated_length": 115.1328125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24320953921414912, "epoch": 0.5276353276353276, "frac_reward_zero_std": 0.5, "grad_norm": 0.07316678762435913, "kl": 0.16505830164533108, "learning_rate": 2.7012299844615554e-06, "loss": 0.0008253681589849293, "num_tokens": 145139825.0, "reward": 2.3776369094848633, "reward_std": 0.5183925628662109, "rewards/code_complexity_reward/mean": 0.928027331829071, "rewards/code_complexity_reward/std": 0.10407835245132446, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 926, "step_time": 33.72297054994851 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 122.37109375, "completions/mean_terminated_length": 122.37109375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23659894987940788, "epoch": 0.5282051282051282, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05882514640688896, "kl": 0.16498675453476608, "learning_rate": 2.696271706309814e-06, "loss": 0.0008254004642367363, "num_tokens": 145269063.0, "reward": 2.35595703125, "reward_std": 0.4844435751438141, "rewards/code_complexity_reward/mean": 0.9327148199081421, "rewards/code_complexity_reward/std": 0.07532741874456406, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 927, "step_time": 45.990120619535446 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 120.8984375, "completions/mean_terminated_length": 120.13307189941406, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.24739812361076474, "epoch": 0.5287749287749288, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05837591364979744, "kl": 0.16950768162496388, "learning_rate": 2.691312651208129e-06, "loss": 0.0008477407391183078, "num_tokens": 145397731.0, "reward": 2.2962892055511475, "reward_std": 0.5001206994056702, "rewards/code_complexity_reward/mean": 0.92138671875, "rewards/code_complexity_reward/std": 0.11864627152681351, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 928, "step_time": 49.69592124503106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 120.890625, "completions/mean_terminated_length": 120.890625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24941144394688308, "epoch": 0.5293447293447293, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05963987484574318, "kl": 0.16639444266911596, "learning_rate": 2.68635283878713e-06, "loss": 0.0008320648921653628, "num_tokens": 145530347.0, "reward": 2.3336915969848633, "reward_std": 0.5266202688217163, "rewards/code_complexity_reward/mean": 0.919921875, "rewards/code_complexity_reward/std": 0.13411884009838104, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 929, "step_time": 53.23980880621821 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 338.0, "completions/max_terminated_length": 338.0, "completions/mean_length": 127.44921875, "completions/mean_terminated_length": 127.44921875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2419690692331642, "epoch": 0.5299145299145299, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05118250101804733, "kl": 0.15547213552054018, "learning_rate": 2.6813922886804477e-06, "loss": 0.0007776428828947246, "num_tokens": 145664857.0, "reward": 2.35693359375, "reward_std": 0.5160375833511353, "rewards/code_complexity_reward/mean": 0.921875, "rewards/code_complexity_reward/std": 0.10473150759935379, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 930, "step_time": 37.48420372232795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 317.0, "completions/max_terminated_length": 317.0, "completions/mean_length": 129.384765625, "completions/mean_terminated_length": 129.384765625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.236864251550287, "epoch": 0.5304843304843305, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06445631384849548, "kl": 0.1744081110227853, "learning_rate": 2.676431020524632e-06, "loss": 0.0008724232320673764, "num_tokens": 145799054.0, "reward": 2.361377239227295, "reward_std": 0.489092081785202, "rewards/code_complexity_reward/mean": 0.9249023199081421, "rewards/code_complexity_reward/std": 0.07668517529964447, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 931, "step_time": 36.230026290751994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 124.080078125, "completions/mean_terminated_length": 124.080078125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2411022528540343, "epoch": 0.531054131054131, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05554966256022453, "kl": 0.14684851397760212, "learning_rate": 2.6714690539590755e-06, "loss": 0.0007342366734519601, "num_tokens": 145931255.0, "reward": 2.403613567352295, "reward_std": 0.5329955816268921, "rewards/code_complexity_reward/mean": 0.9117187261581421, "rewards/code_complexity_reward/std": 0.11836861073970795, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 932, "step_time": 35.00305705331266 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 122.2890625, "completions/mean_terminated_length": 121.52642059326172, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23057022131979465, "epoch": 0.5316239316239316, "frac_reward_zero_std": 0.546875, "grad_norm": 0.06325671076774597, "kl": 0.1921455771662295, "learning_rate": 2.6665064086259334e-06, "loss": 0.0009607979445718229, "num_tokens": 146062635.0, "reward": 2.295947313308716, "reward_std": 0.5002831816673279, "rewards/code_complexity_reward/mean": 0.9208008050918579, "rewards/code_complexity_reward/std": 0.1182965487241745, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 933, "step_time": 49.12031447608024 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 469.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 127.93359375, "completions/mean_terminated_length": 127.93359375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.24556237133219838, "epoch": 0.5321937321937322, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05816458910703659, "kl": 0.15492470155004412, "learning_rate": 2.6615431041700507e-06, "loss": 0.0007746355258859694, "num_tokens": 146194745.0, "reward": 2.405517816543579, "reward_std": 0.5200223922729492, "rewards/code_complexity_reward/mean": 0.9218749403953552, "rewards/code_complexity_reward/std": 0.09189247339963913, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 934, "step_time": 51.97603899426758 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 407.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 125.65234375, "completions/mean_terminated_length": 125.65234375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2327044028788805, "epoch": 0.5327635327635327, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05803702399134636, "kl": 0.1566320132697001, "learning_rate": 2.65657916023888e-06, "loss": 0.0007830619579181075, "num_tokens": 146326455.0, "reward": 2.44873046875, "reward_std": 0.5127755403518677, "rewards/code_complexity_reward/mean": 0.9288085699081421, "rewards/code_complexity_reward/std": 0.07039965689182281, "rewards/code_execution_reward/mean": 0.421875, "rewards/code_execution_reward/std": 0.49434176087379456, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 935, "step_time": 54.27001681737602 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 129.009765625, "completions/mean_terminated_length": 125.99409484863281, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2347163469530642, "epoch": 0.5333333333333333, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06674225628376007, "kl": 0.15422132262028754, "learning_rate": 2.651614596482406e-06, "loss": 0.0007712366641499102, "num_tokens": 146458508.0, "reward": 2.343554973602295, "reward_std": 0.5353113412857056, "rewards/code_complexity_reward/mean": 0.9105468392372131, "rewards/code_complexity_reward/std": 0.13878877460956573, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 936, "step_time": 54.82791930902749 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 489.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 123.900390625, "completions/mean_terminated_length": 123.900390625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23169175232760608, "epoch": 0.5339031339031339, "frac_reward_zero_std": 0.5, "grad_norm": 0.05714459717273712, "kl": 0.14503608318045735, "learning_rate": 2.6466494325530667e-06, "loss": 0.0007253211224451661, "num_tokens": 146588153.0, "reward": 2.3956055641174316, "reward_std": 0.5100220441818237, "rewards/code_complexity_reward/mean": 0.9284179210662842, "rewards/code_complexity_reward/std": 0.08553982526063919, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 937, "step_time": 69.48730009049177 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 290.0, "completions/max_terminated_length": 290.0, "completions/mean_length": 119.55859375, "completions/mean_terminated_length": 119.55859375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23071769787929952, "epoch": 0.5344729344729344, "frac_reward_zero_std": 0.453125, "grad_norm": 0.060997869819402695, "kl": 0.15278225776273757, "learning_rate": 2.641683688105677e-06, "loss": 0.0007638261886313558, "num_tokens": 146718479.0, "reward": 2.4778809547424316, "reward_std": 0.5079748630523682, "rewards/code_complexity_reward/mean": 0.9318358898162842, "rewards/code_complexity_reward/std": 0.05065276473760605, "rewards/code_execution_reward/mean": 0.447265625, "rewards/code_execution_reward/std": 0.4976975917816162, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 938, "step_time": 39.92920105997473 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 128.162109375, "completions/mean_terminated_length": 128.162109375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23520930903032422, "epoch": 0.535042735042735, "frac_reward_zero_std": 0.453125, "grad_norm": 0.061355412006378174, "kl": 0.1705204783938825, "learning_rate": 2.6367173827973463e-06, "loss": 0.0008525546872988343, "num_tokens": 146856098.0, "reward": 2.300341844558716, "reward_std": 0.4755428731441498, "rewards/code_complexity_reward/mean": 0.9280273914337158, "rewards/code_complexity_reward/std": 0.08875423669815063, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 939, "step_time": 37.16617320291698 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 274.0, "completions/max_terminated_length": 274.0, "completions/mean_length": 122.5078125, "completions/mean_terminated_length": 122.5078125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23638475756160915, "epoch": 0.5356125356125356, "frac_reward_zero_std": 0.640625, "grad_norm": 0.04693220928311348, "kl": 0.16143466322682798, "learning_rate": 2.631750536287408e-06, "loss": 0.0008070820476859808, "num_tokens": 146986446.0, "reward": 2.2596192359924316, "reward_std": 0.4208502769470215, "rewards/code_complexity_reward/mean": 0.9325194954872131, "rewards/code_complexity_reward/std": 0.05036170035600662, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 940, "step_time": 34.339668040163815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 414.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 129.662109375, "completions/mean_terminated_length": 129.662109375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23335627163760364, "epoch": 0.5361823361823361, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05899406969547272, "kl": 0.1537207190413028, "learning_rate": 2.6267831682373364e-06, "loss": 0.0007687712786719203, "num_tokens": 147123489.0, "reward": 2.32275390625, "reward_std": 0.4713149964809418, "rewards/code_complexity_reward/mean": 0.9317382574081421, "rewards/code_complexity_reward/std": 0.06755084544420242, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 941, "step_time": 42.227080604061484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 128.8203125, "completions/mean_terminated_length": 128.07044982910156, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23617741209454834, "epoch": 0.5367521367521367, "frac_reward_zero_std": 0.421875, "grad_norm": 0.0599522739648819, "kl": 0.14646647521294653, "learning_rate": 2.62181529831067e-06, "loss": 0.000732718501240015, "num_tokens": 147259021.0, "reward": 2.3749513626098633, "reward_std": 0.5182493925094604, "rewards/code_complexity_reward/mean": 0.9206054210662842, "rewards/code_complexity_reward/std": 0.10713408887386322, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 942, "step_time": 56.941783459857106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 132.0, "completions/mean_terminated_length": 132.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2405220817308873, "epoch": 0.5373219373219373, "frac_reward_zero_std": 0.609375, "grad_norm": 0.05038125813007355, "kl": 0.15488878753967583, "learning_rate": 2.6168469461729344e-06, "loss": 0.0007745387265458703, "num_tokens": 147394685.0, "reward": 2.3150391578674316, "reward_std": 0.4894792437553406, "rewards/code_complexity_reward/mean": 0.9258788824081421, "rewards/code_complexity_reward/std": 0.0976695567369461, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 943, "step_time": 38.96098215971142 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 128.302734375, "completions/mean_terminated_length": 128.302734375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2435975344851613, "epoch": 0.5378917378917379, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05771918594837189, "kl": 0.16041559563018382, "learning_rate": 2.611878131491565e-06, "loss": 0.0008021241519600153, "num_tokens": 147530400.0, "reward": 2.2853028774261475, "reward_std": 0.4757680892944336, "rewards/code_complexity_reward/mean": 0.925000011920929, "rewards/code_complexity_reward/std": 0.10131233185529709, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 944, "step_time": 39.78165880031884 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 443.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 127.916015625, "completions/mean_terminated_length": 127.916015625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.22559391404502094, "epoch": 0.5384615384615384, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05008731409907341, "kl": 0.1668271883390844, "learning_rate": 2.606908873935826e-06, "loss": 0.0008344544912688434, "num_tokens": 147661821.0, "reward": 2.3554201126098633, "reward_std": 0.5048552751541138, "rewards/code_complexity_reward/mean": 0.916015625, "rewards/code_complexity_reward/std": 0.11225035041570663, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 945, "step_time": 43.80758222937584 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 422.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 127.71875, "completions/mean_terminated_length": 127.71875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24019241030327976, "epoch": 0.539031339031339, "frac_reward_zero_std": 0.546875, "grad_norm": 0.053791310638189316, "kl": 0.16670753224752843, "learning_rate": 2.6019391931767374e-06, "loss": 0.0008334882440976799, "num_tokens": 147797789.0, "reward": 2.3009278774261475, "reward_std": 0.5138210654258728, "rewards/code_complexity_reward/mean": 0.9181640148162842, "rewards/code_complexity_reward/std": 0.1329042911529541, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 946, "step_time": 47.19077653251588 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 120.759765625, "completions/mean_terminated_length": 120.759765625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22664424683898687, "epoch": 0.5396011396011396, "frac_reward_zero_std": 0.59375, "grad_norm": 0.054421745240688324, "kl": 0.15466131072025746, "learning_rate": 2.596969108886991e-06, "loss": 0.0007734585087746382, "num_tokens": 147930906.0, "reward": 2.35693359375, "reward_std": 0.527145504951477, "rewards/code_complexity_reward/mean": 0.9219726920127869, "rewards/code_complexity_reward/std": 0.12535902857780457, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 947, "step_time": 37.65305580571294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 500.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 128.1875, "completions/mean_terminated_length": 128.1875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2348505686968565, "epoch": 0.5401709401709401, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05545811727643013, "kl": 0.1630616137990728, "learning_rate": 2.59199864074088e-06, "loss": 0.0008151082438416779, "num_tokens": 148063930.0, "reward": 2.2530763149261475, "reward_std": 0.49076756834983826, "rewards/code_complexity_reward/mean": 0.91650390625, "rewards/code_complexity_reward/std": 0.12735486030578613, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 948, "step_time": 48.25713705923408 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 406.0, "completions/max_terminated_length": 406.0, "completions/mean_length": 130.787109375, "completions/mean_terminated_length": 130.787109375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2448312931228429, "epoch": 0.5407407407407407, "frac_reward_zero_std": 0.53125, "grad_norm": 0.0513082854449749, "kl": 0.16172462166287005, "learning_rate": 2.5870278084142144e-06, "loss": 0.0008088928880169988, "num_tokens": 148199509.0, "reward": 2.2383790016174316, "reward_std": 0.45674359798431396, "rewards/code_complexity_reward/mean": 0.9185546636581421, "rewards/code_complexity_reward/std": 0.10857311636209488, "rewards/code_execution_reward/mean": 0.2265625, "rewards/code_execution_reward/std": 0.4190165400505066, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 949, "step_time": 41.29468002356589 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 327.0, "completions/max_terminated_length": 327.0, "completions/mean_length": 125.3203125, "completions/mean_terminated_length": 125.3203125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23310476588085294, "epoch": 0.5413105413105413, "frac_reward_zero_std": 0.4375, "grad_norm": 0.056193068623542786, "kl": 0.1451058074599132, "learning_rate": 2.582056631584246e-06, "loss": 0.0007255577947944403, "num_tokens": 148333081.0, "reward": 2.31689453125, "reward_std": 0.4822918474674225, "rewards/code_complexity_reward/mean": 0.922656238079071, "rewards/code_complexity_reward/std": 0.08654249459505081, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 950, "step_time": 36.935956972651184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 126.66796875, "completions/mean_terminated_length": 125.91389465332031, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23910746024921536, "epoch": 0.5418803418803418, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05586117506027222, "kl": 0.15843922807835042, "learning_rate": 2.577085129929593e-06, "loss": 0.0007921388023532927, "num_tokens": 148465175.0, "reward": 2.3314454555511475, "reward_std": 0.5001760721206665, "rewards/code_complexity_reward/mean": 0.9235351085662842, "rewards/code_complexity_reward/std": 0.10382980853319168, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 951, "step_time": 49.148253376595676 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 309.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 128.763671875, "completions/mean_terminated_length": 128.763671875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23063878039829433, "epoch": 0.5424501424501424, "frac_reward_zero_std": 0.4375, "grad_norm": 0.055825185030698776, "kl": 0.15037088841199875, "learning_rate": 2.5721133231301547e-06, "loss": 0.0007520612562075257, "num_tokens": 148600854.0, "reward": 2.2464356422424316, "reward_std": 0.4493507444858551, "rewards/code_complexity_reward/mean": 0.9193359017372131, "rewards/code_complexity_reward/std": 0.0994463562965393, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 952, "step_time": 44.220666446723044 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 123.923828125, "completions/mean_terminated_length": 123.923828125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23830513330176473, "epoch": 0.543019943019943, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06007346138358116, "kl": 0.17064976296387613, "learning_rate": 2.567141230867043e-06, "loss": 0.0008529311162419617, "num_tokens": 148733791.0, "reward": 2.365771770477295, "reward_std": 0.49659955501556396, "rewards/code_complexity_reward/mean": 0.9283202886581421, "rewards/code_complexity_reward/std": 0.08769002556800842, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 953, "step_time": 45.41571477800608 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 126.138671875, "completions/mean_terminated_length": 126.138671875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2345753808040172, "epoch": 0.5435897435897435, "frac_reward_zero_std": 0.546875, "grad_norm": 0.051701080054044724, "kl": 0.15308188018389046, "learning_rate": 2.5621688728224965e-06, "loss": 0.0007654677610844374, "num_tokens": 148864774.0, "reward": 2.336719036102295, "reward_std": 0.5117713212966919, "rewards/code_complexity_reward/mean": 0.9258788824081421, "rewards/code_complexity_reward/std": 0.11312674731016159, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 954, "step_time": 47.604430680163205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 331.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 130.142578125, "completions/mean_terminated_length": 130.142578125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24050823412835598, "epoch": 0.5441595441595442, "frac_reward_zero_std": 0.5, "grad_norm": 0.06663521379232407, "kl": 0.15770623285789043, "learning_rate": 2.5571962686798073e-06, "loss": 0.0007885182276368141, "num_tokens": 149001863.0, "reward": 2.28369140625, "reward_std": 0.4896109104156494, "rewards/code_complexity_reward/mean": 0.9229491949081421, "rewards/code_complexity_reward/std": 0.11917727440595627, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 955, "step_time": 37.54196618311107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 125.607421875, "completions/mean_terminated_length": 124.0921630859375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23063887748867273, "epoch": 0.5447293447293448, "frac_reward_zero_std": 0.484375, "grad_norm": 0.052244633436203, "kl": 0.1693061749683693, "learning_rate": 2.5522234381232424e-06, "loss": 0.0008467184961773455, "num_tokens": 149135686.0, "reward": 2.408740282058716, "reward_std": 0.5182817578315735, "rewards/code_complexity_reward/mean": 0.9239257574081421, "rewards/code_complexity_reward/std": 0.0958983525633812, "rewards/code_execution_reward/mean": 0.390625, "rewards/code_execution_reward/std": 0.48836761713027954, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 956, "step_time": 61.342045459896326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 486.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 123.630859375, "completions/mean_terminated_length": 123.630859375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23624830273911357, "epoch": 0.5452991452991452, "frac_reward_zero_std": 0.4375, "grad_norm": 0.07164279371500015, "kl": 0.1537291316781193, "learning_rate": 2.547250400837964e-06, "loss": 0.0007687810575589538, "num_tokens": 149266201.0, "reward": 2.3882813453674316, "reward_std": 0.5119504928588867, "rewards/code_complexity_reward/mean": 0.924023449420929, "rewards/code_complexity_reward/std": 0.09766863286495209, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 957, "step_time": 46.39702033624053 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 126.291015625, "completions/mean_terminated_length": 124.7784423828125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22830513981170952, "epoch": 0.5458689458689459, "frac_reward_zero_std": 0.515625, "grad_norm": 0.0537567064166069, "kl": 0.1340115738566965, "learning_rate": 2.5422771765099524e-06, "loss": 0.0006700929370708764, "num_tokens": 149400542.0, "reward": 2.4452147483825684, "reward_std": 0.5164884924888611, "rewards/code_complexity_reward/mean": 0.9287109375, "rewards/code_complexity_reward/std": 0.0778622031211853, "rewards/code_execution_reward/mean": 0.419921875, "rewards/code_execution_reward/std": 0.4940285086631775, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 958, "step_time": 59.250686287879944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 131.2578125, "completions/mean_terminated_length": 130.51272583007812, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23357106605544686, "epoch": 0.5464387464387465, "frac_reward_zero_std": 0.484375, "grad_norm": 0.049513060599565506, "kl": 0.1494690626859665, "learning_rate": 2.5373037848259295e-06, "loss": 0.0007473885780200362, "num_tokens": 149539890.0, "reward": 2.3441407680511475, "reward_std": 0.5146934390068054, "rewards/code_complexity_reward/mean": 0.9173828363418579, "rewards/code_complexity_reward/std": 0.10966318100690842, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 959, "step_time": 68.6868627909571 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 317.0, "completions/max_terminated_length": 317.0, "completions/mean_length": 124.0625, "completions/mean_terminated_length": 124.0625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.21756852860562503, "epoch": 0.5470085470085471, "frac_reward_zero_std": 0.515625, "grad_norm": 0.056293852627277374, "kl": 0.14539052185136825, "learning_rate": 2.532330245473279e-06, "loss": 0.0007270427886396646, "num_tokens": 149670122.0, "reward": 2.370898723602295, "reward_std": 0.5063711404800415, "rewards/code_complexity_reward/mean": 0.9232421517372131, "rewards/code_complexity_reward/std": 0.09048053622245789, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 960, "step_time": 36.23090007342398 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 406.0, "completions/mean_length": 127.009765625, "completions/mean_terminated_length": 126.25636291503906, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23515363549813628, "epoch": 0.5475783475783476, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05425279214978218, "kl": 0.14644804771523923, "learning_rate": 2.5273565781399684e-06, "loss": 0.0007319908472709358, "num_tokens": 149807271.0, "reward": 2.3128907680511475, "reward_std": 0.4995618164539337, "rewards/code_complexity_reward/mean": 0.918652355670929, "rewards/code_complexity_reward/std": 0.10817259550094604, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 961, "step_time": 48.59473649319261 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 388.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 122.42578125, "completions/mean_terminated_length": 122.42578125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22631168900988996, "epoch": 0.5481481481481482, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05189281329512596, "kl": 0.1526399798458442, "learning_rate": 2.5223828025144733e-06, "loss": 0.0007631211774423718, "num_tokens": 149936089.0, "reward": 2.390380859375, "reward_std": 0.50080806016922, "rewards/code_complexity_reward/mean": 0.930468738079071, "rewards/code_complexity_reward/std": 0.07869641482830048, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 962, "step_time": 39.84272409975529 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 126.0546875, "completions/mean_terminated_length": 125.2994155883789, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2246039011515677, "epoch": 0.5487179487179488, "frac_reward_zero_std": 0.546875, "grad_norm": 0.09041039645671844, "kl": 0.20342903828714043, "learning_rate": 2.517408938285697e-06, "loss": 0.0010163565166294575, "num_tokens": 150066741.0, "reward": 2.3808107376098633, "reward_std": 0.516076385974884, "rewards/code_complexity_reward/mean": 0.919726550579071, "rewards/code_complexity_reward/std": 0.10261514037847519, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 963, "step_time": 54.54643815662712 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 128.716796875, "completions/mean_terminated_length": 127.21372985839844, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23003301047720015, "epoch": 0.5492877492877493, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05662541836500168, "kl": 0.15942117909435183, "learning_rate": 2.512435005142894e-06, "loss": 0.0007972653256729245, "num_tokens": 150201468.0, "reward": 2.3168458938598633, "reward_std": 0.5108779072761536, "rewards/code_complexity_reward/mean": 0.9228515625, "rewards/code_complexity_reward/std": 0.11946255713701248, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 964, "step_time": 80.57439717371017 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 442.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 122.39453125, "completions/mean_terminated_length": 122.39453125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23587281815707684, "epoch": 0.5498575498575499, "frac_reward_zero_std": 0.5625, "grad_norm": 0.06640879064798355, "kl": 0.210566992405802, "learning_rate": 2.507461022775591e-06, "loss": 0.001051964471116662, "num_tokens": 150332934.0, "reward": 2.3698244094848633, "reward_std": 0.4861612915992737, "rewards/code_complexity_reward/mean": 0.931933581829071, "rewards/code_complexity_reward/std": 0.06532151252031326, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 965, "step_time": 55.43770323880017 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 123.224609375, "completions/mean_terminated_length": 123.224609375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23594801709987223, "epoch": 0.5504273504273505, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06528356671333313, "kl": 0.16309043008368462, "learning_rate": 2.5024870108735106e-06, "loss": 0.0008153697708621621, "num_tokens": 150464329.0, "reward": 2.364013671875, "reward_std": 0.5097174048423767, "rewards/code_complexity_reward/mean": 0.9241210222244263, "rewards/code_complexity_reward/std": 0.09513204544782639, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 966, "step_time": 37.753823230974376 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 128.28515625, "completions/mean_terminated_length": 128.28515625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2400600330438465, "epoch": 0.550997150997151, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05592619255185127, "kl": 0.15733959572389722, "learning_rate": 2.4975129891264906e-06, "loss": 0.0007866962114349008, "num_tokens": 150599091.0, "reward": 2.246337890625, "reward_std": 0.4841545522212982, "rewards/code_complexity_reward/mean": 0.913378894329071, "rewards/code_complexity_reward/std": 0.12729962170124054, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 967, "step_time": 41.3511832812801 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 366.0, "completions/mean_length": 121.373046875, "completions/mean_terminated_length": 120.60861206054688, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23627213342115283, "epoch": 0.5515669515669516, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05143818259239197, "kl": 0.16736404749099165, "learning_rate": 2.492538977224409e-06, "loss": 0.0008367898990400136, "num_tokens": 150730474.0, "reward": 2.3385255336761475, "reward_std": 0.5039049983024597, "rewards/code_complexity_reward/mean": 0.923828125, "rewards/code_complexity_reward/std": 0.0965074747800827, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 968, "step_time": 58.25331178866327 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 383.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 118.6796875, "completions/mean_terminated_length": 118.6796875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23226362257264555, "epoch": 0.5521367521367522, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06340117007493973, "kl": 0.14192127052228898, "learning_rate": 2.487564994857107e-06, "loss": 0.0007094062166288495, "num_tokens": 150859358.0, "reward": 2.3219728469848633, "reward_std": 0.523256242275238, "rewards/code_complexity_reward/mean": 0.917285144329071, "rewards/code_complexity_reward/std": 0.13370034098625183, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 969, "step_time": 46.26388185750693 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 116.4765625, "completions/mean_terminated_length": 116.4765625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22844149125739932, "epoch": 0.5527065527065527, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05501887947320938, "kl": 0.16715571947861463, "learning_rate": 2.482591061714304e-06, "loss": 0.0008355114259757102, "num_tokens": 150986714.0, "reward": 2.432422161102295, "reward_std": 0.5046492218971252, "rewards/code_complexity_reward/mean": 0.93212890625, "rewards/code_complexity_reward/std": 0.06332254409790039, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 970, "step_time": 37.48384182341397 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 446.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 124.044921875, "completions/mean_terminated_length": 124.044921875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22745655127801, "epoch": 0.5532763532763533, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05865668132901192, "kl": 0.1665791547857225, "learning_rate": 2.477617197485528e-06, "loss": 0.0008325498783960938, "num_tokens": 151119297.0, "reward": 2.3742189407348633, "reward_std": 0.526898205280304, "rewards/code_complexity_reward/mean": 0.9216797351837158, "rewards/code_complexity_reward/std": 0.11682656407356262, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 971, "step_time": 43.632225304841995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 411.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 126.833984375, "completions/mean_terminated_length": 126.833984375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2350868342909962, "epoch": 0.5538461538461539, "frac_reward_zero_std": 0.546875, "grad_norm": 0.051430054008960724, "kl": 0.15427461685612798, "learning_rate": 2.4726434218600325e-06, "loss": 0.0007716274121776223, "num_tokens": 151253324.0, "reward": 2.358691453933716, "reward_std": 0.4938344359397888, "rewards/code_complexity_reward/mean": 0.9266601204872131, "rewards/code_complexity_reward/std": 0.08768147975206375, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 972, "step_time": 45.93835189938545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 128.95703125, "completions/mean_terminated_length": 128.95703125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2372700220439583, "epoch": 0.5544159544159544, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05653717741370201, "kl": 0.16053497302345932, "learning_rate": 2.4676697545267214e-06, "loss": 0.0008027222356759012, "num_tokens": 151392086.0, "reward": 2.2255373001098633, "reward_std": 0.43492186069488525, "rewards/code_complexity_reward/mean": 0.92578125, "rewards/code_complexity_reward/std": 0.09625764191150665, "rewards/code_execution_reward/mean": 0.205078125, "rewards/code_execution_reward/std": 0.4041535556316376, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 973, "step_time": 42.68236943054944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 341.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 126.435546875, "completions/mean_terminated_length": 126.435546875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22842289507389069, "epoch": 0.554985754985755, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06665603071451187, "kl": 0.1500827904092148, "learning_rate": 2.462696215174071e-06, "loss": 0.0007504898821935058, "num_tokens": 151524773.0, "reward": 2.3526854515075684, "reward_std": 0.5519525408744812, "rewards/code_complexity_reward/mean": 0.9130859375, "rewards/code_complexity_reward/std": 0.14519251883029938, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 974, "step_time": 37.84024197049439 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 305.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 122.349609375, "completions/mean_terminated_length": 122.349609375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2323978936765343, "epoch": 0.5555555555555556, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06232622638344765, "kl": 0.1620650776894763, "learning_rate": 2.4577228234900476e-06, "loss": 0.0008101819548755884, "num_tokens": 151653944.0, "reward": 2.3873536586761475, "reward_std": 0.5102927684783936, "rewards/code_complexity_reward/mean": 0.9284179210662842, "rewards/code_complexity_reward/std": 0.08735086023807526, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 975, "step_time": 42.626972667872906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 352.0, "completions/max_terminated_length": 352.0, "completions/mean_length": 121.3984375, "completions/mean_terminated_length": 121.3984375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23051590658724308, "epoch": 0.5561253561253561, "frac_reward_zero_std": 0.578125, "grad_norm": 0.0501704216003418, "kl": 0.1672632728004828, "learning_rate": 2.452749599162037e-06, "loss": 0.0008364359382539988, "num_tokens": 151783292.0, "reward": 2.317431926727295, "reward_std": 0.47443458437919617, "rewards/code_complexity_reward/mean": 0.9276366829872131, "rewards/code_complexity_reward/std": 0.07995155453681946, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 976, "step_time": 37.56105260178447 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 464.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 119.501953125, "completions/mean_terminated_length": 119.501953125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22788085136562586, "epoch": 0.5566951566951567, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06071331351995468, "kl": 0.15944880212191492, "learning_rate": 2.4477765618767584e-06, "loss": 0.0007975791813805699, "num_tokens": 151913917.0, "reward": 2.4334962368011475, "reward_std": 0.49460503458976746, "rewards/code_complexity_reward/mean": 0.92626953125, "rewards/code_complexity_reward/std": 0.05269961804151535, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 977, "step_time": 45.747420760802925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 462.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 123.681640625, "completions/mean_terminated_length": 123.681640625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23110072594136, "epoch": 0.5572649572649573, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06667788326740265, "kl": 0.1822632682742551, "learning_rate": 2.442803731320194e-06, "loss": 0.0009114322019740939, "num_tokens": 152045466.0, "reward": 2.3592774868011475, "reward_std": 0.5169598460197449, "rewards/code_complexity_reward/mean": 0.91748046875, "rewards/code_complexity_reward/std": 0.10832337290048599, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 978, "step_time": 46.381729485467076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 396.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 124.005859375, "completions/mean_terminated_length": 124.005859375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22668298333883286, "epoch": 0.5578347578347579, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06360066682100296, "kl": 0.15076622844208032, "learning_rate": 2.4378311271775048e-06, "loss": 0.0007539060898125172, "num_tokens": 152177557.0, "reward": 2.436328411102295, "reward_std": 0.5505632758140564, "rewards/code_complexity_reward/mean": 0.9161132574081421, "rewards/code_complexity_reward/std": 0.12562614679336548, "rewards/code_execution_reward/mean": 0.4296875, "rewards/code_execution_reward/std": 0.4955156147480011, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 979, "step_time": 40.90028724633157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 125.61328125, "completions/mean_terminated_length": 125.61328125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22829696093685925, "epoch": 0.5584045584045584, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06469433754682541, "kl": 0.17386810563039035, "learning_rate": 2.4328587691329586e-06, "loss": 0.000869148934725672, "num_tokens": 152312527.0, "reward": 2.324951410293579, "reward_std": 0.5152552723884583, "rewards/code_complexity_reward/mean": 0.9177734851837158, "rewards/code_complexity_reward/std": 0.1265859156847, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 980, "step_time": 40.70175405871123 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 446.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 121.279296875, "completions/mean_terminated_length": 121.279296875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.233255511848256, "epoch": 0.558974358974359, "frac_reward_zero_std": 0.453125, "grad_norm": 0.07285966724157333, "kl": 0.15957463276572526, "learning_rate": 2.4278866768698457e-06, "loss": 0.0007980472291819751, "num_tokens": 152442886.0, "reward": 2.3207521438598633, "reward_std": 0.529423177242279, "rewards/code_complexity_reward/mean": 0.916308581829071, "rewards/code_complexity_reward/std": 0.144205242395401, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843363426625729, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 981, "step_time": 43.766742719337344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 391.0, "completions/max_terminated_length": 391.0, "completions/mean_length": 122.376953125, "completions/mean_terminated_length": 122.376953125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2338870947714895, "epoch": 0.5595441595441596, "frac_reward_zero_std": 0.609375, "grad_norm": 0.04827689379453659, "kl": 0.16139357339125127, "learning_rate": 2.4229148700704076e-06, "loss": 0.000807209056802094, "num_tokens": 152575239.0, "reward": 2.280810594558716, "reward_std": 0.4702126979827881, "rewards/code_complexity_reward/mean": 0.9244140386581421, "rewards/code_complexity_reward/std": 0.09742099046707153, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 982, "step_time": 58.18645931221545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 126.3046875, "completions/mean_terminated_length": 126.3046875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24478064849972725, "epoch": 0.5601139601139601, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05492416024208069, "kl": 0.15666586416773498, "learning_rate": 2.417943368415754e-06, "loss": 0.0007833601557649672, "num_tokens": 152708427.0, "reward": 2.271240472793579, "reward_std": 0.44787412881851196, "rewards/code_complexity_reward/mean": 0.926562488079071, "rewards/code_complexity_reward/std": 0.07799781858921051, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 983, "step_time": 34.22884755767882 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 135.962890625, "completions/mean_terminated_length": 134.48825073242188, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22806353261694312, "epoch": 0.5606837606837607, "frac_reward_zero_std": 0.375, "grad_norm": 0.06680084764957428, "kl": 0.15966003062203526, "learning_rate": 2.412972191585786e-06, "loss": 0.0007984916446730494, "num_tokens": 152847664.0, "reward": 2.3050782680511475, "reward_std": 0.5426624417304993, "rewards/code_complexity_reward/mean": 0.90478515625, "rewards/code_complexity_reward/std": 0.1499561369419098, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 984, "step_time": 51.234781810082495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 127.802734375, "completions/mean_terminated_length": 127.802734375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.21991895698010921, "epoch": 0.5612535612535613, "frac_reward_zero_std": 0.609375, "grad_norm": 0.050014857202768326, "kl": 0.16162324196193367, "learning_rate": 2.408001359259121e-06, "loss": 0.0008080514380708337, "num_tokens": 152981139.0, "reward": 2.235644578933716, "reward_std": 0.4684717059135437, "rewards/code_complexity_reward/mean": 0.9198241829872131, "rewards/code_complexity_reward/std": 0.12532763183116913, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 985, "step_time": 34.55988032743335 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 279.0, "completions/max_terminated_length": 279.0, "completions/mean_length": 117.984375, "completions/mean_terminated_length": 117.984375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22623692895285785, "epoch": 0.5618233618233618, "frac_reward_zero_std": 0.359375, "grad_norm": 0.07449119538068771, "kl": 0.16777464386541396, "learning_rate": 2.4030308911130095e-06, "loss": 0.0008389425347559154, "num_tokens": 153107923.0, "reward": 2.3443849086761475, "reward_std": 0.517448902130127, "rewards/code_complexity_reward/mean": 0.9186522960662842, "rewards/code_complexity_reward/std": 0.11902547627687454, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 986, "step_time": 32.929619835689664 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 126.810546875, "completions/mean_terminated_length": 125.30001068115234, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23336063092574477, "epoch": 0.5623931623931624, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06505198031663895, "kl": 0.1672701039351523, "learning_rate": 2.3980608068232642e-06, "loss": 0.0008364269742742181, "num_tokens": 153243154.0, "reward": 2.3218750953674316, "reward_std": 0.542535126209259, "rewards/code_complexity_reward/mean": 0.9107421636581421, "rewards/code_complexity_reward/std": 0.14494696259498596, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 987, "step_time": 55.95433978643268 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 122.25390625, "completions/mean_terminated_length": 122.25390625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2408806956373155, "epoch": 0.562962962962963, "frac_reward_zero_std": 0.578125, "grad_norm": 0.06802484393119812, "kl": 0.24588174396194518, "learning_rate": 2.3930911260641744e-06, "loss": 0.0012289542937651277, "num_tokens": 153375700.0, "reward": 2.2560060024261475, "reward_std": 0.4568668603897095, "rewards/code_complexity_reward/mean": 0.923046886920929, "rewards/code_complexity_reward/std": 0.0983528345823288, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 988, "step_time": 50.77695545274764 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 295.0, "completions/max_terminated_length": 295.0, "completions/mean_length": 122.91015625, "completions/mean_terminated_length": 122.91015625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24250921281054616, "epoch": 0.5635327635327635, "frac_reward_zero_std": 0.546875, "grad_norm": 0.0566774383187294, "kl": 0.1584868400823325, "learning_rate": 2.3881218685084364e-06, "loss": 0.0007925409590825438, "num_tokens": 153507390.0, "reward": 2.2414064407348633, "reward_std": 0.4726930856704712, "rewards/code_complexity_reward/mean": 0.919726550579071, "rewards/code_complexity_reward/std": 0.12563546001911163, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 989, "step_time": 36.78055451717228 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 437.0, "completions/max_terminated_length": 437.0, "completions/mean_length": 127.400390625, "completions/mean_terminated_length": 127.400390625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23974392958916724, "epoch": 0.5641025641025641, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05684870854020119, "kl": 0.16071464400738478, "learning_rate": 2.3831530538270664e-06, "loss": 0.0008037190418690443, "num_tokens": 153639811.0, "reward": 2.310839891433716, "reward_std": 0.5210018157958984, "rewards/code_complexity_reward/mean": 0.9120116829872131, "rewards/code_complexity_reward/std": 0.13347403705120087, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 990, "step_time": 44.47652171365917 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 130.953125, "completions/mean_terminated_length": 130.953125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2206137317698449, "epoch": 0.5646723646723647, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06363069266080856, "kl": 0.1523681979160756, "learning_rate": 2.3781847016893318e-06, "loss": 0.0007618748350068927, "num_tokens": 153777155.0, "reward": 2.3175783157348633, "reward_std": 0.5214678049087524, "rewards/code_complexity_reward/mean": 0.90966796875, "rewards/code_complexity_reward/std": 0.1313384622335434, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 991, "step_time": 40.43006445840001 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 122.79296875, "completions/mean_terminated_length": 122.79296875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23497396940365434, "epoch": 0.5652421652421652, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06343931704759598, "kl": 0.18835491745267063, "learning_rate": 2.373216831762665e-06, "loss": 0.0009414087398909032, "num_tokens": 153909609.0, "reward": 2.327441692352295, "reward_std": 0.49184533953666687, "rewards/code_complexity_reward/mean": 0.9256835579872131, "rewards/code_complexity_reward/std": 0.09513365477323532, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 992, "step_time": 37.18443303834647 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 125.15625, "completions/mean_terminated_length": 125.15625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2383692436851561, "epoch": 0.5658119658119658, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06927792727947235, "kl": 0.21126187895424664, "learning_rate": 2.3682494637125923e-06, "loss": 0.001056739711202681, "num_tokens": 154043161.0, "reward": 2.3622560501098633, "reward_std": 0.489635705947876, "rewards/code_complexity_reward/mean": 0.9306640625, "rewards/code_complexity_reward/std": 0.06544405221939087, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 993, "step_time": 93.14540679007769 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 379.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 121.04296875, "completions/mean_terminated_length": 121.04296875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.23570705787278712, "epoch": 0.5663817663817664, "frac_reward_zero_std": 0.5, "grad_norm": 0.0665614902973175, "kl": 0.16009006928652525, "learning_rate": 2.3632826172026546e-06, "loss": 0.0008005071431398392, "num_tokens": 154172623.0, "reward": 2.3382816314697266, "reward_std": 0.4970971345901489, "rewards/code_complexity_reward/mean": 0.9276366829872131, "rewards/code_complexity_reward/std": 0.09632626175880432, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 994, "step_time": 40.211982293985784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 128.1796875, "completions/mean_terminated_length": 128.1796875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2298670771997422, "epoch": 0.5669515669515669, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05410550534725189, "kl": 0.15819251991342753, "learning_rate": 2.358316311894324e-06, "loss": 0.0007908776169642806, "num_tokens": 154306731.0, "reward": 2.3952150344848633, "reward_std": 0.5293735861778259, "rewards/code_complexity_reward/mean": 0.919238269329071, "rewards/code_complexity_reward/std": 0.11478733271360397, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 995, "step_time": 47.224212038330734 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 271.0, "completions/max_terminated_length": 271.0, "completions/mean_length": 115.58984375, "completions/mean_terminated_length": 115.58984375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22960680723190308, "epoch": 0.5675213675213675, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05758349597454071, "kl": 0.15402465616352856, "learning_rate": 2.3533505674469337e-06, "loss": 0.0007702796719968319, "num_tokens": 154432225.0, "reward": 2.418457269668579, "reward_std": 0.5047873854637146, "rewards/code_complexity_reward/mean": 0.9346679449081421, "rewards/code_complexity_reward/std": 0.07529645413160324, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 996, "step_time": 33.67009346187115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 311.0, "completions/max_terminated_length": 311.0, "completions/mean_length": 124.044921875, "completions/mean_terminated_length": 124.044921875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2302614429499954, "epoch": 0.5680911680911681, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06349760293960571, "kl": 0.16346033825539052, "learning_rate": 2.3483854035175942e-06, "loss": 0.0008171434164978564, "num_tokens": 154564496.0, "reward": 2.3429689407348633, "reward_std": 0.5570921301841736, "rewards/code_complexity_reward/mean": 0.908984363079071, "rewards/code_complexity_reward/std": 0.15561047196388245, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 997, "step_time": 45.557935521006584 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 123.10546875, "completions/mean_terminated_length": 123.10546875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2372237048111856, "epoch": 0.5686609686609687, "frac_reward_zero_std": 0.53125, "grad_norm": 0.055137310177087784, "kl": 0.1628215272212401, "learning_rate": 2.3434208397611207e-06, "loss": 0.0008142509032040834, "num_tokens": 154695182.0, "reward": 2.3709962368011475, "reward_std": 0.500717282295227, "rewards/code_complexity_reward/mean": 0.92431640625, "rewards/code_complexity_reward/std": 0.08317054808139801, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 998, "step_time": 42.46939242631197 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 121.45703125, "completions/mean_terminated_length": 120.69275665283203, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22863019444048405, "epoch": 0.5692307692307692, "frac_reward_zero_std": 0.40625, "grad_norm": 0.07474376261234283, "kl": 0.1479831220349297, "learning_rate": 2.33845689582995e-06, "loss": 0.0007399716414511204, "num_tokens": 154826296.0, "reward": 2.3995118141174316, "reward_std": 0.5230380296707153, "rewards/code_complexity_reward/mean": 0.9251953363418579, "rewards/code_complexity_reward/std": 0.10463786125183105, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 999, "step_time": 48.83384631574154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 429.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 122.732421875, "completions/mean_terminated_length": 122.732421875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23162688547745347, "epoch": 0.5698005698005698, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06597262620925903, "kl": 0.16193143068812788, "learning_rate": 2.333493591374068e-06, "loss": 0.0008095219964161515, "num_tokens": 154957871.0, "reward": 2.3750977516174316, "reward_std": 0.5414248704910278, "rewards/code_complexity_reward/mean": 0.9147460460662842, "rewards/code_complexity_reward/std": 0.128754124045372, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1000, "step_time": 43.00372860394418 }, { "epoch": 0.5698005698005698, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 172.25, "eval_completions/max_terminated_length": 172.25, "eval_completions/mean_length": 126.7525, "eval_completions/mean_terminated_length": 126.7525, "eval_completions/min_length": 93.43, "eval_completions/min_terminated_length": 93.43, "eval_entropy": 0.22581260725855828, "eval_frac_reward_zero_std": 0.51, "eval_kl": 0.16049077797681094, "eval_loss": 0.0008028876618482172, "eval_num_tokens": 154957871.0, "eval_reward": 2.3036876273155213, "eval_reward_std": 0.20808560203760862, "eval_rewards/code_complexity_reward/mean": 0.9186874842643737, "eval_rewards/code_complexity_reward/std": 0.03534294959157705, "eval_rewards/code_execution_reward/mean": 0.2925, "eval_rewards/code_execution_reward/std": 0.16530491828918456, "eval_rewards/code_syntax_reward/mean": 0.4925, "eval_rewards/code_syntax_reward/std": 0.01632926881313324, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.5, "eval_rewards/xmlcount_reward_func/std": 0.0, "eval_runtime": 799.4717, "eval_samples_per_second": 0.125, "eval_steps_per_second": 0.016, "step": 1000 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 319.0, "completions/max_terminated_length": 319.0, "completions/mean_length": 122.572265625, "completions/mean_terminated_length": 122.572265625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24858853011392057, "epoch": 0.5703703703703704, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06485448032617569, "kl": 0.16125181573443115, "learning_rate": 2.3285309460409257e-06, "loss": 0.0008061715634539723, "num_tokens": 155091292.0, "reward": 2.2831544876098633, "reward_std": 0.5336849689483643, "rewards/code_complexity_reward/mean": 0.908203125, "rewards/code_complexity_reward/std": 0.1547078937292099, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1001, "step_time": 45.40301879774779 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 126.658203125, "completions/mean_terminated_length": 125.90410614013672, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23546271538361907, "epoch": 0.5709401709401709, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05818377807736397, "kl": 0.1682536203879863, "learning_rate": 2.3235689794753683e-06, "loss": 0.000841150525957346, "num_tokens": 155225733.0, "reward": 2.351611375808716, "reward_std": 0.5309672951698303, "rewards/code_complexity_reward/mean": 0.9149413704872131, "rewards/code_complexity_reward/std": 0.12615957856178284, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1002, "step_time": 69.8568269284442 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 438.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 124.564453125, "completions/mean_terminated_length": 124.564453125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22404697630554438, "epoch": 0.5715099715099715, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05709546059370041, "kl": 0.15449402492959052, "learning_rate": 2.3186077113195528e-06, "loss": 0.0007727643242105842, "num_tokens": 155355086.0, "reward": 2.3731446266174316, "reward_std": 0.4986248016357422, "rewards/code_complexity_reward/mean": 0.9206054210662842, "rewards/code_complexity_reward/std": 0.09352952986955643, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1003, "step_time": 81.44445139076561 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 124.935546875, "completions/mean_terminated_length": 124.935546875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23245769459754229, "epoch": 0.5720797720797721, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05491653457283974, "kl": 0.17519260896369815, "learning_rate": 2.3136471612128704e-06, "loss": 0.0008762275683693588, "num_tokens": 155488669.0, "reward": 2.393359422683716, "reward_std": 0.5265224575996399, "rewards/code_complexity_reward/mean": 0.9251953363418579, "rewards/code_complexity_reward/std": 0.11112289875745773, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1004, "step_time": 38.175576601177454 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 130.0390625, "completions/mean_terminated_length": 130.0390625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23003484937362373, "epoch": 0.5726495726495726, "frac_reward_zero_std": 0.5625, "grad_norm": 0.060876358300447464, "kl": 0.14174700644798577, "learning_rate": 2.308687348791872e-06, "loss": 0.0007087607518769801, "num_tokens": 155623833.0, "reward": 2.2137694358825684, "reward_std": 0.46188467741012573, "rewards/code_complexity_reward/mean": 0.90869140625, "rewards/code_complexity_reward/std": 0.12334083020687103, "rewards/code_execution_reward/mean": 0.212890625, "rewards/code_execution_reward/std": 0.409751296043396, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1005, "step_time": 47.25885928608477 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 424.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 127.4453125, "completions/mean_terminated_length": 127.4453125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23366079246625304, "epoch": 0.5732193732193732, "frac_reward_zero_std": 0.359375, "grad_norm": 0.07581678777933121, "kl": 0.15196773584466428, "learning_rate": 2.303728293690186e-06, "loss": 0.0007594700437039137, "num_tokens": 155758653.0, "reward": 2.3641114234924316, "reward_std": 0.5599983930587769, "rewards/code_complexity_reward/mean": 0.9120116829872131, "rewards/code_complexity_reward/std": 0.14971354603767395, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1006, "step_time": 49.68445199448615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 310.0, "completions/mean_length": 122.67578125, "completions/mean_terminated_length": 121.91389465332031, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2308972345199436, "epoch": 0.5737891737891738, "frac_reward_zero_std": 0.53125, "grad_norm": 0.06117118149995804, "kl": 0.15472450107336044, "learning_rate": 2.298770015538446e-06, "loss": 0.0007733589736744761, "num_tokens": 155889671.0, "reward": 2.313720941543579, "reward_std": 0.5145852565765381, "rewards/code_complexity_reward/mean": 0.9200195074081421, "rewards/code_complexity_reward/std": 0.12653982639312744, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1007, "step_time": 48.17107760254294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 462.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 127.03125, "completions/mean_terminated_length": 127.03125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22735199565067887, "epoch": 0.5743589743589743, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06602585315704346, "kl": 0.17196625168435276, "learning_rate": 2.293812533964206e-06, "loss": 0.0008598656859248877, "num_tokens": 156026799.0, "reward": 2.39501953125, "reward_std": 0.5321738123893738, "rewards/code_complexity_reward/mean": 0.9180663824081421, "rewards/code_complexity_reward/std": 0.11907492578029633, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1008, "step_time": 46.140977358445525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 118.84765625, "completions/mean_terminated_length": 118.84765625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22015407215803862, "epoch": 0.5749287749287749, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05147288739681244, "kl": 0.14989876956678927, "learning_rate": 2.2888558685918704e-06, "loss": 0.0007497785845771432, "num_tokens": 156155553.0, "reward": 2.480517625808716, "reward_std": 0.5230804085731506, "rewards/code_complexity_reward/mean": 0.9256835579872131, "rewards/code_complexity_reward/std": 0.0889684408903122, "rewards/code_execution_reward/mean": 0.4609375, "rewards/code_execution_reward/std": 0.4989593029022217, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.03168932721018791, "step": 1009, "step_time": 58.278938544914126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 423.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 130.947265625, "completions/mean_terminated_length": 130.947265625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2300042067654431, "epoch": 0.5754985754985755, "frac_reward_zero_std": 0.5625, "grad_norm": 0.056047920137643814, "kl": 0.17367696831934154, "learning_rate": 2.2839000390426097e-06, "loss": 0.0008686420042067766, "num_tokens": 156292398.0, "reward": 2.2711915969848633, "reward_std": 0.4764668643474579, "rewards/code_complexity_reward/mean": 0.9153320789337158, "rewards/code_complexity_reward/std": 0.1157139241695404, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1010, "step_time": 50.363751573488116 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 416.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 124.333984375, "completions/mean_terminated_length": 124.333984375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22609948180615902, "epoch": 0.576068376068376, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06853087246417999, "kl": 0.15439089701976627, "learning_rate": 2.278945064934289e-06, "loss": 0.0007722359732724726, "num_tokens": 156422961.0, "reward": 2.358447313308716, "reward_std": 0.5231464505195618, "rewards/code_complexity_reward/mean": 0.9209961295127869, "rewards/code_complexity_reward/std": 0.11937370151281357, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1011, "step_time": 58.01510786637664 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 330.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 126.595703125, "completions/mean_terminated_length": 126.595703125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22793809860013425, "epoch": 0.5766381766381766, "frac_reward_zero_std": 0.609375, "grad_norm": 0.048210494220256805, "kl": 0.1591753379907459, "learning_rate": 2.2739909658813834e-06, "loss": 0.0007958381902426481, "num_tokens": 156556954.0, "reward": 2.301025629043579, "reward_std": 0.48008039593696594, "rewards/code_complexity_reward/mean": 0.927929699420929, "rewards/code_complexity_reward/std": 0.09727807343006134, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1012, "step_time": 37.15091550536454 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 121.9375, "completions/mean_terminated_length": 121.9375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22459562472067773, "epoch": 0.5772079772079772, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05824980139732361, "kl": 0.17701945616863668, "learning_rate": 2.269037761494907e-06, "loss": 0.0008850476006045938, "num_tokens": 156689362.0, "reward": 2.3501954078674316, "reward_std": 0.5244860053062439, "rewards/code_complexity_reward/mean": 0.9232422113418579, "rewards/code_complexity_reward/std": 0.1188531145453453, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 1013, "step_time": 40.1881505176425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 469.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 119.623046875, "completions/mean_terminated_length": 119.623046875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23137104720808566, "epoch": 0.5777777777777777, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05952807143330574, "kl": 0.15071424399502575, "learning_rate": 2.264085471382331e-06, "loss": 0.0007532780291512609, "num_tokens": 156818345.0, "reward": 2.37890625, "reward_std": 0.5346729159355164, "rewards/code_complexity_reward/mean": 0.9175781011581421, "rewards/code_complexity_reward/std": 0.1205969899892807, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1014, "step_time": 54.77729028556496 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 326.0, "completions/mean_length": 123.462890625, "completions/mean_terminated_length": 122.70254516601562, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23055773740634322, "epoch": 0.5783475783475783, "frac_reward_zero_std": 0.546875, "grad_norm": 0.054221343249082565, "kl": 0.1655066245002672, "learning_rate": 2.2591341151475077e-06, "loss": 0.0008275557775050402, "num_tokens": 156952590.0, "reward": 2.345263957977295, "reward_std": 0.504515528678894, "rewards/code_complexity_reward/mean": 0.9182616472244263, "rewards/code_complexity_reward/std": 0.10502774268388748, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1015, "step_time": 56.413866576738656 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 479.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 128.892578125, "completions/mean_terminated_length": 128.892578125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24005603813566267, "epoch": 0.5789173789173789, "frac_reward_zero_std": 0.53125, "grad_norm": 0.056002676486968994, "kl": 0.15479426318779588, "learning_rate": 2.254183712390593e-06, "loss": 0.0007740338915027678, "num_tokens": 157088175.0, "reward": 2.33984375, "reward_std": 0.5238764882087708, "rewards/code_complexity_reward/mean": 0.9166015386581421, "rewards/code_complexity_reward/std": 0.12655209004878998, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1016, "step_time": 46.176751440390944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 417.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 119.09765625, "completions/mean_terminated_length": 119.09765625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22548814653418958, "epoch": 0.5794871794871795, "frac_reward_zero_std": 0.484375, "grad_norm": 0.056901950389146805, "kl": 0.1736004534177482, "learning_rate": 2.2492342827079663e-06, "loss": 0.0008678381564095616, "num_tokens": 157216697.0, "reward": 2.39501953125, "reward_std": 0.5141915678977966, "rewards/code_complexity_reward/mean": 0.9229491949081421, "rewards/code_complexity_reward/std": 0.09669535607099533, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1017, "step_time": 99.54848542343825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 120.927734375, "completions/mean_terminated_length": 120.927734375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23221109271980822, "epoch": 0.58005698005698, "frac_reward_zero_std": 0.609375, "grad_norm": 0.05034927651286125, "kl": 0.15567261318210512, "learning_rate": 2.244285845692159e-06, "loss": 0.000778401386924088, "num_tokens": 157345020.0, "reward": 2.342334032058716, "reward_std": 0.48204663395881653, "rewards/code_complexity_reward/mean": 0.9292968511581421, "rewards/code_complexity_reward/std": 0.07566465437412262, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1018, "step_time": 44.50979075115174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 494.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 124.466796875, "completions/mean_terminated_length": 124.466796875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2441985725890845, "epoch": 0.5806267806267806, "frac_reward_zero_std": 0.5, "grad_norm": 0.05833033472299576, "kl": 0.16382356558460742, "learning_rate": 2.2393384209317688e-06, "loss": 0.000819142849650234, "num_tokens": 157478291.0, "reward": 2.3208985328674316, "reward_std": 0.48343425989151, "rewards/code_complexity_reward/mean": 0.9247070550918579, "rewards/code_complexity_reward/std": 0.08852958679199219, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1019, "step_time": 67.27159017324448 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 130.58984375, "completions/mean_terminated_length": 130.58984375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2363176520448178, "epoch": 0.5811965811965812, "frac_reward_zero_std": 0.453125, "grad_norm": 0.0625520646572113, "kl": 0.17626387649215758, "learning_rate": 2.2343920280113897e-06, "loss": 0.0008810353465378284, "num_tokens": 157614761.0, "reward": 2.3740720748901367, "reward_std": 0.5157052874565125, "rewards/code_complexity_reward/mean": 0.9198242425918579, "rewards/code_complexity_reward/std": 0.10030563920736313, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1020, "step_time": 45.060386918485165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 394.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 121.060546875, "completions/mean_terminated_length": 121.060546875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.23341197008267045, "epoch": 0.5817663817663817, "frac_reward_zero_std": 0.625, "grad_norm": 0.05277293175458908, "kl": 0.16129924124106765, "learning_rate": 2.2294466865115282e-06, "loss": 0.000806471158284694, "num_tokens": 157744648.0, "reward": 2.3224122524261475, "reward_std": 0.47729194164276123, "rewards/code_complexity_reward/mean": 0.9235351085662842, "rewards/code_complexity_reward/std": 0.07762246578931808, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1021, "step_time": 50.073753422126174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 121.072265625, "completions/mean_terminated_length": 121.072265625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2318434943445027, "epoch": 0.5823361823361823, "frac_reward_zero_std": 0.578125, "grad_norm": 0.056702621281147, "kl": 0.16501279908698052, "learning_rate": 2.2245024160085304e-06, "loss": 0.0008251176914200187, "num_tokens": 157877477.0, "reward": 2.2728514671325684, "reward_std": 0.4714711010456085, "rewards/code_complexity_reward/mean": 0.9248046875, "rewards/code_complexity_reward/std": 0.10365134477615356, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1022, "step_time": 39.355106715112925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 338.0, "completions/max_terminated_length": 338.0, "completions/mean_length": 127.40625, "completions/mean_terminated_length": 127.40625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2325170268304646, "epoch": 0.582905982905983, "frac_reward_zero_std": 0.5, "grad_norm": 0.05703527852892876, "kl": 0.15891556278802454, "learning_rate": 2.2195592360745044e-06, "loss": 0.0007944563985802233, "num_tokens": 158011941.0, "reward": 2.288818359375, "reward_std": 0.5144755840301514, "rewards/code_complexity_reward/mean": 0.914843738079071, "rewards/code_complexity_reward/std": 0.1417819708585739, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1023, "step_time": 37.53926398605108 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 121.693359375, "completions/mean_terminated_length": 121.693359375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23516896204091609, "epoch": 0.5834757834757834, "frac_reward_zero_std": 0.5, "grad_norm": 0.061980653554201126, "kl": 0.15287324180826545, "learning_rate": 2.214617166277237e-06, "loss": 0.0007645492441952229, "num_tokens": 158144032.0, "reward": 2.372363328933716, "reward_std": 0.49424809217453003, "rewards/code_complexity_reward/mean": 0.922656238079071, "rewards/code_complexity_reward/std": 0.080082967877388, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1024, "step_time": 43.65558885037899 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 124.21484375, "completions/mean_terminated_length": 124.21484375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22648475458845496, "epoch": 0.584045584045584, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06748396158218384, "kl": 0.17169025854673237, "learning_rate": 2.209676226180125e-06, "loss": 0.000858383602462709, "num_tokens": 158275550.0, "reward": 2.3124024868011475, "reward_std": 0.5006845593452454, "rewards/code_complexity_reward/mean": 0.9152343273162842, "rewards/code_complexity_reward/std": 0.1038982942700386, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1025, "step_time": 48.7996532516554 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 125.818359375, "completions/mean_terminated_length": 125.818359375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2214851756580174, "epoch": 0.5846153846153846, "frac_reward_zero_std": 0.515625, "grad_norm": 0.058685120195150375, "kl": 0.16083187237381935, "learning_rate": 2.2047364353420895e-06, "loss": 0.0008044099085964262, "num_tokens": 158408177.0, "reward": 2.3893556594848633, "reward_std": 0.5157418847084045, "rewards/code_complexity_reward/mean": 0.9220702648162842, "rewards/code_complexity_reward/std": 0.09951551258563995, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1026, "step_time": 46.7836876148358 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 133.2109375, "completions/mean_terminated_length": 133.2109375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.23131700209341943, "epoch": 0.5851851851851851, "frac_reward_zero_std": 0.5, "grad_norm": 0.057377420365810394, "kl": 0.1491797547787428, "learning_rate": 2.1997978133175045e-06, "loss": 0.0007460084743797779, "num_tokens": 158544277.0, "reward": 2.3326661586761475, "reward_std": 0.47945162653923035, "rewards/code_complexity_reward/mean": 0.9227538704872131, "rewards/code_complexity_reward/std": 0.07922591269016266, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 1027, "step_time": 40.1402899492532 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 296.0, "completions/max_terminated_length": 296.0, "completions/mean_length": 115.91796875, "completions/mean_terminated_length": 115.91796875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2281700058374554, "epoch": 0.5857549857549857, "frac_reward_zero_std": 0.515625, "grad_norm": 0.06149086728692055, "kl": 0.1579522801330313, "learning_rate": 2.1948603796561163e-06, "loss": 0.0007896603783592582, "num_tokens": 158669427.0, "reward": 2.354492425918579, "reward_std": 0.4969543516635895, "rewards/code_complexity_reward/mean": 0.9288085699081421, "rewards/code_complexity_reward/std": 0.0871664509177208, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1028, "step_time": 34.26087632589042 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 127.0859375, "completions/mean_terminated_length": 126.33267974853516, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23743768269196153, "epoch": 0.5863247863247864, "frac_reward_zero_std": 0.4375, "grad_norm": 0.07057392597198486, "kl": 0.15771814598701894, "learning_rate": 2.189924153902968e-06, "loss": 0.0007885646773502231, "num_tokens": 158802471.0, "reward": 2.373779535293579, "reward_std": 0.5185853838920593, "rewards/code_complexity_reward/mean": 0.9214843511581421, "rewards/code_complexity_reward/std": 0.09796033054590225, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1029, "step_time": 53.949344732798636 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 126.26171875, "completions/mean_terminated_length": 124.7490234375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2285554560367018, "epoch": 0.5868945868945868, "frac_reward_zero_std": 0.625, "grad_norm": 0.05282042548060417, "kl": 0.16754080285318196, "learning_rate": 2.1849891555983186e-06, "loss": 0.0008374980534426868, "num_tokens": 158935909.0, "reward": 2.3779296875, "reward_std": 0.5275099873542786, "rewards/code_complexity_reward/mean": 0.9209960699081421, "rewards/code_complexity_reward/std": 0.12006840109825134, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1030, "step_time": 50.46450274530798 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 122.6015625, "completions/mean_terminated_length": 122.6015625, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.23904240760020912, "epoch": 0.5874643874643874, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05992846190929413, "kl": 0.15351308847311884, "learning_rate": 2.1800554042775724e-06, "loss": 0.0007673670770600438, "num_tokens": 159066593.0, "reward": 2.275683879852295, "reward_std": 0.47721007466316223, "rewards/code_complexity_reward/mean": 0.9237304329872131, "rewards/code_complexity_reward/std": 0.10463032126426697, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1031, "step_time": 52.8974651126191 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 120.2890625, "completions/mean_terminated_length": 120.2890625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22776977811008692, "epoch": 0.588034188034188, "frac_reward_zero_std": 0.5, "grad_norm": 0.06196313351392746, "kl": 0.1489278784720227, "learning_rate": 2.1751229194711925e-06, "loss": 0.0007446244126185775, "num_tokens": 159199405.0, "reward": 2.386230707168579, "reward_std": 0.5291141271591187, "rewards/code_complexity_reward/mean": 0.9239257574081421, "rewards/code_complexity_reward/std": 0.11782776564359665, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1032, "step_time": 37.2531651314348 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 128.083984375, "completions/mean_terminated_length": 128.083984375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.22135092969983816, "epoch": 0.5886039886039887, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05083516240119934, "kl": 0.14910297794267535, "learning_rate": 2.170191720704633e-06, "loss": 0.0007454039878211915, "num_tokens": 159331480.0, "reward": 2.365966796875, "reward_std": 0.48342403769493103, "rewards/code_complexity_reward/mean": 0.932421863079071, "rewards/code_complexity_reward/std": 0.0629788413643837, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1033, "step_time": 35.128837826661766 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 125.330078125, "completions/mean_terminated_length": 125.330078125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2357555686030537, "epoch": 0.5891737891737892, "frac_reward_zero_std": 0.484375, "grad_norm": 0.21618512272834778, "kl": 0.30640779179520905, "learning_rate": 2.165261827498255e-06, "loss": 0.0015344570856541395, "num_tokens": 159463097.0, "reward": 2.3385252952575684, "reward_std": 0.48388051986694336, "rewards/code_complexity_reward/mean": 0.9274413585662842, "rewards/code_complexity_reward/std": 0.07677384465932846, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1034, "step_time": 38.12699946667999 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 122.22265625, "completions/mean_terminated_length": 122.22265625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22209472605027258, "epoch": 0.5897435897435898, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05475519597530365, "kl": 0.15821062889881432, "learning_rate": 2.1603332593672497e-06, "loss": 0.0007913920562714338, "num_tokens": 159592843.0, "reward": 2.3150391578674316, "reward_std": 0.497565895318985, "rewards/code_complexity_reward/mean": 0.926953136920929, "rewards/code_complexity_reward/std": 0.11281009763479233, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1035, "step_time": 43.458715884014964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 121.75, "completions/mean_terminated_length": 121.75, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2305290913209319, "epoch": 0.5903133903133904, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05586186796426773, "kl": 0.16345367347821593, "learning_rate": 2.1554060358215674e-06, "loss": 0.0008171148365363479, "num_tokens": 159722275.0, "reward": 2.333740472793579, "reward_std": 0.5031753778457642, "rewards/code_complexity_reward/mean": 0.9214843511581421, "rewards/code_complexity_reward/std": 0.10425086319446564, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1036, "step_time": 44.16129305027425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 382.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 126.595703125, "completions/mean_terminated_length": 126.595703125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24026170396246016, "epoch": 0.5908831908831909, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05198875069618225, "kl": 0.15378091938327998, "learning_rate": 2.150480176365831e-06, "loss": 0.0007688846671953797, "num_tokens": 159854116.0, "reward": 2.313525676727295, "reward_std": 0.4929978847503662, "rewards/code_complexity_reward/mean": 0.9200195074081421, "rewards/code_complexity_reward/std": 0.10224772989749908, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1037, "step_time": 39.97176253516227 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 121.50390625, "completions/mean_terminated_length": 120.7397232055664, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2283136723563075, "epoch": 0.5914529914529915, "frac_reward_zero_std": 0.5, "grad_norm": 0.054373849183321, "kl": 0.15716313512530178, "learning_rate": 2.1455557004992684e-06, "loss": 0.0007857922464609146, "num_tokens": 159985934.0, "reward": 2.388427734375, "reward_std": 0.49189531803131104, "rewards/code_complexity_reward/mean": 0.9283202886581421, "rewards/code_complexity_reward/std": 0.057070620357990265, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1038, "step_time": 58.47581253387034 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 477.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 126.63671875, "completions/mean_terminated_length": 126.63671875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23603624640963972, "epoch": 0.5920227920227921, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05589143931865692, "kl": 0.1675294324522838, "learning_rate": 2.1406326277156257e-06, "loss": 0.0008376055047847331, "num_tokens": 160119204.0, "reward": 2.2997071743011475, "reward_std": 0.4795473515987396, "rewards/code_complexity_reward/mean": 0.92919921875, "rewards/code_complexity_reward/std": 0.09606719762086868, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1039, "step_time": 54.2838700665161 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 290.0, "completions/max_terminated_length": 290.0, "completions/mean_length": 126.4609375, "completions/mean_terminated_length": 126.4609375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23242000746540725, "epoch": 0.5925925925925926, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06058115139603615, "kl": 0.17283728357870132, "learning_rate": 2.135710977503098e-06, "loss": 0.000864196743350476, "num_tokens": 160253416.0, "reward": 2.324023485183716, "reward_std": 0.4955886900424957, "rewards/code_complexity_reward/mean": 0.9217773079872131, "rewards/code_complexity_reward/std": 0.1042134165763855, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1040, "step_time": 44.278491511940956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 126.830078125, "completions/mean_terminated_length": 126.830078125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23456407291814685, "epoch": 0.5931623931623932, "frac_reward_zero_std": 0.53125, "grad_norm": 0.058322008699178696, "kl": 0.17497618473134935, "learning_rate": 2.130790769344248e-06, "loss": 0.0008749777916818857, "num_tokens": 160390145.0, "reward": 2.2298340797424316, "reward_std": 0.45907869935035706, "rewards/code_complexity_reward/mean": 0.9232422113418579, "rewards/code_complexity_reward/std": 0.1188119426369667, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1041, "step_time": 54.82664940971881 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 119.51171875, "completions/mean_terminated_length": 119.51171875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2244573337957263, "epoch": 0.5937321937321938, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05257266387343407, "kl": 0.1701350490329787, "learning_rate": 2.12587202271593e-06, "loss": 0.0008506180602125823, "num_tokens": 160519439.0, "reward": 2.423095941543579, "reward_std": 0.5153894424438477, "rewards/code_complexity_reward/mean": 0.9231444597244263, "rewards/code_complexity_reward/std": 0.08694912493228912, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1042, "step_time": 44.51921568159014 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 122.634765625, "completions/mean_terminated_length": 122.634765625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2321073543280363, "epoch": 0.5943019943019943, "frac_reward_zero_std": 0.65625, "grad_norm": 0.0526898168027401, "kl": 0.1572190555743873, "learning_rate": 2.1209547570892135e-06, "loss": 0.0007861003396101296, "num_tokens": 160648724.0, "reward": 2.3189454078674316, "reward_std": 0.48731711506843567, "rewards/code_complexity_reward/mean": 0.925000011920929, "rewards/code_complexity_reward/std": 0.09656528383493423, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1043, "step_time": 38.633292177692056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 261.0, "completions/max_terminated_length": 261.0, "completions/mean_length": 121.072265625, "completions/mean_terminated_length": 121.072265625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23141890321858227, "epoch": 0.5948717948717949, "frac_reward_zero_std": 0.625, "grad_norm": 0.048431120812892914, "kl": 0.159606566070579, "learning_rate": 2.116038991929304e-06, "loss": 0.0007980180671438575, "num_tokens": 160778737.0, "reward": 2.3619141578674316, "reward_std": 0.4851762056350708, "rewards/code_complexity_reward/mean": 0.9308593273162842, "rewards/code_complexity_reward/std": 0.07549473643302917, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1044, "step_time": 34.50497299153358 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 280.0, "completions/max_terminated_length": 280.0, "completions/mean_length": 122.123046875, "completions/mean_terminated_length": 122.123046875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23419300513342023, "epoch": 0.5954415954415955, "frac_reward_zero_std": 0.609375, "grad_norm": 0.0588151253759861, "kl": 0.21725855604745448, "learning_rate": 2.1111247466954697e-06, "loss": 0.0010875824373215437, "num_tokens": 160909952.0, "reward": 2.2842774391174316, "reward_std": 0.488249272108078, "rewards/code_complexity_reward/mean": 0.919628918170929, "rewards/code_complexity_reward/std": 0.11230749636888504, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1045, "step_time": 46.22330033779144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 122.95703125, "completions/mean_terminated_length": 122.95703125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2277402044273913, "epoch": 0.596011396011396, "frac_reward_zero_std": 0.5625, "grad_norm": 0.07302623987197876, "kl": 0.15942930232267827, "learning_rate": 2.1062120408409588e-06, "loss": 0.00079708406701684, "num_tokens": 161040306.0, "reward": 2.3249025344848633, "reward_std": 0.5219637751579285, "rewards/code_complexity_reward/mean": 0.920214831829071, "rewards/code_complexity_reward/std": 0.13240420818328857, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1046, "step_time": 40.169247166253626 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 125.701171875, "completions/mean_terminated_length": 125.701171875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23498800373636186, "epoch": 0.5965811965811966, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06584423780441284, "kl": 0.17654832126572728, "learning_rate": 2.1013008938129287e-06, "loss": 0.0008827798301354051, "num_tokens": 161172753.0, "reward": 2.3248047828674316, "reward_std": 0.503332793712616, "rewards/code_complexity_reward/mean": 0.9188475608825684, "rewards/code_complexity_reward/std": 0.10571367293596268, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1047, "step_time": 34.6675154492259 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 429.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 129.041015625, "completions/mean_terminated_length": 129.041015625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22924511600285769, "epoch": 0.5971509971509972, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05363794416189194, "kl": 0.15827707666903734, "learning_rate": 2.0963913250523633e-06, "loss": 0.0007917354814708233, "num_tokens": 161307174.0, "reward": 2.3470702171325684, "reward_std": 0.5118829607963562, "rewards/code_complexity_reward/mean": 0.9168944954872131, "rewards/code_complexity_reward/std": 0.1120557114481926, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1048, "step_time": 43.450517379678786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 447.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 134.693359375, "completions/mean_terminated_length": 134.693359375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2233181744813919, "epoch": 0.5977207977207977, "frac_reward_zero_std": 0.46875, "grad_norm": 0.059563957154750824, "kl": 0.15668723243288696, "learning_rate": 2.091483353994003e-06, "loss": 0.00078344508074224, "num_tokens": 161443977.0, "reward": 2.254150390625, "reward_std": 0.5082188248634338, "rewards/code_complexity_reward/mean": 0.9058593511581421, "rewards/code_complexity_reward/std": 0.14298619329929352, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 1049, "step_time": 53.057346411049366 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 128.59375, "completions/mean_terminated_length": 127.84344482421875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23291198769584298, "epoch": 0.5982905982905983, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06424860656261444, "kl": 0.1551980460062623, "learning_rate": 2.0865770000662592e-06, "loss": 0.0007760653970763087, "num_tokens": 161578265.0, "reward": 2.2760744094848633, "reward_std": 0.4766990840435028, "rewards/code_complexity_reward/mean": 0.9248046875, "rewards/code_complexity_reward/std": 0.10547608882188797, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1050, "step_time": 48.986249602399766 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 127.69140625, "completions/mean_terminated_length": 126.9393310546875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2383693524170667, "epoch": 0.5988603988603989, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05144836753606796, "kl": 0.1732903152005747, "learning_rate": 2.081672282691146e-06, "loss": 0.0008666741196066141, "num_tokens": 161713763.0, "reward": 2.2276859283447266, "reward_std": 0.44137901067733765, "rewards/code_complexity_reward/mean": 0.9245116710662842, "rewards/code_complexity_reward/std": 0.09671591222286224, "rewards/code_execution_reward/mean": 0.208984375, "rewards/code_execution_reward/std": 0.40698084235191345, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 1051, "step_time": 50.142285759560764 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 396.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 124.966796875, "completions/mean_terminated_length": 124.966796875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23564808839000762, "epoch": 0.5994301994301995, "frac_reward_zero_std": 0.609375, "grad_norm": 0.0531722828745842, "kl": 0.17311902868095785, "learning_rate": 2.076769221284194e-06, "loss": 0.0008659341838210821, "num_tokens": 161848066.0, "reward": 2.319140672683716, "reward_std": 0.4938044846057892, "rewards/code_complexity_reward/mean": 0.9239257574081421, "rewards/code_complexity_reward/std": 0.10202844440937042, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1052, "step_time": 41.664639932103455 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 126.31640625, "completions/mean_terminated_length": 125.5616455078125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24256942700594664, "epoch": 0.6, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06841352581977844, "kl": 0.15977990627288818, "learning_rate": 2.071867835254383e-06, "loss": 0.000799204281065613, "num_tokens": 161979868.0, "reward": 2.2728517055511475, "reward_std": 0.5215772986412048, "rewards/code_complexity_reward/mean": 0.9079101085662842, "rewards/code_complexity_reward/std": 0.14708825945854187, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1053, "step_time": 59.08459727279842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 125.84765625, "completions/mean_terminated_length": 125.84765625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2311008358374238, "epoch": 0.6005698005698006, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06965584307909012, "kl": 0.2562855512369424, "learning_rate": 2.06696814400406e-06, "loss": 0.0012808431638404727, "num_tokens": 162114158.0, "reward": 2.360888719558716, "reward_std": 0.5027369856834412, "rewards/code_complexity_reward/mean": 0.9244140386581421, "rewards/code_complexity_reward/std": 0.09822120517492294, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1054, "step_time": 41.913012514822185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 483.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 126.650390625, "completions/mean_terminated_length": 126.650390625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.225700369104743, "epoch": 0.6011396011396012, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05441320687532425, "kl": 0.1420588359469548, "learning_rate": 2.062070166928861e-06, "loss": 0.0007103472598828375, "num_tokens": 162252739.0, "reward": 2.333301067352295, "reward_std": 0.48173025250434875, "rewards/code_complexity_reward/mean": 0.9217773079872131, "rewards/code_complexity_reward/std": 0.08702163398265839, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1055, "step_time": 51.03523966949433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 128.05859375, "completions/mean_terminated_length": 128.05859375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.23570960783399642, "epoch": 0.6017094017094017, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06491728872060776, "kl": 0.17226482648402452, "learning_rate": 2.0571739234176393e-06, "loss": 0.0008614612743258476, "num_tokens": 162385417.0, "reward": 2.349414110183716, "reward_std": 0.5105153322219849, "rewards/code_complexity_reward/mean": 0.9193359613418579, "rewards/code_complexity_reward/std": 0.10560211539268494, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1056, "step_time": 50.033340299502015 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 503.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 123.466796875, "completions/mean_terminated_length": 123.466796875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2310812110081315, "epoch": 0.6022792022792023, "frac_reward_zero_std": 0.5625, "grad_norm": 0.0567687563598156, "kl": 0.16117426194250584, "learning_rate": 2.0522794328523817e-06, "loss": 0.0008060112013481557, "num_tokens": 162517176.0, "reward": 2.3251953125, "reward_std": 0.5047208666801453, "rewards/code_complexity_reward/mean": 0.919238269329071, "rewards/code_complexity_reward/std": 0.10883661359548569, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1057, "step_time": 58.12640654202551 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 121.05078125, "completions/mean_terminated_length": 121.05078125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23273242986761034, "epoch": 0.6028490028490029, "frac_reward_zero_std": 0.59375, "grad_norm": 0.055921345949172974, "kl": 0.15660326974466443, "learning_rate": 2.047386714608142e-06, "loss": 0.0007830369868315756, "num_tokens": 162650634.0, "reward": 2.3688478469848633, "reward_std": 0.5060386657714844, "rewards/code_complexity_reward/mean": 0.923632800579071, "rewards/code_complexity_reward/std": 0.08785375952720642, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1058, "step_time": 39.94549986254424 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 275.0, "completions/mean_length": 124.77734375, "completions/mean_terminated_length": 124.01956939697266, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23357156733982265, "epoch": 0.6034188034188034, "frac_reward_zero_std": 0.5625, "grad_norm": 0.054618120193481445, "kl": 0.16051704494748265, "learning_rate": 2.0424957880529517e-06, "loss": 0.0008024891722016037, "num_tokens": 162784752.0, "reward": 2.3619630336761475, "reward_std": 0.5110971927642822, "rewards/code_complexity_reward/mean": 0.9276367425918579, "rewards/code_complexity_reward/std": 0.10408900678157806, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 1059, "step_time": 47.59349625837058 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 118.533203125, "completions/mean_terminated_length": 118.533203125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22068510041572154, "epoch": 0.603988603988604, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05920594185590744, "kl": 0.15853543381672353, "learning_rate": 2.0376066725477544e-06, "loss": 0.0007927544065751135, "num_tokens": 162912873.0, "reward": 2.3656740188598633, "reward_std": 0.5035050511360168, "rewards/code_complexity_reward/mean": 0.9270507097244263, "rewards/code_complexity_reward/std": 0.09649276733398438, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1060, "step_time": 37.53120976500213 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 124.849609375, "completions/mean_terminated_length": 124.849609375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23152055265381932, "epoch": 0.6045584045584046, "frac_reward_zero_std": 0.609375, "grad_norm": 0.05201860889792442, "kl": 0.17110345198307186, "learning_rate": 2.0327193874463217e-06, "loss": 0.0008557487744837999, "num_tokens": 163047004.0, "reward": 2.3768067359924316, "reward_std": 0.4988640546798706, "rewards/code_complexity_reward/mean": 0.9313477277755737, "rewards/code_complexity_reward/std": 0.085772305727005, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1061, "step_time": 35.40607319306582 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 125.029296875, "completions/mean_terminated_length": 125.029296875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23051019525155425, "epoch": 0.6051282051282051, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06436637043952942, "kl": 0.16005806112661958, "learning_rate": 2.027833952095182e-06, "loss": 0.000800244277343154, "num_tokens": 163182259.0, "reward": 2.3466796875, "reward_std": 0.5460575222969055, "rewards/code_complexity_reward/mean": 0.9126952886581421, "rewards/code_complexity_reward/std": 0.14485622942447662, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1062, "step_time": 50.276565230451524 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 424.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 122.93359375, "completions/mean_terminated_length": 122.93359375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23484130413271487, "epoch": 0.6056980056980057, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06584717333316803, "kl": 0.1555991981877014, "learning_rate": 2.0229503858335387e-06, "loss": 0.0007777765858918428, "num_tokens": 163315705.0, "reward": 2.2707033157348633, "reward_std": 0.45565444231033325, "rewards/code_complexity_reward/mean": 0.92431640625, "rewards/code_complexity_reward/std": 0.07763329893350601, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1063, "step_time": 43.61427304428071 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 296.0, "completions/max_terminated_length": 296.0, "completions/mean_length": 123.35546875, "completions/mean_terminated_length": 123.35546875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23306275624781847, "epoch": 0.6062678062678063, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05451205372810364, "kl": 0.16503737249877304, "learning_rate": 2.0180687079931994e-06, "loss": 0.0008251373656094074, "num_tokens": 163448263.0, "reward": 2.394726514816284, "reward_std": 0.48974964022636414, "rewards/code_complexity_reward/mean": 0.9324219226837158, "rewards/code_complexity_reward/std": 0.05143340304493904, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1064, "step_time": 35.20453601703048 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 120.9296875, "completions/mean_terminated_length": 120.9296875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23304089158773422, "epoch": 0.6068376068376068, "frac_reward_zero_std": 0.515625, "grad_norm": 0.08095379918813705, "kl": 0.18864110298454762, "learning_rate": 2.013188937898494e-06, "loss": 0.0009431577054783702, "num_tokens": 163579091.0, "reward": 2.3543944358825684, "reward_std": 0.48887351155281067, "rewards/code_complexity_reward/mean": 0.93115234375, "rewards/code_complexity_reward/std": 0.0764375627040863, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1065, "step_time": 44.763398157432675 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 388.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 124.1171875, "completions/mean_terminated_length": 124.1171875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.22948066401295364, "epoch": 0.6074074074074074, "frac_reward_zero_std": 0.625, "grad_norm": 0.049997180700302124, "kl": 0.164768994320184, "learning_rate": 2.0083110948662e-06, "loss": 0.0008235409040935338, "num_tokens": 163710239.0, "reward": 2.3531737327575684, "reward_std": 0.5009101033210754, "rewards/code_complexity_reward/mean": 0.92138671875, "rewards/code_complexity_reward/std": 0.08711864799261093, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1066, "step_time": 49.277649939991534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 121.873046875, "completions/mean_terminated_length": 121.873046875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22501980932429433, "epoch": 0.607977207977208, "frac_reward_zero_std": 0.5, "grad_norm": 0.0677524134516716, "kl": 0.16518831171561033, "learning_rate": 2.00343519820547e-06, "loss": 0.0008259828318841755, "num_tokens": 163842190.0, "reward": 2.3645997047424316, "reward_std": 0.5385010242462158, "rewards/code_complexity_reward/mean": 0.919140636920929, "rewards/code_complexity_reward/std": 0.13236092031002045, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1067, "step_time": 73.021689273417 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 351.0, "completions/max_terminated_length": 351.0, "completions/mean_length": 125.03125, "completions/mean_terminated_length": 125.03125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2294187555089593, "epoch": 0.6085470085470085, "frac_reward_zero_std": 0.5, "grad_norm": 0.05459730327129364, "kl": 0.15979930944740772, "learning_rate": 1.9985612672177468e-06, "loss": 0.0007993243634700775, "num_tokens": 163975606.0, "reward": 2.3932619094848633, "reward_std": 0.5052232146263123, "rewards/code_complexity_reward/mean": 0.91796875, "rewards/code_complexity_reward/std": 0.08617425709962845, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1068, "step_time": 38.460106018930674 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 310.0, "completions/max_terminated_length": 310.0, "completions/mean_length": 123.798828125, "completions/mean_terminated_length": 123.798828125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23766279080882668, "epoch": 0.6091168091168091, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05425569787621498, "kl": 0.15361657610628754, "learning_rate": 1.993689321196697e-06, "loss": 0.0007683208677917719, "num_tokens": 164106575.0, "reward": 2.355517864227295, "reward_std": 0.4942972660064697, "rewards/code_complexity_reward/mean": 0.9288085699081421, "rewards/code_complexity_reward/std": 0.09654068946838379, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1069, "step_time": 35.379033122211695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 126.87890625, "completions/mean_terminated_length": 126.87890625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23361365590244532, "epoch": 0.6096866096866097, "frac_reward_zero_std": 0.5, "grad_norm": 0.0553274042904377, "kl": 0.14678296202328056, "learning_rate": 1.9888193794281256e-06, "loss": 0.0007337449351325631, "num_tokens": 164239377.0, "reward": 2.3228516578674316, "reward_std": 0.5155937075614929, "rewards/code_complexity_reward/mean": 0.9152343273162842, "rewards/code_complexity_reward/std": 0.12544380128383636, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1070, "step_time": 38.4856641581282 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 140.673828125, "completions/mean_terminated_length": 137.0118408203125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23741275910288095, "epoch": 0.6102564102564103, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06370911002159119, "kl": 0.16883919178508222, "learning_rate": 1.983951461189907e-06, "loss": 0.0008445943240076303, "num_tokens": 164380090.0, "reward": 2.295459270477295, "reward_std": 0.5482628345489502, "rewards/code_complexity_reward/mean": 0.9027343988418579, "rewards/code_complexity_reward/std": 0.15697214007377625, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 1071, "step_time": 50.35715991538018 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 122.630859375, "completions/mean_terminated_length": 121.86888122558594, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23391154292039573, "epoch": 0.6108262108262108, "frac_reward_zero_std": 0.46875, "grad_norm": 0.060705509036779404, "kl": 0.15170027944259346, "learning_rate": 1.9790855857519026e-06, "loss": 0.0007585901767015457, "num_tokens": 164510469.0, "reward": 2.3426270484924316, "reward_std": 0.5024456977844238, "rewards/code_complexity_reward/mean": 0.9264647960662842, "rewards/code_complexity_reward/std": 0.10359394550323486, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1072, "step_time": 47.633519265800714 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 255.0, "completions/mean_length": 117.380859375, "completions/mean_terminated_length": 116.60861206054688, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.22152745770290494, "epoch": 0.6113960113960114, "frac_reward_zero_std": 0.59375, "grad_norm": 0.060066107660532, "kl": 0.1562617621384561, "learning_rate": 1.974221772375888e-06, "loss": 0.0007815341232344508, "num_tokens": 164636648.0, "reward": 2.4064455032348633, "reward_std": 0.5188284516334534, "rewards/code_complexity_reward/mean": 0.927734375, "rewards/code_complexity_reward/std": 0.09483915567398071, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 1073, "step_time": 55.463563565164804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 270.0, "completions/max_terminated_length": 270.0, "completions/mean_length": 122.482421875, "completions/mean_terminated_length": 122.482421875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.21919287368655205, "epoch": 0.611965811965812, "frac_reward_zero_std": 0.671875, "grad_norm": 0.04661429673433304, "kl": 0.15668728528544307, "learning_rate": 1.9693600403154783e-06, "loss": 0.0007834004354663193, "num_tokens": 164769079.0, "reward": 2.335742235183716, "reward_std": 0.4860800504684448, "rewards/code_complexity_reward/mean": 0.9261718988418579, "rewards/code_complexity_reward/std": 0.07850482314825058, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1074, "step_time": 33.40430398657918 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 383.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 129.775390625, "completions/mean_terminated_length": 129.775390625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.233283567475155, "epoch": 0.6125356125356125, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06627000123262405, "kl": 0.16556635382585227, "learning_rate": 1.964500408816046e-06, "loss": 0.0008278591558337212, "num_tokens": 164906140.0, "reward": 2.374267578125, "reward_std": 0.5017442107200623, "rewards/code_complexity_reward/mean": 0.9218749403953552, "rewards/code_complexity_reward/std": 0.09157247841358185, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1075, "step_time": 39.63899602834135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 132.576171875, "completions/mean_terminated_length": 131.8336639404297, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22747030528262258, "epoch": 0.6131054131054131, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06540544331073761, "kl": 0.1577114969259128, "learning_rate": 1.959642897114652e-06, "loss": 0.0007886771345511079, "num_tokens": 165044483.0, "reward": 2.3397459983825684, "reward_std": 0.5339454412460327, "rewards/code_complexity_reward/mean": 0.914746105670929, "rewards/code_complexity_reward/std": 0.13257326185703278, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 1076, "step_time": 48.75808573421091 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 368.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 123.138671875, "completions/mean_terminated_length": 123.138671875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23555528884753585, "epoch": 0.6136752136752137, "frac_reward_zero_std": 0.5, "grad_norm": 0.059009917080402374, "kl": 0.15195504343137145, "learning_rate": 1.954787524439963e-06, "loss": 0.0007596609648317099, "num_tokens": 165177338.0, "reward": 2.306884765625, "reward_std": 0.49538251757621765, "rewards/code_complexity_reward/mean": 0.919238269329071, "rewards/code_complexity_reward/std": 0.11150108277797699, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1077, "step_time": 38.72789135668427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 387.0, "completions/max_terminated_length": 387.0, "completions/mean_length": 127.390625, "completions/mean_terminated_length": 127.390625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2332878508605063, "epoch": 0.6142450142450142, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05630039796233177, "kl": 0.16905807610601187, "learning_rate": 1.94993431001218e-06, "loss": 0.0008454842027276754, "num_tokens": 165310578.0, "reward": 2.374804735183716, "reward_std": 0.4906962215900421, "rewards/code_complexity_reward/mean": 0.9349609613418579, "rewards/code_complexity_reward/std": 0.0647016242146492, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1078, "step_time": 47.139903298579156 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 129.736328125, "completions/mean_terminated_length": 128.98825073242188, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23213910521008074, "epoch": 0.6148148148148148, "frac_reward_zero_std": 0.5625, "grad_norm": 0.054171670228242874, "kl": 0.1705298739252612, "learning_rate": 1.945083273042958e-06, "loss": 0.0008525596931576729, "num_tokens": 165446571.0, "reward": 2.2750978469848633, "reward_std": 0.4444950222969055, "rewards/code_complexity_reward/mean": 0.931640625, "rewards/code_complexity_reward/std": 0.06712464243173599, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1079, "step_time": 50.56132238917053 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 371.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 124.73046875, "completions/mean_terminated_length": 124.73046875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22048108256421983, "epoch": 0.6153846153846154, "frac_reward_zero_std": 0.40625, "grad_norm": 0.08410967886447906, "kl": 0.1511832777177915, "learning_rate": 1.9402344327353373e-06, "loss": 0.0007559378864243627, "num_tokens": 165582113.0, "reward": 2.4267091751098633, "reward_std": 0.5360495448112488, "rewards/code_complexity_reward/mean": 0.92578125, "rewards/code_complexity_reward/std": 0.11120833456516266, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1080, "step_time": 48.28594478126615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 503.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 132.177734375, "completions/mean_terminated_length": 132.177734375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22674452583305538, "epoch": 0.6159544159544159, "frac_reward_zero_std": 0.5, "grad_norm": 0.05567837506532669, "kl": 0.14547171886079013, "learning_rate": 1.9353878082836573e-06, "loss": 0.0007272647926583886, "num_tokens": 165716876.0, "reward": 2.3987793922424316, "reward_std": 0.5406153202056885, "rewards/code_complexity_reward/mean": 0.9183593988418579, "rewards/code_complexity_reward/std": 0.12007353454828262, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1081, "step_time": 56.913857384584844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 460.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 127.79296875, "completions/mean_terminated_length": 127.79296875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2410375566687435, "epoch": 0.6165242165242165, "frac_reward_zero_std": 0.484375, "grad_norm": 0.059405479580163956, "kl": 0.16034692188259214, "learning_rate": 1.93054341887349e-06, "loss": 0.0008018993539735675, "num_tokens": 165852002.0, "reward": 2.2201662063598633, "reward_std": 0.47016918659210205, "rewards/code_complexity_reward/mean": 0.9168945550918579, "rewards/code_complexity_reward/std": 0.13279592990875244, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 1082, "step_time": 44.19519677571952 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 427.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 124.169921875, "completions/mean_terminated_length": 124.169921875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22904472309164703, "epoch": 0.6170940170940171, "frac_reward_zero_std": 0.53125, "grad_norm": 0.06000248342752457, "kl": 0.16520700161345303, "learning_rate": 1.9257012836815563e-06, "loss": 0.0008261075709015131, "num_tokens": 165984745.0, "reward": 2.4175782203674316, "reward_std": 0.5231491327285767, "rewards/code_complexity_reward/mean": 0.9307616949081421, "rewards/code_complexity_reward/std": 0.09710086137056351, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1083, "step_time": 49.402635912410915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 125.52734375, "completions/mean_terminated_length": 125.52734375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23031345778144896, "epoch": 0.6176638176638176, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06400005519390106, "kl": 0.14049424487166107, "learning_rate": 1.9208614218756565e-06, "loss": 0.0007026228122413158, "num_tokens": 166115623.0, "reward": 2.2837891578674316, "reward_std": 0.48779720067977905, "rewards/code_complexity_reward/mean": 0.916210949420929, "rewards/code_complexity_reward/std": 0.11182920634746552, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1084, "step_time": 57.93612010218203 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 396.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 126.029296875, "completions/mean_terminated_length": 126.029296875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23228283040225506, "epoch": 0.6182336182336182, "frac_reward_zero_std": 0.65625, "grad_norm": 0.05390700697898865, "kl": 0.15563327656127512, "learning_rate": 1.9160238526145915e-06, "loss": 0.0007783013279549778, "num_tokens": 166249582.0, "reward": 2.2833008766174316, "reward_std": 0.44267183542251587, "rewards/code_complexity_reward/mean": 0.9284179210662842, "rewards/code_complexity_reward/std": 0.05166538804769516, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1085, "step_time": 41.238648822531104 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 125.724609375, "completions/mean_terminated_length": 125.724609375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23603819659911096, "epoch": 0.6188034188034188, "frac_reward_zero_std": 0.5, "grad_norm": 0.05965705215930939, "kl": 0.1521200481802225, "learning_rate": 1.911188595048084e-06, "loss": 0.0007608777959831059, "num_tokens": 166381273.0, "reward": 2.392578125, "reward_std": 0.4942290186882019, "rewards/code_complexity_reward/mean": 0.93212890625, "rewards/code_complexity_reward/std": 0.0651504248380661, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1086, "step_time": 37.73074329085648 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 122.421875, "completions/mean_terminated_length": 122.421875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23204087745398283, "epoch": 0.6193732193732193, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06695234775543213, "kl": 0.15409394109155983, "learning_rate": 1.90635566831671e-06, "loss": 0.000770151149481535, "num_tokens": 166511513.0, "reward": 2.375, "reward_std": 0.5088958740234375, "rewards/code_complexity_reward/mean": 0.9231445789337158, "rewards/code_complexity_reward/std": 0.09850385040044785, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1087, "step_time": 70.70881448220462 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 122.708984375, "completions/mean_terminated_length": 122.708984375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23509692028164864, "epoch": 0.6199430199430199, "frac_reward_zero_std": 0.5, "grad_norm": 0.06134476140141487, "kl": 0.1495391217758879, "learning_rate": 1.9015250915518148e-06, "loss": 0.0007475800812244415, "num_tokens": 166641724.0, "reward": 2.3158693313598633, "reward_std": 0.5221496820449829, "rewards/code_complexity_reward/mean": 0.91943359375, "rewards/code_complexity_reward/std": 0.13241055607795715, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1088, "step_time": 42.72378965374082 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 127.326171875, "completions/mean_terminated_length": 126.5733871459961, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23578065307810903, "epoch": 0.6205128205128205, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05023961514234543, "kl": 0.1637926724506542, "learning_rate": 1.8966968838754452e-06, "loss": 0.0008190562948584557, "num_tokens": 166774451.0, "reward": 2.3189942836761475, "reward_std": 0.5140746831893921, "rewards/code_complexity_reward/mean": 0.91455078125, "rewards/code_complexity_reward/std": 0.11844637244939804, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1089, "step_time": 55.91841675341129 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 381.0, "completions/max_terminated_length": 381.0, "completions/mean_length": 132.283203125, "completions/mean_terminated_length": 132.283203125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23483983147889376, "epoch": 0.6210826210826211, "frac_reward_zero_std": 0.5, "grad_norm": 0.06434827297925949, "kl": 0.15435624681413174, "learning_rate": 1.8918710644002662e-06, "loss": 0.0007720771245658398, "num_tokens": 166913108.0, "reward": 2.2396974563598633, "reward_std": 0.4995228350162506, "rewards/code_complexity_reward/mean": 0.9075194597244263, "rewards/code_complexity_reward/std": 0.14319905638694763, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 1090, "step_time": 40.97378041408956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 128.701171875, "completions/mean_terminated_length": 128.701171875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2291452488861978, "epoch": 0.6216524216524216, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05886059254407883, "kl": 0.1663315922487527, "learning_rate": 1.8870476522294912e-06, "loss": 0.0008317246101796627, "num_tokens": 167049235.0, "reward": 2.333740472793579, "reward_std": 0.489238977432251, "rewards/code_complexity_reward/mean": 0.9216797351837158, "rewards/code_complexity_reward/std": 0.08868858963251114, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1091, "step_time": 37.23303921055049 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 128.51171875, "completions/mean_terminated_length": 128.51171875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23538332618772984, "epoch": 0.6222222222222222, "frac_reward_zero_std": 0.625, "grad_norm": 0.05353008583188057, "kl": 0.1537520915735513, "learning_rate": 1.8822266664568029e-06, "loss": 0.0007686157478019595, "num_tokens": 167183865.0, "reward": 2.3709473609924316, "reward_std": 0.49434858560562134, "rewards/code_complexity_reward/mean": 0.9305664300918579, "rewards/code_complexity_reward/std": 0.0812588706612587, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1092, "step_time": 45.53367271181196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 127.939453125, "completions/mean_terminated_length": 127.939453125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.24022657447494566, "epoch": 0.6227920227920228, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06376616656780243, "kl": 0.16147635970264673, "learning_rate": 1.8774081261662798e-06, "loss": 0.0008074066136032343, "num_tokens": 167322626.0, "reward": 2.2540040016174316, "reward_std": 0.4560113251209259, "rewards/code_complexity_reward/mean": 0.9271484017372131, "rewards/code_complexity_reward/std": 0.09603323042392731, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1093, "step_time": 38.289379618130624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 498.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 124.263671875, "completions/mean_terminated_length": 124.263671875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22550524375401437, "epoch": 0.6233618233618233, "frac_reward_zero_std": 0.53125, "grad_norm": 0.06348437070846558, "kl": 0.1635620892047882, "learning_rate": 1.872592050432322e-06, "loss": 0.0008178529678843915, "num_tokens": 167458585.0, "reward": 2.357421875, "reward_std": 0.48376134037971497, "rewards/code_complexity_reward/mean": 0.9253906011581421, "rewards/code_complexity_reward/std": 0.06657576560974121, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1094, "step_time": 66.19488540943712 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 123.439453125, "completions/mean_terminated_length": 123.439453125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23065582336857915, "epoch": 0.6239316239316239, "frac_reward_zero_std": 0.59375, "grad_norm": 0.0534420907497406, "kl": 0.16643035120796412, "learning_rate": 1.8677784583195685e-06, "loss": 0.0008322163484990597, "num_tokens": 167589386.0, "reward": 2.3549318313598633, "reward_std": 0.47675037384033203, "rewards/code_complexity_reward/mean": 0.932910144329071, "rewards/code_complexity_reward/std": 0.06836020946502686, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1095, "step_time": 37.82557685114443 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 501.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 132.357421875, "completions/mean_terminated_length": 132.357421875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23160220705904067, "epoch": 0.6245014245014245, "frac_reward_zero_std": 0.515625, "grad_norm": 0.058799147605895996, "kl": 0.1534165672492236, "learning_rate": 1.8629673688828309e-06, "loss": 0.0007673267973586917, "num_tokens": 167725889.0, "reward": 2.238818645477295, "reward_std": 0.48307281732559204, "rewards/code_complexity_reward/mean": 0.9156249761581421, "rewards/code_complexity_reward/std": 0.1330435574054718, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1096, "step_time": 54.95530574861914 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 417.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 131.595703125, "completions/mean_terminated_length": 131.595703125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23708151560276747, "epoch": 0.625071225071225, "frac_reward_zero_std": 0.390625, "grad_norm": 0.07575089484453201, "kl": 0.15613219735678285, "learning_rate": 1.858158801167012e-06, "loss": 0.000781145179644227, "num_tokens": 167859186.0, "reward": 2.464648723602295, "reward_std": 0.503957211971283, "rewards/code_complexity_reward/mean": 0.9339843988418579, "rewards/code_complexity_reward/std": 0.05558216571807861, "rewards/code_execution_reward/mean": 0.431640625, "rewards/code_execution_reward/std": 0.4957893490791321, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1097, "step_time": 42.178602151572704 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 307.0, "completions/max_terminated_length": 307.0, "completions/mean_length": 119.04296875, "completions/mean_terminated_length": 119.04296875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.21914896345697343, "epoch": 0.6256410256410256, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05861486494541168, "kl": 0.14798448025248945, "learning_rate": 1.8533527742070336e-06, "loss": 0.0007398823508992791, "num_tokens": 167986728.0, "reward": 2.4842774868011475, "reward_std": 0.5369259715080261, "rewards/code_complexity_reward/mean": 0.92333984375, "rewards/code_complexity_reward/std": 0.10481172055006027, "rewards/code_execution_reward/mean": 0.466796875, "rewards/code_execution_reward/std": 0.4993842542171478, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1098, "step_time": 48.3789958935231 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 126.248046875, "completions/mean_terminated_length": 126.248046875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21723977010697126, "epoch": 0.6262108262108262, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05262073129415512, "kl": 0.14780974166933447, "learning_rate": 1.8485493070277575e-06, "loss": 0.0007391432300209999, "num_tokens": 168121015.0, "reward": 2.3278322219848633, "reward_std": 0.5008425116539001, "rewards/code_complexity_reward/mean": 0.925097644329071, "rewards/code_complexity_reward/std": 0.10547622293233871, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1099, "step_time": 46.15494341030717 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 444.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 121.8671875, "completions/mean_terminated_length": 121.8671875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23526632017455995, "epoch": 0.6267806267806267, "frac_reward_zero_std": 0.40625, "grad_norm": 0.07483572512865067, "kl": 0.17696747882291675, "learning_rate": 1.8437484186439158e-06, "loss": 0.0008847998105920851, "num_tokens": 168253171.0, "reward": 2.3529295921325684, "reward_std": 0.5195975303649902, "rewards/code_complexity_reward/mean": 0.916015625, "rewards/code_complexity_reward/std": 0.12158180773258209, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1100, "step_time": 63.28738943487406 }, { "epoch": 0.6267806267806267, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 180.26, "eval_completions/max_terminated_length": 180.26, "eval_completions/mean_length": 127.5425, "eval_completions/mean_terminated_length": 127.5425, "eval_completions/min_length": 92.18, "eval_completions/min_terminated_length": 92.18, "eval_entropy": 0.22937969893217086, "eval_frac_reward_zero_std": 0.46, "eval_kl": 0.14933144398033618, "eval_loss": 0.0007468069670721889, "eval_num_tokens": 168253171.0, "eval_reward": 2.326750124692917, "eval_reward_std": 0.2145951318182051, "eval_rewards/code_complexity_reward/mean": 0.9229999858140946, "eval_rewards/code_complexity_reward/std": 0.030586474221199752, "eval_rewards/code_execution_reward/mean": 0.30875, "eval_rewards/code_execution_reward/std": 0.1772594505548477, "eval_rewards/code_syntax_reward/mean": 0.495, "eval_rewards/code_syntax_reward/std": 0.01292115181684494, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.5, "eval_rewards/xmlcount_reward_func/std": 0.0, "eval_runtime": 841.9095, "eval_samples_per_second": 0.119, "eval_steps_per_second": 0.015, "step": 1100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 126.1171875, "completions/mean_terminated_length": 126.1171875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22268082574009895, "epoch": 0.6273504273504273, "frac_reward_zero_std": 0.671875, "grad_norm": 0.046024300158023834, "kl": 0.15436142461840063, "learning_rate": 1.8389501280600288e-06, "loss": 0.0007717060507275164, "num_tokens": 168386199.0, "reward": 2.4073243141174316, "reward_std": 0.5008383989334106, "rewards/code_complexity_reward/mean": 0.9300781488418579, "rewards/code_complexity_reward/std": 0.07638860493898392, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1101, "step_time": 41.91806854587048 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 131.677734375, "completions/mean_terminated_length": 131.677734375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.22998545691370964, "epoch": 0.627920227920228, "frac_reward_zero_std": 0.5, "grad_norm": 0.05542615056037903, "kl": 0.15760075929574668, "learning_rate": 1.8341544542703368e-06, "loss": 0.0007882026256993413, "num_tokens": 168524002.0, "reward": 2.344287395477295, "reward_std": 0.5090163946151733, "rewards/code_complexity_reward/mean": 0.9253906011581421, "rewards/code_complexity_reward/std": 0.11488888412714005, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1102, "step_time": 38.80940331052989 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 305.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 124.48828125, "completions/mean_terminated_length": 124.48828125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24353987909853458, "epoch": 0.6284900284900284, "frac_reward_zero_std": 0.53125, "grad_norm": 0.06017749384045601, "kl": 0.1815555189969018, "learning_rate": 1.8293614162587182e-06, "loss": 0.000907841429580003, "num_tokens": 168658060.0, "reward": 2.2354493141174316, "reward_std": 0.43157902359962463, "rewards/code_complexity_reward/mean": 0.9303710460662842, "rewards/code_complexity_reward/std": 0.08612362295389175, "rewards/code_execution_reward/mean": 0.208984375, "rewards/code_execution_reward/std": 0.40698084235191345, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1103, "step_time": 44.777443065308034 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 125.83984375, "completions/mean_terminated_length": 125.83984375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23276164033450186, "epoch": 0.629059829059829, "frac_reward_zero_std": 0.5625, "grad_norm": 0.06613392382860184, "kl": 0.1653401852818206, "learning_rate": 1.824571032998619e-06, "loss": 0.0008266683435067534, "num_tokens": 168789858.0, "reward": 2.354980707168579, "reward_std": 0.5020599365234375, "rewards/code_complexity_reward/mean": 0.9239257574081421, "rewards/code_complexity_reward/std": 0.09569407254457474, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1104, "step_time": 33.96164138428867 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 270.0, "completions/max_terminated_length": 270.0, "completions/mean_length": 122.720703125, "completions/mean_terminated_length": 122.720703125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23914419603534043, "epoch": 0.6296296296296297, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06515676528215408, "kl": 0.15588181652128696, "learning_rate": 1.8197833234529776e-06, "loss": 0.000779302092269063, "num_tokens": 168919171.0, "reward": 2.345752000808716, "reward_std": 0.5095254778862, "rewards/code_complexity_reward/mean": 0.9229492545127869, "rewards/code_complexity_reward/std": 0.11298170685768127, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1105, "step_time": 43.588459918275476 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 323.0, "completions/max_terminated_length": 323.0, "completions/mean_length": 122.01171875, "completions/mean_terminated_length": 122.01171875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23298831935971975, "epoch": 0.6301994301994301, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05379534512758255, "kl": 0.15278156800195575, "learning_rate": 1.8149983065741456e-06, "loss": 0.0007637381786480546, "num_tokens": 169048777.0, "reward": 2.4432618618011475, "reward_std": 0.49527958035469055, "rewards/code_complexity_reward/mean": 0.93603515625, "rewards/code_complexity_reward/std": 0.049114495515823364, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1106, "step_time": 36.338553791865706 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 125.70703125, "completions/mean_terminated_length": 125.70703125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23423234815709293, "epoch": 0.6307692307692307, "frac_reward_zero_std": 0.5, "grad_norm": 0.05739621818065643, "kl": 0.17007212387397885, "learning_rate": 1.810216001303818e-06, "loss": 0.0008503167773596942, "num_tokens": 169181667.0, "reward": 2.3916015625, "reward_std": 0.48661917448043823, "rewards/code_complexity_reward/mean": 0.9312499761581421, "rewards/code_complexity_reward/std": 0.05177854374051094, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1107, "step_time": 42.78470767196268 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 268.0, "completions/max_terminated_length": 268.0, "completions/mean_length": 120.095703125, "completions/mean_terminated_length": 120.095703125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22670953604392707, "epoch": 0.6313390313390314, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05669459328055382, "kl": 0.15977327164728194, "learning_rate": 1.8054364265729535e-06, "loss": 0.0007988520665094256, "num_tokens": 169310876.0, "reward": 2.4689455032348633, "reward_std": 0.5289028286933899, "rewards/code_complexity_reward/mean": 0.9265624284744263, "rewards/code_complexity_reward/std": 0.09720909595489502, "rewards/code_execution_reward/mean": 0.447265625, "rewards/code_execution_reward/std": 0.4976975917816162, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1108, "step_time": 33.44387407694012 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 124.525390625, "completions/mean_terminated_length": 124.525390625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23532329918816686, "epoch": 0.631908831908832, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06824333220720291, "kl": 0.1461621664930135, "learning_rate": 1.8006596013017052e-06, "loss": 0.0007309811189770699, "num_tokens": 169441737.0, "reward": 2.357715129852295, "reward_std": 0.4970107078552246, "rewards/code_complexity_reward/mean": 0.9276367425918579, "rewards/code_complexity_reward/std": 0.08793611079454422, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1109, "step_time": 53.2351435944438 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 126.91015625, "completions/mean_terminated_length": 126.91015625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23575312714092433, "epoch": 0.6324786324786325, "frac_reward_zero_std": 0.484375, "grad_norm": 0.0634990930557251, "kl": 0.15287336346227676, "learning_rate": 1.795885544399338e-06, "loss": 0.0007641760166734457, "num_tokens": 169575475.0, "reward": 2.3377928733825684, "reward_std": 0.5079335570335388, "rewards/code_complexity_reward/mean": 0.91943359375, "rewards/code_complexity_reward/std": 0.10528253763914108, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1110, "step_time": 57.66723224893212 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 397.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 131.26171875, "completions/mean_terminated_length": 131.26171875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23523568850941956, "epoch": 0.6330484330484331, "frac_reward_zero_std": 0.546875, "grad_norm": 0.055397409945726395, "kl": 0.15322062640916556, "learning_rate": 1.7911142747641624e-06, "loss": 0.0007661134004592896, "num_tokens": 169711969.0, "reward": 2.339111566543579, "reward_std": 0.5260157585144043, "rewards/code_complexity_reward/mean": 0.9141601324081421, "rewards/code_complexity_reward/std": 0.12590058147907257, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1111, "step_time": 81.70725801680237 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 132.25390625, "completions/mean_terminated_length": 132.25390625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2306113934610039, "epoch": 0.6336182336182337, "frac_reward_zero_std": 0.390625, "grad_norm": 0.0695333480834961, "kl": 0.14506880496628582, "learning_rate": 1.786345811283451e-06, "loss": 0.0007254750235006213, "num_tokens": 169847275.0, "reward": 2.3214845657348633, "reward_std": 0.5129987597465515, "rewards/code_complexity_reward/mean": 0.920703113079071, "rewards/code_complexity_reward/std": 0.12335450202226639, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1112, "step_time": 50.74257034063339 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 493.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 128.859375, "completions/mean_terminated_length": 128.859375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2373156005050987, "epoch": 0.6341880341880342, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06320343166589737, "kl": 0.15850104903802276, "learning_rate": 1.7815801728333715e-06, "loss": 0.0007925385143607855, "num_tokens": 169980707.0, "reward": 2.3454103469848633, "reward_std": 0.50692218542099, "rewards/code_complexity_reward/mean": 0.9231445789337158, "rewards/code_complexity_reward/std": 0.10946621000766754, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1113, "step_time": 46.747329438105226 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 285.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 124.951171875, "completions/mean_terminated_length": 124.951171875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.22718887450173497, "epoch": 0.6347578347578348, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06318408250808716, "kl": 0.1529749531764537, "learning_rate": 1.7768173782789091e-06, "loss": 0.000764491967856884, "num_tokens": 170113450.0, "reward": 2.2623047828674316, "reward_std": 0.49417844414711, "rewards/code_complexity_reward/mean": 0.920117199420929, "rewards/code_complexity_reward/std": 0.13206765055656433, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1114, "step_time": 33.92584057338536 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 126.419921875, "completions/mean_terminated_length": 126.419921875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2343603279441595, "epoch": 0.6353276353276354, "frac_reward_zero_std": 0.34375, "grad_norm": 0.07176797091960907, "kl": 0.15719129366334528, "learning_rate": 1.7720574464737878e-06, "loss": 0.0007861237390898168, "num_tokens": 170243649.0, "reward": 2.3580565452575684, "reward_std": 0.5029265880584717, "rewards/code_complexity_reward/mean": 0.9235351085662842, "rewards/code_complexity_reward/std": 0.09573999792337418, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1115, "step_time": 37.38386777136475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 414.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 129.7890625, "completions/mean_terminated_length": 129.7890625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23521346668712795, "epoch": 0.6358974358974359, "frac_reward_zero_std": 0.484375, "grad_norm": 0.07867199927568436, "kl": 0.1600161138921976, "learning_rate": 1.7673003962604027e-06, "loss": 0.0008000730304047465, "num_tokens": 170377261.0, "reward": 2.3634276390075684, "reward_std": 0.5189855694770813, "rewards/code_complexity_reward/mean": 0.919140636920929, "rewards/code_complexity_reward/std": 0.11345336586236954, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1116, "step_time": 55.17984122876078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 130.44921875, "completions/mean_terminated_length": 128.95294189453125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23629102855920792, "epoch": 0.6364672364672365, "frac_reward_zero_std": 0.546875, "grad_norm": 0.06346779316663742, "kl": 0.1688961099134758, "learning_rate": 1.7625462464697385e-06, "loss": 0.0008446135907433927, "num_tokens": 170513523.0, "reward": 2.3058106899261475, "reward_std": 0.49338498711586, "rewards/code_complexity_reward/mean": 0.91845703125, "rewards/code_complexity_reward/std": 0.10090666264295578, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1117, "step_time": 48.65066136978567 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 423.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 126.982421875, "completions/mean_terminated_length": 126.982421875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23499195487238467, "epoch": 0.6370370370370371, "frac_reward_zero_std": 0.546875, "grad_norm": 0.058293960988521576, "kl": 0.1609489320544526, "learning_rate": 1.7577950159213029e-06, "loss": 0.0008048773161135614, "num_tokens": 170650170.0, "reward": 2.2984375953674316, "reward_std": 0.47045019268989563, "rewards/code_complexity_reward/mean": 0.9189453125, "rewards/code_complexity_reward/std": 0.07796619832515717, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 1118, "step_time": 42.650728864595294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 319.0, "completions/max_terminated_length": 319.0, "completions/mean_length": 127.205078125, "completions/mean_terminated_length": 127.205078125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23315223678946495, "epoch": 0.6376068376068376, "frac_reward_zero_std": 0.453125, "grad_norm": 0.11191292852163315, "kl": 0.16593821649439633, "learning_rate": 1.7530467234230432e-06, "loss": 0.0008299194741994143, "num_tokens": 170784011.0, "reward": 2.3701171875, "reward_std": 0.5164851546287537, "rewards/code_complexity_reward/mean": 0.9205077886581421, "rewards/code_complexity_reward/std": 0.1077674850821495, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1119, "step_time": 46.01038230676204 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 130.40234375, "completions/mean_terminated_length": 129.65557861328125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23464403161779046, "epoch": 0.6381766381766382, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06699872016906738, "kl": 0.15466137242037803, "learning_rate": 1.7483013877712802e-06, "loss": 0.0007730770157650113, "num_tokens": 170918441.0, "reward": 2.339404582977295, "reward_std": 0.5166705250740051, "rewards/code_complexity_reward/mean": 0.9183593988418579, "rewards/code_complexity_reward/std": 0.11537755280733109, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1120, "step_time": 48.07642398774624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 126.494140625, "completions/mean_terminated_length": 125.7397232055664, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22416977165266871, "epoch": 0.6387464387464388, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05091512203216553, "kl": 0.1581331135239452, "learning_rate": 1.7435590277506258e-06, "loss": 0.0007905553793534636, "num_tokens": 171054342.0, "reward": 2.3456056118011475, "reward_std": 0.516818642616272, "rewards/code_complexity_reward/mean": 0.9210937023162842, "rewards/code_complexity_reward/std": 0.11166930943727493, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1121, "step_time": 49.19823390431702 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 133.443359375, "completions/mean_terminated_length": 132.70254516601562, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24542712699621916, "epoch": 0.6393162393162393, "frac_reward_zero_std": 0.375, "grad_norm": 0.05966249108314514, "kl": 0.16086973762139678, "learning_rate": 1.7388196621339175e-06, "loss": 0.0008044888381846249, "num_tokens": 171190401.0, "reward": 2.2719240188598633, "reward_std": 0.498693585395813, "rewards/code_complexity_reward/mean": 0.9093749523162842, "rewards/code_complexity_reward/std": 0.1233898252248764, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1122, "step_time": 60.860070147551596 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 489.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 126.060546875, "completions/mean_terminated_length": 126.060546875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22454542038030922, "epoch": 0.6398860398860399, "frac_reward_zero_std": 0.5, "grad_norm": 0.06407929211854935, "kl": 0.16799655789509416, "learning_rate": 1.7340833096821357e-06, "loss": 0.0008398049976676702, "num_tokens": 171321984.0, "reward": 2.3412110805511475, "reward_std": 0.49787306785583496, "rewards/code_complexity_reward/mean": 0.927734375, "rewards/code_complexity_reward/std": 0.09723347425460815, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1123, "step_time": 47.184051715768874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 368.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 123.720703125, "completions/mean_terminated_length": 123.720703125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22445071185939014, "epoch": 0.6404558404558405, "frac_reward_zero_std": 0.59375, "grad_norm": 0.052024323493242264, "kl": 0.15314526436850429, "learning_rate": 1.7293499891443332e-06, "loss": 0.0007659035618416965, "num_tokens": 171456873.0, "reward": 2.410644769668579, "reward_std": 0.5004879832267761, "rewards/code_complexity_reward/mean": 0.9317382574081421, "rewards/code_complexity_reward/std": 0.06884215772151947, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1124, "step_time": 43.48825963772833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 414.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 125.287109375, "completions/mean_terminated_length": 125.287109375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.237702505197376, "epoch": 0.6410256410256411, "frac_reward_zero_std": 0.65625, "grad_norm": 0.05231228843331337, "kl": 0.15904434060212225, "learning_rate": 1.7246197192575637e-06, "loss": 0.0007954140892252326, "num_tokens": 171588140.0, "reward": 2.3607420921325684, "reward_std": 0.49390795826911926, "rewards/code_complexity_reward/mean": 0.92578125, "rewards/code_complexity_reward/std": 0.08387260138988495, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1125, "step_time": 41.37635988462716 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 359.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 132.2265625, "completions/mean_terminated_length": 132.2265625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23307264083996415, "epoch": 0.6415954415954416, "frac_reward_zero_std": 0.5, "grad_norm": 0.0736512839794159, "kl": 0.13565395586192608, "learning_rate": 1.719892518746801e-06, "loss": 0.0006783237913623452, "num_tokens": 171725280.0, "reward": 2.319384813308716, "reward_std": 0.49090200662612915, "rewards/code_complexity_reward/mean": 0.9219726324081421, "rewards/code_complexity_reward/std": 0.0993649810552597, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1126, "step_time": 44.44489096943289 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 352.0, "completions/max_terminated_length": 352.0, "completions/mean_length": 127.751953125, "completions/mean_terminated_length": 127.751953125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2397334212437272, "epoch": 0.6421652421652422, "frac_reward_zero_std": 0.671875, "grad_norm": 0.05119362473487854, "kl": 0.1592496520606801, "learning_rate": 1.7151684063248726e-06, "loss": 0.0007963357493281364, "num_tokens": 171860809.0, "reward": 2.3663086891174316, "reward_std": 0.49593767523765564, "rewards/code_complexity_reward/mean": 0.9263671636581421, "rewards/code_complexity_reward/std": 0.07924599200487137, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1127, "step_time": 45.55047440622002 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 447.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 128.849609375, "completions/mean_terminated_length": 128.849609375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.23312145075760782, "epoch": 0.6427350427350428, "frac_reward_zero_std": 0.53125, "grad_norm": 0.07128462940454483, "kl": 0.15906690002884716, "learning_rate": 1.7104474006923776e-06, "loss": 0.0007951159495860338, "num_tokens": 171994596.0, "reward": 2.3561525344848633, "reward_std": 0.5055816173553467, "rewards/code_complexity_reward/mean": 0.925097644329071, "rewards/code_complexity_reward/std": 0.0978236272931099, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1128, "step_time": 45.38516684342176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 453.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 132.037109375, "completions/mean_terminated_length": 132.037109375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22767718764953315, "epoch": 0.6433048433048433, "frac_reward_zero_std": 0.53125, "grad_norm": 0.054150912910699844, "kl": 0.16538754536304623, "learning_rate": 1.7057295205376205e-06, "loss": 0.0008270774269476533, "num_tokens": 172130887.0, "reward": 2.38818359375, "reward_std": 0.5190613865852356, "rewards/code_complexity_reward/mean": 0.9180663824081421, "rewards/code_complexity_reward/std": 0.10194186121225357, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1129, "step_time": 44.28145437501371 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 454.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 135.91796875, "completions/mean_terminated_length": 135.91796875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.23888148344121873, "epoch": 0.6438746438746439, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06113174930214882, "kl": 0.16157261689659208, "learning_rate": 1.7010147845365292e-06, "loss": 0.0008079442195594311, "num_tokens": 172269861.0, "reward": 2.255908489227295, "reward_std": 0.4851418137550354, "rewards/code_complexity_reward/mean": 0.9170898199081421, "rewards/code_complexity_reward/std": 0.12762288749217987, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1130, "step_time": 62.26621806062758 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 404.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 129.380859375, "completions/mean_terminated_length": 129.380859375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23286803509108722, "epoch": 0.6444444444444445, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06336740404367447, "kl": 0.14925303356721997, "learning_rate": 1.6963032113525907e-06, "loss": 0.0007461574859917164, "num_tokens": 172404488.0, "reward": 2.3109376430511475, "reward_std": 0.49509087204933167, "rewards/code_complexity_reward/mean": 0.92236328125, "rewards/code_complexity_reward/std": 0.10432374477386475, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 1131, "step_time": 41.526721719652414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 282.0, "completions/max_terminated_length": 282.0, "completions/mean_length": 129.796875, "completions/mean_terminated_length": 129.796875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23231881647370756, "epoch": 0.645014245014245, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06514489650726318, "kl": 0.14868002710863948, "learning_rate": 1.6915948196367672e-06, "loss": 0.0007437822059728205, "num_tokens": 172541232.0, "reward": 2.333544969558716, "reward_std": 0.5041646957397461, "rewards/code_complexity_reward/mean": 0.9273437857627869, "rewards/code_complexity_reward/std": 0.10409603267908096, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1132, "step_time": 36.31375841423869 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 335.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 123.728515625, "completions/mean_terminated_length": 123.728515625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23230397189036012, "epoch": 0.6455840455840456, "frac_reward_zero_std": 0.53125, "grad_norm": 0.055239491164684296, "kl": 0.1602072613313794, "learning_rate": 1.6868896280274288e-06, "loss": 0.0008011074969545007, "num_tokens": 172672933.0, "reward": 2.2956056594848633, "reward_std": 0.500611424446106, "rewards/code_complexity_reward/mean": 0.9172850847244263, "rewards/code_complexity_reward/std": 0.11450161039829254, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1133, "step_time": 36.728746350854635 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 122.62109375, "completions/mean_terminated_length": 122.62109375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.22388331266120076, "epoch": 0.6461538461538462, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05878415331244469, "kl": 0.14872355107218027, "learning_rate": 1.6821876551502786e-06, "loss": 0.0007436329033225775, "num_tokens": 172802715.0, "reward": 2.3819828033447266, "reward_std": 0.5209845900535583, "rewards/code_complexity_reward/mean": 0.9247070550918579, "rewards/code_complexity_reward/std": 0.11208708584308624, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1134, "step_time": 41.74569829273969 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 400.0, "completions/max_terminated_length": 400.0, "completions/mean_length": 129.955078125, "completions/mean_terminated_length": 129.955078125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23307408136315644, "epoch": 0.6467236467236467, "frac_reward_zero_std": 0.578125, "grad_norm": 0.051887623965740204, "kl": 0.1543945271987468, "learning_rate": 1.677488919618277e-06, "loss": 0.0007721217116340995, "num_tokens": 172936044.0, "reward": 2.341797113418579, "reward_std": 0.4802163243293762, "rewards/code_complexity_reward/mean": 0.9283202886581421, "rewards/code_complexity_reward/std": 0.0752602145075798, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1135, "step_time": 49.40588859748095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 126.841796875, "completions/mean_terminated_length": 126.841796875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24400258576497436, "epoch": 0.6472934472934473, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06153138354420662, "kl": 0.15310925233643502, "learning_rate": 1.6727934400315698e-06, "loss": 0.0007659501279704273, "num_tokens": 173069531.0, "reward": 2.3500490188598633, "reward_std": 0.5144785046577454, "rewards/code_complexity_reward/mean": 0.9211913347244263, "rewards/code_complexity_reward/std": 0.10466209053993225, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.04950176179409027, "step": 1136, "step_time": 35.33330123499036 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 119.814453125, "completions/mean_terminated_length": 119.814453125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22534209536388516, "epoch": 0.6478632478632479, "frac_reward_zero_std": 0.609375, "grad_norm": 0.04903896152973175, "kl": 0.14669440197758377, "learning_rate": 1.6681012349774123e-06, "loss": 0.0007332699606195092, "num_tokens": 173197100.0, "reward": 2.4044435024261475, "reward_std": 0.5040016174316406, "rewards/code_complexity_reward/mean": 0.9287109375, "rewards/code_complexity_reward/std": 0.07633930444717407, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1137, "step_time": 38.3041369151324 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 352.0, "completions/max_terminated_length": 352.0, "completions/mean_length": 130.822265625, "completions/mean_terminated_length": 130.822265625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24373412877321243, "epoch": 0.6484330484330484, "frac_reward_zero_std": 0.5, "grad_norm": 0.05532021075487137, "kl": 0.1490927212871611, "learning_rate": 1.6634123230301014e-06, "loss": 0.0007454143487848341, "num_tokens": 173336465.0, "reward": 2.2899904251098633, "reward_std": 0.46310409903526306, "rewards/code_complexity_reward/mean": 0.930371105670929, "rewards/code_complexity_reward/std": 0.07926063984632492, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 1138, "step_time": 38.75185564439744 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 126.890625, "completions/mean_terminated_length": 126.890625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23535462515428662, "epoch": 0.649002849002849, "frac_reward_zero_std": 0.609375, "grad_norm": 0.049489811062812805, "kl": 0.1561864729737863, "learning_rate": 1.6587267227508933e-06, "loss": 0.0007810996030457318, "num_tokens": 173472313.0, "reward": 2.4735350608825684, "reward_std": 0.4946264922618866, "rewards/code_complexity_reward/mean": 0.93798828125, "rewards/code_complexity_reward/std": 0.02376614511013031, "rewards/code_execution_reward/mean": 0.435546875, "rewards/code_execution_reward/std": 0.49631330370903015, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1139, "step_time": 37.25345916673541 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 378.0, "completions/max_terminated_length": 378.0, "completions/mean_length": 128.861328125, "completions/mean_terminated_length": 128.861328125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24149646051228046, "epoch": 0.6495726495726496, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06526965647935867, "kl": 0.15086408134084195, "learning_rate": 1.654044452687939e-06, "loss": 0.0007541439263150096, "num_tokens": 173604938.0, "reward": 2.3321776390075684, "reward_std": 0.5725505352020264, "rewards/code_complexity_reward/mean": 0.90185546875, "rewards/code_complexity_reward/std": 0.16941095888614655, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 1140, "step_time": 39.64865814615041 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 130.412109375, "completions/mean_terminated_length": 130.412109375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22642799629829824, "epoch": 0.6501424501424501, "frac_reward_zero_std": 0.546875, "grad_norm": 0.056249067187309265, "kl": 0.16033178218640387, "learning_rate": 1.6493655313762036e-06, "loss": 0.0008020773530006409, "num_tokens": 173740021.0, "reward": 2.3365235328674316, "reward_std": 0.5070674419403076, "rewards/code_complexity_reward/mean": 0.9210937023162842, "rewards/code_complexity_reward/std": 0.11184441298246384, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1141, "step_time": 41.231242588721216 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 134.466796875, "completions/mean_terminated_length": 133.7279815673828, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23683414398692548, "epoch": 0.6507122507122507, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05175904557108879, "kl": 0.15270274016074836, "learning_rate": 1.644689977337398e-06, "loss": 0.0007635836955159903, "num_tokens": 173880180.0, "reward": 2.2675294876098633, "reward_std": 0.4712856709957123, "rewards/code_complexity_reward/mean": 0.92431640625, "rewards/code_complexity_reward/std": 0.10360205173492432, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 1142, "step_time": 49.66060279868543 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 130.35546875, "completions/mean_terminated_length": 130.35546875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.24102315795607865, "epoch": 0.6512820512820513, "frac_reward_zero_std": 0.53125, "grad_norm": 0.062407080084085464, "kl": 0.15321721858344972, "learning_rate": 1.6400178090799013e-06, "loss": 0.0007659635157324374, "num_tokens": 174018610.0, "reward": 2.2946290969848633, "reward_std": 0.4873301386833191, "rewards/code_complexity_reward/mean": 0.923144519329071, "rewards/code_complexity_reward/std": 0.10494837909936905, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1143, "step_time": 45.2003970881924 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 127.9609375, "completions/mean_terminated_length": 127.9609375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24374888162128627, "epoch": 0.6518518518518519, "frac_reward_zero_std": 0.5625, "grad_norm": 0.054159849882125854, "kl": 0.16276063490658998, "learning_rate": 1.6353490450986924e-06, "loss": 0.0008138672565110028, "num_tokens": 174153198.0, "reward": 2.308837890625, "reward_std": 0.5052379965782166, "rewards/code_complexity_reward/mean": 0.920214831829071, "rewards/code_complexity_reward/std": 0.11901648342609406, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1144, "step_time": 38.697256200015545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 396.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 121.306640625, "completions/mean_terminated_length": 121.306640625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23616155050694942, "epoch": 0.6524216524216524, "frac_reward_zero_std": 0.625, "grad_norm": 0.04899701848626137, "kl": 0.1513902946608141, "learning_rate": 1.630683703875275e-06, "loss": 0.0007569340523332357, "num_tokens": 174284955.0, "reward": 2.3255372047424316, "reward_std": 0.45974281430244446, "rewards/code_complexity_reward/mean": 0.9357421398162842, "rewards/code_complexity_reward/std": 0.0483771488070488, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1145, "step_time": 49.84803128242493 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 124.853515625, "completions/mean_terminated_length": 124.09588623046875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24367141560651362, "epoch": 0.652991452991453, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05771809071302414, "kl": 0.15908450505230576, "learning_rate": 1.6260218038775994e-06, "loss": 0.0007954129250720143, "num_tokens": 174417272.0, "reward": 2.2998046875, "reward_std": 0.5077505707740784, "rewards/code_complexity_reward/mean": 0.9141601324081421, "rewards/code_complexity_reward/std": 0.11984855681657791, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1146, "step_time": 55.28460945934057 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 499.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 125.48046875, "completions/mean_terminated_length": 125.48046875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24245010199956596, "epoch": 0.6535612535612536, "frac_reward_zero_std": 0.546875, "grad_norm": 0.06244508922100067, "kl": 0.15341069549322128, "learning_rate": 1.621363363559997e-06, "loss": 0.0007670503109693527, "num_tokens": 174550846.0, "reward": 2.2786622047424316, "reward_std": 0.48585671186447144, "rewards/code_complexity_reward/mean": 0.9142577648162842, "rewards/code_complexity_reward/std": 0.1130082979798317, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1147, "step_time": 48.41716365888715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 331.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 122.45703125, "completions/mean_terminated_length": 122.45703125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.22050023637712002, "epoch": 0.6541310541310541, "frac_reward_zero_std": 0.625, "grad_norm": 0.05584075301885605, "kl": 0.14947482524439692, "learning_rate": 1.6167084013631031e-06, "loss": 0.0007473518489859998, "num_tokens": 174681632.0, "reward": 2.3160157203674316, "reward_std": 0.48441195487976074, "rewards/code_complexity_reward/mean": 0.9201171398162842, "rewards/code_complexity_reward/std": 0.10070973634719849, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1148, "step_time": 36.91124960966408 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 126.92578125, "completions/mean_terminated_length": 126.92578125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2297428548336029, "epoch": 0.6547008547008547, "frac_reward_zero_std": 0.515625, "grad_norm": 0.06404148042201996, "kl": 0.16396201425231993, "learning_rate": 1.6120569357137855e-06, "loss": 0.0008203122415579855, "num_tokens": 174815538.0, "reward": 2.3192384243011475, "reward_std": 0.4890243411064148, "rewards/code_complexity_reward/mean": 0.92626953125, "rewards/code_complexity_reward/std": 0.10402068495750427, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1149, "step_time": 36.34482603892684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 287.0, "completions/max_terminated_length": 287.0, "completions/mean_length": 130.724609375, "completions/mean_terminated_length": 130.724609375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.235672245034948, "epoch": 0.6552706552706553, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05411260202527046, "kl": 0.14825370826292783, "learning_rate": 1.6074089850250677e-06, "loss": 0.0007413565181195736, "num_tokens": 174953901.0, "reward": 2.408740520477295, "reward_std": 0.5009145736694336, "rewards/code_complexity_reward/mean": 0.929980456829071, "rewards/code_complexity_reward/std": 0.07791686058044434, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1150, "step_time": 35.733223146758974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 127.83203125, "completions/mean_terminated_length": 127.08023071289062, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2380967519711703, "epoch": 0.6558404558404558, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06554190069437027, "kl": 0.1519737709313631, "learning_rate": 1.6027645676960638e-06, "loss": 0.0007595901843160391, "num_tokens": 175087247.0, "reward": 2.3440918922424316, "reward_std": 0.5291414856910706, "rewards/code_complexity_reward/mean": 0.9108397960662842, "rewards/code_complexity_reward/std": 0.12595884501934052, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.019864002242684364, "step": 1151, "step_time": 48.99637116957456 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 382.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 121.46875, "completions/mean_terminated_length": 121.46875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2270834797527641, "epoch": 0.6564102564102564, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05356881394982338, "kl": 0.16237076243851334, "learning_rate": 1.5981237021118962e-06, "loss": 0.0008115150267258286, "num_tokens": 175217607.0, "reward": 2.34228515625, "reward_std": 0.5026194453239441, "rewards/code_complexity_reward/mean": 0.9239257574081421, "rewards/code_complexity_reward/std": 0.10359874367713928, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1152, "step_time": 41.33082789834589 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 123.1015625, "completions/mean_terminated_length": 123.1015625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.21892039920203388, "epoch": 0.656980056980057, "frac_reward_zero_std": 0.46875, "grad_norm": 0.059408772736787796, "kl": 0.1479068446205929, "learning_rate": 1.5934864066436316e-06, "loss": 0.0007395254215225577, "num_tokens": 175350651.0, "reward": 2.4645018577575684, "reward_std": 0.5446240901947021, "rewards/code_complexity_reward/mean": 0.9206054210662842, "rewards/code_complexity_reward/std": 0.11403566598892212, "rewards/code_execution_reward/mean": 0.451171875, "rewards/code_execution_reward/std": 0.498096764087677, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1153, "step_time": 38.61027605459094 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 311.0, "completions/max_terminated_length": 311.0, "completions/mean_length": 125.33203125, "completions/mean_terminated_length": 125.33203125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23808114416897297, "epoch": 0.6575498575498575, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05237441509962082, "kl": 0.1546329875709489, "learning_rate": 1.5888526996482023e-06, "loss": 0.00077343441080302, "num_tokens": 175482933.0, "reward": 2.4075684547424316, "reward_std": 0.5040322542190552, "rewards/code_complexity_reward/mean": 0.9320312738418579, "rewards/code_complexity_reward/std": 0.07744308561086655, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1154, "step_time": 44.54491860233247 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 135.6015625, "completions/mean_terminated_length": 134.12550354003906, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2424721687566489, "epoch": 0.6581196581196581, "frac_reward_zero_std": 0.515625, "grad_norm": 0.057270728051662445, "kl": 0.1632370832376182, "learning_rate": 1.5842225994683344e-06, "loss": 0.0008163676830008626, "num_tokens": 175621745.0, "reward": 2.1805665493011475, "reward_std": 0.4331265687942505, "rewards/code_complexity_reward/mean": 0.9199218153953552, "rewards/code_complexity_reward/std": 0.12645858526229858, "rewards/code_execution_reward/mean": 0.169921875, "rewards/code_execution_reward/std": 0.3759314715862274, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1155, "step_time": 69.16494607832283 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 122.974609375, "completions/mean_terminated_length": 122.974609375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2249363362789154, "epoch": 0.6586894586894587, "frac_reward_zero_std": 0.65625, "grad_norm": 0.049062274396419525, "kl": 0.1585088197607547, "learning_rate": 1.5795961244324791e-06, "loss": 0.000792241538874805, "num_tokens": 175754580.0, "reward": 2.3880860805511475, "reward_std": 0.49332597851753235, "rewards/code_complexity_reward/mean": 0.93359375, "rewards/code_complexity_reward/std": 0.05847352743148804, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1156, "step_time": 40.56336172390729 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 451.0, "completions/mean_length": 134.291015625, "completions/mean_terminated_length": 133.55186462402344, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.24118194379843771, "epoch": 0.6592592592592592, "frac_reward_zero_std": 0.46875, "grad_norm": 0.058775950223207474, "kl": 0.15499348822049797, "learning_rate": 1.574973292854734e-06, "loss": 0.0007751635275781155, "num_tokens": 175893665.0, "reward": 2.3265626430511475, "reward_std": 0.48391443490982056, "rewards/code_complexity_reward/mean": 0.92724609375, "rewards/code_complexity_reward/std": 0.07916413247585297, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1157, "step_time": 48.31725493166596 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 267.0, "completions/max_terminated_length": 267.0, "completions/mean_length": 123.470703125, "completions/mean_terminated_length": 123.470703125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2296791363041848, "epoch": 0.6598290598290598, "frac_reward_zero_std": 0.484375, "grad_norm": 0.058107808232307434, "kl": 0.1381272297585383, "learning_rate": 1.5703541230347774e-06, "loss": 0.0006905998452566564, "num_tokens": 176026546.0, "reward": 2.384326457977295, "reward_std": 0.526917040348053, "rewards/code_complexity_reward/mean": 0.9205077886581421, "rewards/code_complexity_reward/std": 0.1117786094546318, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1158, "step_time": 34.19021963700652 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 134.91796875, "completions/mean_terminated_length": 134.91796875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.24100663675926626, "epoch": 0.6603988603988604, "frac_reward_zero_std": 0.53125, "grad_norm": 0.08326663076877594, "kl": 0.15807155088987201, "learning_rate": 1.5657386332577896e-06, "loss": 0.0007902884972281754, "num_tokens": 176163240.0, "reward": 2.271777391433716, "reward_std": 0.46692919731140137, "rewards/code_complexity_reward/mean": 0.9247070550918579, "rewards/code_complexity_reward/std": 0.0985705703496933, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1159, "step_time": 42.59188989922404 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 422.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 129.5703125, "completions/mean_terminated_length": 129.5703125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23804237856529653, "epoch": 0.6609686609686609, "frac_reward_zero_std": 0.375, "grad_norm": 0.06223364546895027, "kl": 0.14609284640755504, "learning_rate": 1.5611268417943842e-06, "loss": 0.0007303510792553425, "num_tokens": 176299668.0, "reward": 2.4603028297424316, "reward_std": 0.5583493709564209, "rewards/code_complexity_reward/mean": 0.9203124642372131, "rewards/code_complexity_reward/std": 0.12896357476711273, "rewards/code_execution_reward/mean": 0.44921875, "rewards/code_execution_reward/std": 0.497901052236557, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1160, "step_time": 43.004539088346064 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 351.0, "completions/max_terminated_length": 351.0, "completions/mean_length": 127.66796875, "completions/mean_terminated_length": 127.66796875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22961792163550854, "epoch": 0.6615384615384615, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05340545251965523, "kl": 0.16173067199997604, "learning_rate": 1.5565187669005355e-06, "loss": 0.000808789161965251, "num_tokens": 176431586.0, "reward": 2.3449220657348633, "reward_std": 0.47933170199394226, "rewards/code_complexity_reward/mean": 0.929492175579071, "rewards/code_complexity_reward/std": 0.07565329223871231, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1161, "step_time": 37.88778489269316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 126.638671875, "completions/mean_terminated_length": 126.638671875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2329721050336957, "epoch": 0.6621082621082621, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06812135875225067, "kl": 0.14344359724782407, "learning_rate": 1.5519144268175055e-06, "loss": 0.0007173888152465224, "num_tokens": 176565593.0, "reward": 2.330810546875, "reward_std": 0.5092531442642212, "rewards/code_complexity_reward/mean": 0.9175781011581421, "rewards/code_complexity_reward/std": 0.11400721222162247, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1162, "step_time": 46.89965727366507 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 132.88671875, "completions/mean_terminated_length": 132.88671875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23590037319809198, "epoch": 0.6626780626780627, "frac_reward_zero_std": 0.5, "grad_norm": 0.06495973467826843, "kl": 0.14803439856041223, "learning_rate": 1.5473138397717693e-06, "loss": 0.0007402526098303497, "num_tokens": 176702975.0, "reward": 2.379687547683716, "reward_std": 0.5011703372001648, "rewards/code_complexity_reward/mean": 0.9232421517372131, "rewards/code_complexity_reward/std": 0.08315474539995193, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1163, "step_time": 44.81380531005561 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 123.732421875, "completions/mean_terminated_length": 123.732421875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22262496734037995, "epoch": 0.6632478632478632, "frac_reward_zero_std": 0.625, "grad_norm": 0.05554589629173279, "kl": 0.15344001806806773, "learning_rate": 1.542717023974949e-06, "loss": 0.0007672180654481053, "num_tokens": 176835334.0, "reward": 2.408984661102295, "reward_std": 0.5132657289505005, "rewards/code_complexity_reward/mean": 0.9271484613418579, "rewards/code_complexity_reward/std": 0.0819602832198143, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1164, "step_time": 41.37449342571199 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 383.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 132.787109375, "completions/mean_terminated_length": 132.787109375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23576009343378246, "epoch": 0.6638176638176638, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05138485133647919, "kl": 0.1594529850408435, "learning_rate": 1.5381239976237378e-06, "loss": 0.0007973198662512004, "num_tokens": 176975009.0, "reward": 2.3248047828674316, "reward_std": 0.5055973529815674, "rewards/code_complexity_reward/mean": 0.919628918170929, "rewards/code_complexity_reward/std": 0.11557082086801529, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1165, "step_time": 42.872379675507545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 130.03125, "completions/mean_terminated_length": 129.28375244140625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22852786886505783, "epoch": 0.6643874643874644, "frac_reward_zero_std": 0.421875, "grad_norm": 0.067618228495121, "kl": 0.17521356348879635, "learning_rate": 1.5335347788998249e-06, "loss": 0.0008761482313275337, "num_tokens": 177107633.0, "reward": 2.286376953125, "reward_std": 0.5457326769828796, "rewards/code_complexity_reward/mean": 0.9083007574081421, "rewards/code_complexity_reward/std": 0.16209985315799713, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1166, "step_time": 56.21710927411914 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 136.705078125, "completions/mean_terminated_length": 136.705078125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.22896552411839366, "epoch": 0.6649572649572649, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06279700994491577, "kl": 0.15222535794600844, "learning_rate": 1.52894938596983e-06, "loss": 0.0007612677873112261, "num_tokens": 177249458.0, "reward": 2.36474609375, "reward_std": 0.5010554790496826, "rewards/code_complexity_reward/mean": 0.92431640625, "rewards/code_complexity_reward/std": 0.08791742473840714, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 1167, "step_time": 55.64217258710414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 323.0, "completions/max_terminated_length": 323.0, "completions/mean_length": 120.341796875, "completions/mean_terminated_length": 120.341796875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23796477844007313, "epoch": 0.6655270655270655, "frac_reward_zero_std": 0.609375, "grad_norm": 0.06043066084384918, "kl": 0.1766526063438505, "learning_rate": 1.5243678369852261e-06, "loss": 0.0008831251179799438, "num_tokens": 177377665.0, "reward": 2.469531536102295, "reward_std": 0.5092197060585022, "rewards/code_complexity_reward/mean": 0.9339843988418579, "rewards/code_complexity_reward/std": 0.06393294036388397, "rewards/code_execution_reward/mean": 0.4375, "rewards/code_execution_reward/std": 0.49656352400779724, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1168, "step_time": 37.27694413345307 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 134.396484375, "completions/mean_terminated_length": 133.65753173828125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.23492081696167588, "epoch": 0.6660968660968661, "frac_reward_zero_std": 0.546875, "grad_norm": 0.050611723214387894, "kl": 0.1504359581740573, "learning_rate": 1.5197901500822724e-06, "loss": 0.0007524309912696481, "num_tokens": 177512804.0, "reward": 2.2892088890075684, "reward_std": 0.4838811159133911, "rewards/code_complexity_reward/mean": 0.91796875, "rewards/code_complexity_reward/std": 0.10411952435970306, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1169, "step_time": 48.718818517401814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 311.0, "completions/max_terminated_length": 311.0, "completions/mean_length": 129.763671875, "completions/mean_terminated_length": 129.763671875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2403448864351958, "epoch": 0.6666666666666666, "frac_reward_zero_std": 0.53125, "grad_norm": 0.056541044265031815, "kl": 0.15579443343449384, "learning_rate": 1.5152163433819365e-06, "loss": 0.0007793365512043238, "num_tokens": 177649931.0, "reward": 2.253467082977295, "reward_std": 0.4430689811706543, "rewards/code_complexity_reward/mean": 0.9273437261581421, "rewards/code_complexity_reward/std": 0.08733034878969193, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1170, "step_time": 42.82708792760968 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 126.69140625, "completions/mean_terminated_length": 125.9373779296875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23961403733119369, "epoch": 0.6672364672364672, "frac_reward_zero_std": 0.453125, "grad_norm": 0.07305030524730682, "kl": 0.15351115900557488, "learning_rate": 1.5106464349898299e-06, "loss": 0.0007673820946365595, "num_tokens": 177784245.0, "reward": 2.4178223609924316, "reward_std": 0.5406142473220825, "rewards/code_complexity_reward/mean": 0.9214843511581421, "rewards/code_complexity_reward/std": 0.11677893996238708, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1171, "step_time": 57.51679781265557 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 124.8828125, "completions/mean_terminated_length": 124.8828125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2401346368715167, "epoch": 0.6678062678062678, "frac_reward_zero_std": 0.5625, "grad_norm": 0.056371334940195084, "kl": 0.1543724772054702, "learning_rate": 1.5060804429961284e-06, "loss": 0.0007717548869550228, "num_tokens": 177918289.0, "reward": 2.4396486282348633, "reward_std": 0.5362680554389954, "rewards/code_complexity_reward/mean": 0.9275391101837158, "rewards/code_complexity_reward/std": 0.10774052888154984, "rewards/code_execution_reward/mean": 0.41796875, "rewards/code_execution_reward/std": 0.4937073290348053, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1172, "step_time": 41.359495741315186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 125.173828125, "completions/mean_terminated_length": 125.173828125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23473186907358468, "epoch": 0.6683760683760683, "frac_reward_zero_std": 0.4375, "grad_norm": 0.061910614371299744, "kl": 0.1684743290534243, "learning_rate": 1.5015183854755078e-06, "loss": 0.0008425905834883451, "num_tokens": 178050042.0, "reward": 2.315722703933716, "reward_std": 0.4933356046676636, "rewards/code_complexity_reward/mean": 0.923632800579071, "rewards/code_complexity_reward/std": 0.10504908114671707, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1173, "step_time": 36.0487834084779 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 131.25, "completions/mean_terminated_length": 131.25, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23895679228007793, "epoch": 0.6689458689458689, "frac_reward_zero_std": 0.578125, "grad_norm": 0.059482596814632416, "kl": 0.16246813023462892, "learning_rate": 1.496960280487068e-06, "loss": 0.0008122895378619432, "num_tokens": 178185298.0, "reward": 2.3338379859924316, "reward_std": 0.5043196678161621, "rewards/code_complexity_reward/mean": 0.9154296517372131, "rewards/code_complexity_reward/std": 0.1151711493730545, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 1174, "step_time": 39.04499179497361 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 129.908203125, "completions/mean_terminated_length": 129.908203125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.22769243083894253, "epoch": 0.6695156695156695, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05415249988436699, "kl": 0.16437309957109392, "learning_rate": 1.492406146074262e-06, "loss": 0.0008221640018746257, "num_tokens": 178319467.0, "reward": 2.395068645477295, "reward_std": 0.4853644073009491, "rewards/code_complexity_reward/mean": 0.9351562261581421, "rewards/code_complexity_reward/std": 0.05009164661169052, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1175, "step_time": 47.68172964733094 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 428.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 134.365234375, "completions/mean_terminated_length": 134.365234375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22341226576827466, "epoch": 0.67008547008547, "frac_reward_zero_std": 0.484375, "grad_norm": 0.0585196316242218, "kl": 0.14608727651648223, "learning_rate": 1.4878560002648266e-06, "loss": 0.0007303393795154989, "num_tokens": 178455966.0, "reward": 2.479248046875, "reward_std": 0.494465708732605, "rewards/code_complexity_reward/mean": 0.9263671636581421, "rewards/code_complexity_reward/std": 0.036318954080343246, "rewards/code_execution_reward/mean": 0.453125, "rewards/code_execution_reward/std": 0.4982847273349762, "rewards/code_syntax_reward/mean": 0.5, "rewards/code_syntax_reward/std": 0.0, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1176, "step_time": 51.0167223084718 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 323.0, "completions/max_terminated_length": 323.0, "completions/mean_length": 125.013671875, "completions/mean_terminated_length": 125.013671875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22938761790283024, "epoch": 0.6706552706552706, "frac_reward_zero_std": 0.5, "grad_norm": 0.06374508887529373, "kl": 0.1497145399916917, "learning_rate": 1.4833098610707069e-06, "loss": 0.0007485207170248032, "num_tokens": 178587677.0, "reward": 2.3729982376098633, "reward_std": 0.5260039567947388, "rewards/code_complexity_reward/mean": 0.9227539300918579, "rewards/code_complexity_reward/std": 0.11966536194086075, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1177, "step_time": 42.28385605942458 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 371.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 121.99609375, "completions/mean_terminated_length": 121.99609375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22213005111552775, "epoch": 0.6712250712250712, "frac_reward_zero_std": 0.515625, "grad_norm": 0.059269774705171585, "kl": 0.16174075519666076, "learning_rate": 1.4787677464879918e-06, "loss": 0.0008086316520348191, "num_tokens": 178718843.0, "reward": 2.404345750808716, "reward_std": 0.5003344416618347, "rewards/code_complexity_reward/mean": 0.9315429925918579, "rewards/code_complexity_reward/std": 0.06662218272686005, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1178, "step_time": 48.91300445050001 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 393.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 135.1796875, "completions/mean_terminated_length": 135.1796875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23049846314825118, "epoch": 0.6717948717948717, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05198212340474129, "kl": 0.15865293494425714, "learning_rate": 1.4742296744968338e-06, "loss": 0.0007931911386549473, "num_tokens": 178856295.0, "reward": 2.2753419876098633, "reward_std": 0.4508132040500641, "rewards/code_complexity_reward/mean": 0.9287109375, "rewards/code_complexity_reward/std": 0.07748428732156754, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1179, "step_time": 80.10253976657987 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 128.158203125, "completions/mean_terminated_length": 127.40704345703125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2363938852213323, "epoch": 0.6723646723646723, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05665365979075432, "kl": 0.1449460657313466, "learning_rate": 1.4696956630613867e-06, "loss": 0.0007247317698784173, "num_tokens": 178991944.0, "reward": 2.3126468658447266, "reward_std": 0.4991357624530792, "rewards/code_complexity_reward/mean": 0.9267578125, "rewards/code_complexity_reward/std": 0.11168009042739868, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1180, "step_time": 49.28879629354924 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 483.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 137.541015625, "completions/mean_terminated_length": 137.541015625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23828872968442738, "epoch": 0.672934472934473, "frac_reward_zero_std": 0.5, "grad_norm": 0.05644087493419647, "kl": 0.14998320559971035, "learning_rate": 1.4651657301297273e-06, "loss": 0.0007499116472899914, "num_tokens": 179130485.0, "reward": 2.29296875, "reward_std": 0.5506706237792969, "rewards/code_complexity_reward/mean": 0.9058593511581421, "rewards/code_complexity_reward/std": 0.16541723906993866, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1181, "step_time": 47.3870694776997 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 389.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 128.3125, "completions/mean_terminated_length": 128.3125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24552192934788764, "epoch": 0.6735042735042736, "frac_reward_zero_std": 0.5, "grad_norm": 0.06399370729923248, "kl": 0.15940504474565387, "learning_rate": 1.46063989363379e-06, "loss": 0.0007970895385369658, "num_tokens": 179265045.0, "reward": 2.334521770477295, "reward_std": 0.47853732109069824, "rewards/code_complexity_reward/mean": 0.9281250238418579, "rewards/code_complexity_reward/std": 0.06931309401988983, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1182, "step_time": 42.7442631283775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 122.4296875, "completions/mean_terminated_length": 122.4296875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23316331743262708, "epoch": 0.674074074074074, "frac_reward_zero_std": 0.671875, "grad_norm": 0.0510912761092186, "kl": 0.16458938119467348, "learning_rate": 1.4561181714892916e-06, "loss": 0.000822678382974118, "num_tokens": 179397121.0, "reward": 2.3309569358825684, "reward_std": 0.48764950037002563, "rewards/code_complexity_reward/mean": 0.92236328125, "rewards/code_complexity_reward/std": 0.08782476931810379, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1183, "step_time": 39.41698204353452 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 133.107421875, "completions/mean_terminated_length": 132.36595153808594, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23470585397444665, "epoch": 0.6746438746438747, "frac_reward_zero_std": 0.421875, "grad_norm": 0.053601980209350586, "kl": 0.14625168801285326, "learning_rate": 1.451600581595662e-06, "loss": 0.000731275649741292, "num_tokens": 179533448.0, "reward": 2.3121583461761475, "reward_std": 0.49208498001098633, "rewards/code_complexity_reward/mean": 0.92333984375, "rewards/code_complexity_reward/std": 0.1063869297504425, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1184, "step_time": 50.03754240646958 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 460.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 138.86328125, "completions/mean_terminated_length": 138.86328125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24067538185045123, "epoch": 0.6752136752136753, "frac_reward_zero_std": 0.59375, "grad_norm": 0.05343587324023247, "kl": 0.150873490027152, "learning_rate": 1.4470871418359761e-06, "loss": 0.0007543019019067287, "num_tokens": 179676706.0, "reward": 2.261767864227295, "reward_std": 0.47721004486083984, "rewards/code_complexity_reward/mean": 0.9193359017372131, "rewards/code_complexity_reward/std": 0.11669430881738663, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 1185, "step_time": 48.40006676223129 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 127.89453125, "completions/mean_terminated_length": 127.89453125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.224232017993927, "epoch": 0.6757834757834758, "frac_reward_zero_std": 0.53125, "grad_norm": 0.13170494139194489, "kl": 0.15189519431442022, "learning_rate": 1.4425778700768758e-06, "loss": 0.0007595627103000879, "num_tokens": 179810308.0, "reward": 2.37939453125, "reward_std": 0.5559698939323425, "rewards/code_complexity_reward/mean": 0.9140625, "rewards/code_complexity_reward/std": 0.13900506496429443, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1186, "step_time": 46.887382712215185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 328.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 122.427734375, "completions/mean_terminated_length": 122.427734375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23060085414908826, "epoch": 0.6763532763532764, "frac_reward_zero_std": 0.5, "grad_norm": 0.05822319537401199, "kl": 0.14312389155384153, "learning_rate": 1.4380727841685083e-06, "loss": 0.0007153606275096536, "num_tokens": 179940623.0, "reward": 2.3929688930511475, "reward_std": 0.499813437461853, "rewards/code_complexity_reward/mean": 0.927734375, "rewards/code_complexity_reward/std": 0.06746995449066162, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1187, "step_time": 48.101872467435896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 129.1796875, "completions/mean_terminated_length": 129.1796875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23116180463694036, "epoch": 0.676923076923077, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05717221274971962, "kl": 0.14884907728992403, "learning_rate": 1.4335719019444472e-06, "loss": 0.0007442950736731291, "num_tokens": 180073995.0, "reward": 2.289599657058716, "reward_std": 0.4612032473087311, "rewards/code_complexity_reward/mean": 0.9244140386581421, "rewards/code_complexity_reward/std": 0.07591354101896286, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 1188, "step_time": 43.82335096690804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 387.0, "completions/max_terminated_length": 387.0, "completions/mean_length": 128.595703125, "completions/mean_terminated_length": 128.595703125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2355308374390006, "epoch": 0.6774928774928775, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06882636994123459, "kl": 0.15774244314525276, "learning_rate": 1.4290752412216286e-06, "loss": 0.0007884894730523229, "num_tokens": 180208372.0, "reward": 2.3345704078674316, "reward_std": 0.5422701835632324, "rewards/code_complexity_reward/mean": 0.9123046398162842, "rewards/code_complexity_reward/std": 0.14536195993423462, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1189, "step_time": 48.855972471646965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 357.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 131.505859375, "completions/mean_terminated_length": 131.505859375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.23385776998475194, "epoch": 0.6780626780626781, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05663622170686722, "kl": 0.15432913252152503, "learning_rate": 1.4245828198002752e-06, "loss": 0.0007718389388173819, "num_tokens": 180342415.0, "reward": 2.3507325649261475, "reward_std": 0.4781023859977722, "rewards/code_complexity_reward/mean": 0.9328124523162842, "rewards/code_complexity_reward/std": 0.0642394945025444, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1190, "step_time": 51.52969419397414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 383.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 129.970703125, "completions/mean_terminated_length": 129.970703125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2403282264713198, "epoch": 0.6786324786324787, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06339943408966064, "kl": 0.1468573168385774, "learning_rate": 1.4200946554638305e-06, "loss": 0.0007340672891587019, "num_tokens": 180475680.0, "reward": 2.317187786102295, "reward_std": 0.46562352776527405, "rewards/code_complexity_reward/mean": 0.930957019329071, "rewards/code_complexity_reward/std": 0.06564196199178696, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1191, "step_time": 46.885051359422505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 133.615234375, "completions/mean_terminated_length": 129.1284637451172, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23729562480002642, "epoch": 0.6792022792022792, "frac_reward_zero_std": 0.5, "grad_norm": 0.0883951410651207, "kl": 0.14664584840647876, "learning_rate": 1.4156107659788836e-06, "loss": 0.000733178632799536, "num_tokens": 180613939.0, "reward": 2.3205080032348633, "reward_std": 0.5755894184112549, "rewards/code_complexity_reward/mean": 0.900683581829071, "rewards/code_complexity_reward/std": 0.17386625707149506, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 1192, "step_time": 49.072742578573525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 133.23828125, "completions/mean_terminated_length": 133.23828125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.24298499361611903, "epoch": 0.6797720797720798, "frac_reward_zero_std": 0.453125, "grad_norm": 0.0667843222618103, "kl": 0.15386075526475906, "learning_rate": 1.4111311690951047e-06, "loss": 0.0007693184306845069, "num_tokens": 180752069.0, "reward": 2.332519769668579, "reward_std": 0.5251429080963135, "rewards/code_complexity_reward/mean": 0.9209960699081421, "rewards/code_complexity_reward/std": 0.12634243071079254, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1193, "step_time": 48.22443950828165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 328.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 122.95703125, "completions/mean_terminated_length": 122.95703125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22930157906375825, "epoch": 0.6803418803418804, "frac_reward_zero_std": 0.5, "grad_norm": 0.07253802567720413, "kl": 0.15242520614992827, "learning_rate": 1.4066558825451676e-06, "loss": 0.0007621733238920569, "num_tokens": 180880911.0, "reward": 2.4158692359924316, "reward_std": 0.5639820098876953, "rewards/code_complexity_reward/mean": 0.9123046398162842, "rewards/code_complexity_reward/std": 0.14380542933940887, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 1194, "step_time": 46.94524802081287 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 391.0, "completions/max_terminated_length": 391.0, "completions/mean_length": 133.05859375, "completions/mean_terminated_length": 133.05859375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2327549543697387, "epoch": 0.6809116809116809, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05890914052724838, "kl": 0.1715667024254799, "learning_rate": 1.4021849240446874e-06, "loss": 0.000857917417306453, "num_tokens": 181020613.0, "reward": 2.280517816543579, "reward_std": 0.5009557008743286, "rewards/code_complexity_reward/mean": 0.9172852039337158, "rewards/code_complexity_reward/std": 0.12374240905046463, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1195, "step_time": 50.220877047628164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 130.142578125, "completions/mean_terminated_length": 130.142578125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23645366355776787, "epoch": 0.6814814814814815, "frac_reward_zero_std": 0.59375, "grad_norm": 0.051235176622867584, "kl": 0.14835781732108444, "learning_rate": 1.3977183112921438e-06, "loss": 0.0007418036111630499, "num_tokens": 181154814.0, "reward": 2.3721680641174316, "reward_std": 0.511962354183197, "rewards/code_complexity_reward/mean": 0.9215819835662842, "rewards/code_complexity_reward/std": 0.09856126457452774, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1196, "step_time": 38.460924081504345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 133.626953125, "completions/mean_terminated_length": 132.88648986816406, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.22400055336765945, "epoch": 0.6820512820512821, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06499191373586655, "kl": 0.15535912138875574, "learning_rate": 1.3932560619688134e-06, "loss": 0.0007766554481349885, "num_tokens": 181291471.0, "reward": 2.4798827171325684, "reward_std": 0.5269225239753723, "rewards/code_complexity_reward/mean": 0.923828125, "rewards/code_complexity_reward/std": 0.08785659074783325, "rewards/code_execution_reward/mean": 0.458984375, "rewards/code_execution_reward/std": 0.49880221486091614, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1197, "step_time": 49.308255426585674 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 130.34765625, "completions/mean_terminated_length": 130.34765625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22794597921893, "epoch": 0.6826210826210827, "frac_reward_zero_std": 0.5, "grad_norm": 0.06272444128990173, "kl": 0.1547918029827997, "learning_rate": 1.388798193738703e-06, "loss": 0.0007737616542726755, "num_tokens": 181426913.0, "reward": 2.2716307640075684, "reward_std": 0.4758627414703369, "rewards/code_complexity_reward/mean": 0.9210937023162842, "rewards/code_complexity_reward/std": 0.11332190036773682, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1198, "step_time": 38.395153812132776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 397.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 131.55859375, "completions/mean_terminated_length": 131.55859375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22558193281292915, "epoch": 0.6831908831908832, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05661467835307121, "kl": 0.14889216551091522, "learning_rate": 1.3843447242484725e-06, "loss": 0.0007445556111633778, "num_tokens": 181563647.0, "reward": 2.3211913108825684, "reward_std": 0.495779812335968, "rewards/code_complexity_reward/mean": 0.92236328125, "rewards/code_complexity_reward/std": 0.10437063872814178, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1199, "step_time": 42.61736520193517 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 444.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 129.419921875, "completions/mean_terminated_length": 129.419921875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2338833708781749, "epoch": 0.6837606837606838, "frac_reward_zero_std": 0.484375, "grad_norm": 0.08461295068264008, "kl": 0.15541940939147025, "learning_rate": 1.3798956711273736e-06, "loss": 0.00077723030699417, "num_tokens": 181697982.0, "reward": 2.3541016578674316, "reward_std": 0.5232956409454346, "rewards/code_complexity_reward/mean": 0.9181640148162842, "rewards/code_complexity_reward/std": 0.11838008463382721, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1200, "step_time": 55.94027252495289 }, { "epoch": 0.6837606837606838, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 183.36, "eval_completions/max_terminated_length": 183.36, "eval_completions/mean_length": 131.8375, "eval_completions/mean_terminated_length": 131.8375, "eval_completions/min_length": 93.83, "eval_completions/min_terminated_length": 93.83, "eval_entropy": 0.23239785470068455, "eval_frac_reward_zero_std": 0.41, "eval_kl": 0.149003172442317, "eval_loss": 0.0007453425205312669, "eval_num_tokens": 181697982.0, "eval_reward": 2.3142501187324522, "eval_reward_std": 0.2442215383797884, "eval_rewards/code_complexity_reward/mean": 0.917937484383583, "eval_rewards/code_complexity_reward/std": 0.04670793751254678, "eval_rewards/code_execution_reward/mean": 0.305, "eval_rewards/code_execution_reward/std": 0.18399026960134507, "eval_rewards/code_syntax_reward/mean": 0.491875, "eval_rewards/code_syntax_reward/std": 0.02175998643040657, "eval_rewards/reasoning_present_reward_func/mean": 0.09975000150501728, "eval_rewards/reasoning_present_reward_func/std": 0.0004629100486636162, "eval_rewards/xmlcount_reward_func/mean": 0.4996875, "eval_rewards/xmlcount_reward_func/std": 0.000578637570142746, "eval_runtime": 850.1157, "eval_samples_per_second": 0.118, "eval_steps_per_second": 0.015, "step": 1200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 127.12109375, "completions/mean_terminated_length": 127.12109375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23846090934239328, "epoch": 0.6843304843304844, "frac_reward_zero_std": 0.546875, "grad_norm": 0.06165429204702377, "kl": 0.14611120824702084, "learning_rate": 1.3754510519871716e-06, "loss": 0.0007304815808311105, "num_tokens": 181827684.0, "reward": 2.287890911102295, "reward_std": 0.5111667513847351, "rewards/code_complexity_reward/mean": 0.9105468392372131, "rewards/code_complexity_reward/std": 0.13195835053920746, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1201, "step_time": 54.21155703626573 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 128.2265625, "completions/mean_terminated_length": 128.2265625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2437476022168994, "epoch": 0.6849002849002849, "frac_reward_zero_std": 0.5, "grad_norm": 0.06044531241059303, "kl": 0.15310887480154634, "learning_rate": 1.3710108844220828e-06, "loss": 0.0007655006484128535, "num_tokens": 181964664.0, "reward": 2.3470215797424316, "reward_std": 0.5479745268821716, "rewards/code_complexity_reward/mean": 0.9152343273162842, "rewards/code_complexity_reward/std": 0.14386455714702606, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1202, "step_time": 41.19197986461222 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 282.0, "completions/max_terminated_length": 282.0, "completions/mean_length": 127.29296875, "completions/mean_terminated_length": 127.29296875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23901903885416687, "epoch": 0.6854700854700855, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05582733079791069, "kl": 0.14871348137967288, "learning_rate": 1.366575186008699e-06, "loss": 0.0007435556617565453, "num_tokens": 182102806.0, "reward": 2.3868165016174316, "reward_std": 0.5033907294273376, "rewards/code_complexity_reward/mean": 0.9264647960662842, "rewards/code_complexity_reward/std": 0.07647955417633057, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1203, "step_time": 35.61247096396983 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 453.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 142.642578125, "completions/mean_terminated_length": 142.642578125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22743596928194165, "epoch": 0.6860398860398861, "frac_reward_zero_std": 0.5, "grad_norm": 0.051358725875616074, "kl": 0.15256227611098439, "learning_rate": 1.3621439743059228e-06, "loss": 0.0007626960286870599, "num_tokens": 182252311.0, "reward": 2.2945802211761475, "reward_std": 0.5043850541114807, "rewards/code_complexity_reward/mean": 0.91552734375, "rewards/code_complexity_reward/std": 0.12118186056613922, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1204, "step_time": 55.21987647563219 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 135.505859375, "completions/mean_terminated_length": 134.76907348632812, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23463490046560764, "epoch": 0.6866096866096866, "frac_reward_zero_std": 0.5, "grad_norm": 0.05422716960310936, "kl": 0.14551599463447928, "learning_rate": 1.3577172668548965e-06, "loss": 0.0007277876138687134, "num_tokens": 182391762.0, "reward": 2.2792482376098633, "reward_std": 0.4980689287185669, "rewards/code_complexity_reward/mean": 0.9128906726837158, "rewards/code_complexity_reward/std": 0.12659874558448792, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1205, "step_time": 59.37924641184509 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 132.189453125, "completions/mean_terminated_length": 131.44618225097656, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23628516355529428, "epoch": 0.6871794871794872, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05967162176966667, "kl": 0.15805232094135135, "learning_rate": 1.3532950811789296e-06, "loss": 0.0007902115467004478, "num_tokens": 182527947.0, "reward": 2.280712842941284, "reward_std": 0.4929397404193878, "rewards/code_complexity_reward/mean": 0.919238269329071, "rewards/code_complexity_reward/std": 0.11917855590581894, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1206, "step_time": 48.175748580135405 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 127.328125, "completions/mean_terminated_length": 126.5753402709961, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.22370639070868492, "epoch": 0.6877492877492878, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06162892282009125, "kl": 0.15144737716764212, "learning_rate": 1.3488774347834326e-06, "loss": 0.0007573972688987851, "num_tokens": 182663035.0, "reward": 2.3756837844848633, "reward_std": 0.5169740915298462, "rewards/code_complexity_reward/mean": 0.9208984375, "rewards/code_complexity_reward/std": 0.10483784228563309, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1207, "step_time": 48.9942285772413 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 135.271484375, "completions/mean_terminated_length": 134.53424072265625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2247184389270842, "epoch": 0.6883190883190883, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06388332694768906, "kl": 0.14712555264122784, "learning_rate": 1.3444643451558498e-06, "loss": 0.0007354976842179894, "num_tokens": 182799262.0, "reward": 2.2958984375, "reward_std": 0.5091196298599243, "rewards/code_complexity_reward/mean": 0.9133788347244263, "rewards/code_complexity_reward/std": 0.13194207847118378, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1208, "step_time": 49.76583607401699 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 123.66796875, "completions/mean_terminated_length": 123.66796875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22873742366209626, "epoch": 0.6888888888888889, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05982634052634239, "kl": 0.14724188286345452, "learning_rate": 1.3400558297655836e-06, "loss": 0.0007361047901213169, "num_tokens": 182931908.0, "reward": 2.386474609375, "reward_std": 0.5085558891296387, "rewards/code_complexity_reward/mean": 0.9273437261581421, "rewards/code_complexity_reward/std": 0.08733034878969193, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1209, "step_time": 38.06419591791928 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 466.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 134.51953125, "completions/mean_terminated_length": 134.51953125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23369782138615847, "epoch": 0.6894586894586895, "frac_reward_zero_std": 0.484375, "grad_norm": 0.062323007732629776, "kl": 0.17005007446277887, "learning_rate": 1.3356519060639302e-06, "loss": 0.0008501154952682555, "num_tokens": 183068198.0, "reward": 2.3534669876098633, "reward_std": 0.4930475056171417, "rewards/code_complexity_reward/mean": 0.9248046875, "rewards/code_complexity_reward/std": 0.08364237844944, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1210, "step_time": 54.467754336073995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 132.71484375, "completions/mean_terminated_length": 131.97259521484375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2317509213462472, "epoch": 0.69002849002849, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06433609873056412, "kl": 0.16075260913930833, "learning_rate": 1.331252591484012e-06, "loss": 0.000803539005573839, "num_tokens": 183205460.0, "reward": 2.4066896438598633, "reward_std": 0.5116838812828064, "rewards/code_complexity_reward/mean": 0.9280272722244263, "rewards/code_complexity_reward/std": 0.0862947627902031, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1211, "step_time": 60.39645512402058 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 127.060546875, "completions/mean_terminated_length": 126.30724334716797, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2421147048007697, "epoch": 0.6905982905982906, "frac_reward_zero_std": 0.5, "grad_norm": 0.05924813821911812, "kl": 0.1512962746201083, "learning_rate": 1.326857903440702e-06, "loss": 0.0007564693805761635, "num_tokens": 183338011.0, "reward": 2.4203615188598633, "reward_std": 0.5254335403442383, "rewards/code_complexity_reward/mean": 0.926464855670929, "rewards/code_complexity_reward/std": 0.09558657556772232, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 1212, "step_time": 47.9440192701295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 137.912109375, "completions/mean_terminated_length": 137.912109375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22696354985237122, "epoch": 0.6911680911680912, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05800488218665123, "kl": 0.16156343999318779, "learning_rate": 1.3224678593305612e-06, "loss": 0.0008081374689936638, "num_tokens": 183475694.0, "reward": 2.3130860328674316, "reward_std": 0.510359525680542, "rewards/code_complexity_reward/mean": 0.9170898199081421, "rewards/code_complexity_reward/std": 0.12479308247566223, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1213, "step_time": 58.27993384562433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 128.02734375, "completions/mean_terminated_length": 128.02734375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.22535312362015247, "epoch": 0.6917378917378917, "frac_reward_zero_std": 0.484375, "grad_norm": 0.061300888657569885, "kl": 0.15627450786996633, "learning_rate": 1.318082476531769e-06, "loss": 0.0007813952397555113, "num_tokens": 183608060.0, "reward": 2.338134765625, "reward_std": 0.544265627861023, "rewards/code_complexity_reward/mean": 0.9104491472244263, "rewards/code_complexity_reward/std": 0.14386743307113647, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1214, "step_time": 43.490861374884844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 123.712890625, "completions/mean_terminated_length": 123.712890625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22887173248454928, "epoch": 0.6923076923076923, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06436195224523544, "kl": 0.15205461834557354, "learning_rate": 1.313701772404048e-06, "loss": 0.0007602018304169178, "num_tokens": 183738625.0, "reward": 2.3606934547424316, "reward_std": 0.4911687970161438, "rewards/code_complexity_reward/mean": 0.9359375238418579, "rewards/code_complexity_reward/std": 0.07466494292020798, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1215, "step_time": 53.285675342194736 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 316.0, "completions/max_terminated_length": 316.0, "completions/mean_length": 124.37109375, "completions/mean_terminated_length": 124.37109375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.24599446333013475, "epoch": 0.6928774928774929, "frac_reward_zero_std": 0.4375, "grad_norm": 0.08140486478805542, "kl": 0.16898857289925218, "learning_rate": 1.3093257642886048e-06, "loss": 0.0008450730820186436, "num_tokens": 183872703.0, "reward": 2.3114748001098633, "reward_std": 0.5142788887023926, "rewards/code_complexity_reward/mean": 0.9267578125, "rewards/code_complexity_reward/std": 0.13218335807323456, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1216, "step_time": 45.03925686143339 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 329.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 126.73828125, "completions/mean_terminated_length": 126.73828125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2262923016678542, "epoch": 0.6934472934472935, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06235829368233681, "kl": 0.15139905572868884, "learning_rate": 1.3049544695080534e-06, "loss": 0.0007568957516923547, "num_tokens": 184005457.0, "reward": 2.3970704078674316, "reward_std": 0.4883187413215637, "rewards/code_complexity_reward/mean": 0.9328124523162842, "rewards/code_complexity_reward/std": 0.04992656037211418, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1217, "step_time": 46.902277490124106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 451.0, "completions/max_terminated_length": 451.0, "completions/mean_length": 135.158203125, "completions/mean_terminated_length": 135.158203125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23746633366681635, "epoch": 0.694017094017094, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06182041019201279, "kl": 0.15312191110569984, "learning_rate": 1.3005879053663525e-06, "loss": 0.0007653897628188133, "num_tokens": 184143330.0, "reward": 2.333740234375, "reward_std": 0.49783340096473694, "rewards/code_complexity_reward/mean": 0.9205077886581421, "rewards/code_complexity_reward/std": 0.09994781762361526, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1218, "step_time": 44.075918816030025 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 125.060546875, "completions/mean_terminated_length": 124.30332946777344, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2229375378228724, "epoch": 0.6945868945868946, "frac_reward_zero_std": 0.5, "grad_norm": 0.06061180680990219, "kl": 0.15157657803501934, "learning_rate": 1.2962260891487325e-06, "loss": 0.000757806294132024, "num_tokens": 184274833.0, "reward": 2.3731935024261475, "reward_std": 0.5192341804504395, "rewards/code_complexity_reward/mean": 0.92578125, "rewards/code_complexity_reward/std": 0.10876181721687317, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1219, "step_time": 47.527129208669066 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 130.39453125, "completions/mean_terminated_length": 130.39453125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2298953509889543, "epoch": 0.6951566951566952, "frac_reward_zero_std": 0.6875, "grad_norm": 0.04736846685409546, "kl": 0.15103459730744362, "learning_rate": 1.2918690381216282e-06, "loss": 0.0007550434675067663, "num_tokens": 184410411.0, "reward": 2.300293207168579, "reward_std": 0.4746044874191284, "rewards/code_complexity_reward/mean": 0.9307616949081421, "rewards/code_complexity_reward/std": 0.08944274485111237, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1220, "step_time": 39.53531977534294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 138.69140625, "completions/mean_terminated_length": 137.9608612060547, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.22995151462964714, "epoch": 0.6957264957264957, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05399850383400917, "kl": 0.16584903351031244, "learning_rate": 1.2875167695326146e-06, "loss": 0.0008293819846585393, "num_tokens": 184549877.0, "reward": 2.3509278297424316, "reward_std": 0.527312159538269, "rewards/code_complexity_reward/mean": 0.9118163585662842, "rewards/code_complexity_reward/std": 0.12248256057500839, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 1221, "step_time": 49.801141910254955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 123.759765625, "completions/mean_terminated_length": 123.759765625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23460134118795395, "epoch": 0.6962962962962963, "frac_reward_zero_std": 0.546875, "grad_norm": 0.07758290320634842, "kl": 0.15769178443588316, "learning_rate": 1.2831693006103318e-06, "loss": 0.0007889963453635573, "num_tokens": 184678498.0, "reward": 2.323779582977295, "reward_std": 0.5026419758796692, "rewards/code_complexity_reward/mean": 0.9203125238418579, "rewards/code_complexity_reward/std": 0.11547359079122543, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1222, "step_time": 37.27405049465597 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 132.958984375, "completions/mean_terminated_length": 132.958984375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2368496956769377, "epoch": 0.6968660968660969, "frac_reward_zero_std": 0.515625, "grad_norm": 0.062423236668109894, "kl": 0.1460226266644895, "learning_rate": 1.278826648564421e-06, "loss": 0.0007297627744264901, "num_tokens": 184814533.0, "reward": 2.2472171783447266, "reward_std": 0.46631473302841187, "rewards/code_complexity_reward/mean": 0.9208984375, "rewards/code_complexity_reward/std": 0.11937039345502853, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1223, "step_time": 39.44728745985776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 382.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 133.810546875, "completions/mean_terminated_length": 133.810546875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23660911549814045, "epoch": 0.6974358974358974, "frac_reward_zero_std": 0.4375, "grad_norm": 0.061212074011564255, "kl": 0.13701581163331866, "learning_rate": 1.2744888305854566e-06, "loss": 0.000684996775817126, "num_tokens": 184955780.0, "reward": 2.318603515625, "reward_std": 0.5525044202804565, "rewards/code_complexity_reward/mean": 0.9041016101837158, "rewards/code_complexity_reward/std": 0.15267598628997803, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 1224, "step_time": 41.407601109705865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 125.3046875, "completions/mean_terminated_length": 125.3046875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23490127734839916, "epoch": 0.698005698005698, "frac_reward_zero_std": 0.5625, "grad_norm": 0.06319881230592728, "kl": 0.14660081395413727, "learning_rate": 1.270155863844878e-06, "loss": 0.0007330081425607204, "num_tokens": 185090376.0, "reward": 2.3294920921325684, "reward_std": 0.4956851899623871, "rewards/code_complexity_reward/mean": 0.9248046875, "rewards/code_complexity_reward/std": 0.10398122668266296, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1225, "step_time": 37.87342565692961 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 131.6015625, "completions/mean_terminated_length": 131.6015625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23726067156530917, "epoch": 0.6985754985754986, "frac_reward_zero_std": 0.65625, "grad_norm": 0.04458816722035408, "kl": 0.1465492834104225, "learning_rate": 1.265827765494917e-06, "loss": 0.0007329077343456447, "num_tokens": 185228820.0, "reward": 2.3033204078674316, "reward_std": 0.4753911793231964, "rewards/code_complexity_reward/mean": 0.9240233898162842, "rewards/code_complexity_reward/std": 0.08841407299041748, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1226, "step_time": 42.6346625611186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 418.0, "completions/max_terminated_length": 418.0, "completions/mean_length": 128.841796875, "completions/mean_terminated_length": 128.841796875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23142736521549523, "epoch": 0.6991452991452991, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05416635051369667, "kl": 0.1487462930381298, "learning_rate": 1.2615045526685382e-06, "loss": 0.0007435987936332822, "num_tokens": 185364595.0, "reward": 2.421679973602295, "reward_std": 0.5090196132659912, "rewards/code_complexity_reward/mean": 0.930957019329071, "rewards/code_complexity_reward/std": 0.07561659067869186, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1227, "step_time": 50.7735063591972 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 130.80078125, "completions/mean_terminated_length": 130.80078125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23385215131565928, "epoch": 0.6997150997150997, "frac_reward_zero_std": 0.484375, "grad_norm": 0.064646415412426, "kl": 0.1723334628622979, "learning_rate": 1.2571862424793623e-06, "loss": 0.0008618084248155355, "num_tokens": 185499581.0, "reward": 2.4332032203674316, "reward_std": 0.5103704929351807, "rewards/code_complexity_reward/mean": 0.9237304925918579, "rewards/code_complexity_reward/std": 0.07361472398042679, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1228, "step_time": 40.06573230866343 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 463.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 133.392578125, "completions/mean_terminated_length": 133.392578125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2298889432568103, "epoch": 0.7002849002849003, "frac_reward_zero_std": 0.453125, "grad_norm": 0.0649145096540451, "kl": 0.148923336295411, "learning_rate": 1.2528728520216077e-06, "loss": 0.0007446094532497227, "num_tokens": 185637542.0, "reward": 2.452197551727295, "reward_std": 0.5251095294952393, "rewards/code_complexity_reward/mean": 0.9237304925918579, "rewards/code_complexity_reward/std": 0.09686045348644257, "rewards/code_execution_reward/mean": 0.43359375, "rewards/code_execution_reward/std": 0.4960552453994751, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1229, "step_time": 57.56036171130836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 367.0, "completions/max_terminated_length": 367.0, "completions/mean_length": 132.15625, "completions/mean_terminated_length": 132.15625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2496970093343407, "epoch": 0.7008547008547008, "frac_reward_zero_std": 0.5, "grad_norm": 0.06441503018140793, "kl": 0.14914961881004274, "learning_rate": 1.2485643983700122e-06, "loss": 0.0007457341998815536, "num_tokens": 185774110.0, "reward": 2.1890625953674316, "reward_std": 0.4695602357387543, "rewards/code_complexity_reward/mean": 0.9132812023162842, "rewards/code_complexity_reward/std": 0.1550183743238449, "rewards/code_execution_reward/mean": 0.189453125, "rewards/code_execution_reward/std": 0.3922513723373413, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1230, "step_time": 39.46089053619653 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 133.5859375, "completions/mean_terminated_length": 133.5859375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22769123665057123, "epoch": 0.7014245014245014, "frac_reward_zero_std": 0.53125, "grad_norm": 1.6577262878417969, "kl": 0.697170631843619, "learning_rate": 1.244260898579776e-06, "loss": 0.0034844563342630863, "num_tokens": 185911930.0, "reward": 2.30712890625, "reward_std": 0.49939343333244324, "rewards/code_complexity_reward/mean": 0.9175781011581421, "rewards/code_complexity_reward/std": 0.1083301231265068, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 1231, "step_time": 42.7208192916587 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 411.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 130.341796875, "completions/mean_terminated_length": 130.341796875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2374554716516286, "epoch": 0.701994301994302, "frac_reward_zero_std": 0.578125, "grad_norm": 0.056227173656225204, "kl": 0.1525900944834575, "learning_rate": 1.2399623696864863e-06, "loss": 0.000762789451982826, "num_tokens": 186050705.0, "reward": 2.3514649868011475, "reward_std": 0.4966352581977844, "rewards/code_complexity_reward/mean": 0.92529296875, "rewards/code_complexity_reward/std": 0.09231694787740707, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1232, "step_time": 44.142261603847146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 435.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 135.919921875, "completions/mean_terminated_length": 135.919921875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.24108349555172026, "epoch": 0.7025641025641025, "frac_reward_zero_std": 0.484375, "grad_norm": 0.060709405690431595, "kl": 0.14974047045689076, "learning_rate": 1.2356688287060524e-06, "loss": 0.0007486734539270401, "num_tokens": 186190064.0, "reward": 2.2823731899261475, "reward_std": 0.5275187492370605, "rewards/code_complexity_reward/mean": 0.9064452648162842, "rewards/code_complexity_reward/std": 0.15146854519844055, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1233, "step_time": 45.33978269249201 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 131.650390625, "completions/mean_terminated_length": 131.650390625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23044970887713134, "epoch": 0.7031339031339031, "frac_reward_zero_std": 0.515625, "grad_norm": 0.0649295225739479, "kl": 0.1558472914621234, "learning_rate": 1.2313802926346422e-06, "loss": 0.0007791669340804219, "num_tokens": 186326813.0, "reward": 2.285449504852295, "reward_std": 0.5462989211082458, "rewards/code_complexity_reward/mean": 0.9061523675918579, "rewards/code_complexity_reward/std": 0.16553960740566254, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1234, "step_time": 56.58094254694879 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 377.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 135.44140625, "completions/mean_terminated_length": 135.44140625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23914741724729538, "epoch": 0.7037037037037037, "frac_reward_zero_std": 0.59375, "grad_norm": 0.053930845111608505, "kl": 0.14384128793608397, "learning_rate": 1.2270967784486071e-06, "loss": 0.0007190048927441239, "num_tokens": 186464639.0, "reward": 2.2774415016174316, "reward_std": 0.47398456931114197, "rewards/code_complexity_reward/mean": 0.9235351085662842, "rewards/code_complexity_reward/std": 0.10495457053184509, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1235, "step_time": 46.123278005979955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 132.263671875, "completions/mean_terminated_length": 132.263671875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22332841041497886, "epoch": 0.7042735042735043, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06602542102336884, "kl": 0.16236926964484155, "learning_rate": 1.222818303104423e-06, "loss": 0.0008118898258544505, "num_tokens": 186597278.0, "reward": 2.31884765625, "reward_std": 0.48727571964263916, "rewards/code_complexity_reward/mean": 0.9249023199081421, "rewards/code_complexity_reward/std": 0.09605725109577179, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1236, "step_time": 35.793593402951956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 127.912109375, "completions/mean_terminated_length": 127.912109375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.23927207151427865, "epoch": 0.7048433048433048, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05706116929650307, "kl": 0.15396956575568765, "learning_rate": 1.2185448835386157e-06, "loss": 0.0007698509143665433, "num_tokens": 186731225.0, "reward": 2.266406297683716, "reward_std": 0.46950486302375793, "rewards/code_complexity_reward/mean": 0.9212890267372131, "rewards/code_complexity_reward/std": 0.09588345885276794, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1237, "step_time": 46.45878542866558 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 133.05078125, "completions/mean_terminated_length": 132.3092041015625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2332425129134208, "epoch": 0.7054131054131054, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06292397528886795, "kl": 0.14619719237089157, "learning_rate": 1.2142765366677014e-06, "loss": 0.000730993808247149, "num_tokens": 186872715.0, "reward": 2.3458008766174316, "reward_std": 0.5196476578712463, "rewards/code_complexity_reward/mean": 0.917187511920929, "rewards/code_complexity_reward/std": 0.11389518529176712, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1238, "step_time": 60.11867171525955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 314.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 134.181640625, "completions/mean_terminated_length": 134.181640625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.234542258316651, "epoch": 0.705982905982906, "frac_reward_zero_std": 0.5, "grad_norm": 0.06112579256296158, "kl": 0.16355535562615842, "learning_rate": 1.2100132793881116e-06, "loss": 0.0008178256684914231, "num_tokens": 187009568.0, "reward": 2.2740235328674316, "reward_std": 0.4485670030117035, "rewards/code_complexity_reward/mean": 0.9289062023162842, "rewards/code_complexity_reward/std": 0.07645762711763382, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1239, "step_time": 45.88611823040992 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 362.0, "completions/max_terminated_length": 362.0, "completions/mean_length": 129.73046875, "completions/mean_terminated_length": 129.73046875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23210791405290365, "epoch": 0.7065527065527065, "frac_reward_zero_std": 0.375, "grad_norm": 0.06503135710954666, "kl": 0.1661777765257284, "learning_rate": 1.205755128576135e-06, "loss": 0.0008305726223625243, "num_tokens": 187145134.0, "reward": 2.3473145961761475, "reward_std": 0.5170137882232666, "rewards/code_complexity_reward/mean": 0.9176757335662842, "rewards/code_complexity_reward/std": 0.1190514788031578, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1240, "step_time": 49.77557050343603 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 136.8359375, "completions/mean_terminated_length": 136.8359375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2366640530526638, "epoch": 0.7071225071225071, "frac_reward_zero_std": 0.46875, "grad_norm": 0.054721757769584656, "kl": 0.15106990351341665, "learning_rate": 1.201502101087841e-06, "loss": 0.0007551205926574767, "num_tokens": 187281962.0, "reward": 2.2716798782348633, "reward_std": 0.43362656235694885, "rewards/code_complexity_reward/mean": 0.930468738079071, "rewards/code_complexity_reward/std": 0.05139325559139252, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1241, "step_time": 58.363094496540725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 491.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 132.6875, "completions/mean_terminated_length": 132.6875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23973907297477126, "epoch": 0.7076923076923077, "frac_reward_zero_std": 0.53125, "grad_norm": 0.051858752965927124, "kl": 0.1487143230624497, "learning_rate": 1.1972542137590237e-06, "loss": 0.0007435230654664338, "num_tokens": 187419690.0, "reward": 2.2950196266174316, "reward_std": 0.4674321115016937, "rewards/code_complexity_reward/mean": 0.929394543170929, "rewards/code_complexity_reward/std": 0.08589476346969604, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1242, "step_time": 77.05512745678425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 134.2734375, "completions/mean_terminated_length": 133.53424072265625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2222814168781042, "epoch": 0.7082621082621082, "frac_reward_zero_std": 0.546875, "grad_norm": 0.05582905188202858, "kl": 0.15868595393840224, "learning_rate": 1.1930114834051243e-06, "loss": 0.0007934584864415228, "num_tokens": 187556214.0, "reward": 2.382519483566284, "reward_std": 0.5148547887802124, "rewards/code_complexity_reward/mean": 0.925488293170929, "rewards/code_complexity_reward/std": 0.10232022404670715, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 1243, "step_time": 67.30342494882643 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 130.166015625, "completions/mean_terminated_length": 128.66864013671875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22942037507891655, "epoch": 0.7088319088319088, "frac_reward_zero_std": 0.53125, "grad_norm": 0.05526171624660492, "kl": 0.1477243696572259, "learning_rate": 1.1887739268211743e-06, "loss": 0.0007386209326796234, "num_tokens": 187690539.0, "reward": 2.3482911586761475, "reward_std": 0.5304945111274719, "rewards/code_complexity_reward/mean": 0.9201171398162842, "rewards/code_complexity_reward/std": 0.12611764669418335, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1244, "step_time": 49.64488561078906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 135.068359375, "completions/mean_terminated_length": 134.33071899414062, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2347910322714597, "epoch": 0.7094017094017094, "frac_reward_zero_std": 0.546875, "grad_norm": 0.0707087442278862, "kl": 0.15540727460756898, "learning_rate": 1.1845415607817219e-06, "loss": 0.0007771914824843407, "num_tokens": 187827726.0, "reward": 2.240966796875, "reward_std": 0.4800039827823639, "rewards/code_complexity_reward/mean": 0.9205077886581421, "rewards/code_complexity_reward/std": 0.13333497941493988, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1245, "step_time": 47.71844664774835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 131.150390625, "completions/mean_terminated_length": 131.150390625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23203720268793404, "epoch": 0.7099715099715099, "frac_reward_zero_std": 0.5, "grad_norm": 0.0613148994743824, "kl": 0.13878392532933503, "learning_rate": 1.180314402040768e-06, "loss": 0.0006936965510249138, "num_tokens": 187964539.0, "reward": 2.36328125, "reward_std": 0.4902467429637909, "rewards/code_complexity_reward/mean": 0.9302734136581421, "rewards/code_complexity_reward/std": 0.07618293166160583, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1246, "step_time": 38.25249280128628 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 131.021484375, "completions/mean_terminated_length": 130.2759246826172, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22200383129529655, "epoch": 0.7105413105413105, "frac_reward_zero_std": 0.53125, "grad_norm": 0.06102154031395912, "kl": 0.15542250429280102, "learning_rate": 1.1760924673317033e-06, "loss": 0.0007770856609568, "num_tokens": 188097286.0, "reward": 2.4022951126098633, "reward_std": 0.5209653973579407, "rewards/code_complexity_reward/mean": 0.924609363079071, "rewards/code_complexity_reward/std": 0.10014589875936508, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1247, "step_time": 68.31246996577829 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 133.34765625, "completions/mean_terminated_length": 130.3661346435547, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23589520738460124, "epoch": 0.7111111111111111, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06034427508711815, "kl": 0.1423877403140068, "learning_rate": 1.171875773367235e-06, "loss": 0.0007121505914255977, "num_tokens": 188235360.0, "reward": 2.3058595657348633, "reward_std": 0.5326151251792908, "rewards/code_complexity_reward/mean": 0.9158203601837158, "rewards/code_complexity_reward/std": 0.14448000490665436, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 1248, "step_time": 55.15283035766333 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 462.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 127.1796875, "completions/mean_terminated_length": 127.1796875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2249110892880708, "epoch": 0.7116809116809116, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06757640093564987, "kl": 0.14673170947935432, "learning_rate": 1.1676643368393286e-06, "loss": 0.0007336122216656804, "num_tokens": 188368084.0, "reward": 2.3833985328674316, "reward_std": 0.5143800973892212, "rewards/code_complexity_reward/mean": 0.926953136920929, "rewards/code_complexity_reward/std": 0.10032282769680023, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1249, "step_time": 55.61117851361632 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 127.94140625, "completions/mean_terminated_length": 127.94140625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23650395055301487, "epoch": 0.7122507122507122, "frac_reward_zero_std": 0.5, "grad_norm": 0.06619168817996979, "kl": 0.14139892242383212, "learning_rate": 1.1634581744191336e-06, "loss": 0.0007066525286063552, "num_tokens": 188502534.0, "reward": 2.4088869094848633, "reward_std": 0.5029308199882507, "rewards/code_complexity_reward/mean": 0.9329100847244263, "rewards/code_complexity_reward/std": 0.07595399767160416, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1250, "step_time": 36.87449887301773 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 122.39453125, "completions/mean_terminated_length": 122.39453125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22663812804967165, "epoch": 0.7128205128205128, "frac_reward_zero_std": 0.5, "grad_norm": 0.05960922688245773, "kl": 0.14972076460253447, "learning_rate": 1.159257302756926e-06, "loss": 0.0007487639086320996, "num_tokens": 188630272.0, "reward": 2.3874025344848633, "reward_std": 0.5060060024261475, "rewards/code_complexity_reward/mean": 0.932324230670929, "rewards/code_complexity_reward/std": 0.09628657251596451, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 1251, "step_time": 76.25630730763078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 295.0, "completions/max_terminated_length": 295.0, "completions/mean_length": 126.115234375, "completions/mean_terminated_length": 126.115234375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.22185029089450836, "epoch": 0.7133903133903133, "frac_reward_zero_std": 0.484375, "grad_norm": 0.07081814855337143, "kl": 0.15902344381902367, "learning_rate": 1.155061738482034e-06, "loss": 0.0007950199651531875, "num_tokens": 188761523.0, "reward": 2.3458008766174316, "reward_std": 0.5253150463104248, "rewards/code_complexity_reward/mean": 0.9206054210662842, "rewards/code_complexity_reward/std": 0.12566934525966644, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1252, "step_time": 35.957549904473126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 131.8046875, "completions/mean_terminated_length": 131.8046875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.24254391505382955, "epoch": 0.7139601139601139, "frac_reward_zero_std": 0.421875, "grad_norm": 0.061524663120508194, "kl": 0.15308510570321232, "learning_rate": 1.1508714982027797e-06, "loss": 0.0007655025692656636, "num_tokens": 188897831.0, "reward": 2.3294923305511475, "reward_std": 0.488264262676239, "rewards/code_complexity_reward/mean": 0.929882824420929, "rewards/code_complexity_reward/std": 0.08845037966966629, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 1253, "step_time": 39.61461532767862 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 295.0, "completions/max_terminated_length": 295.0, "completions/mean_length": 126.509765625, "completions/mean_terminated_length": 126.509765625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2403423390351236, "epoch": 0.7145299145299145, "frac_reward_zero_std": 0.5, "grad_norm": 0.0643434077501297, "kl": 0.1484020153293386, "learning_rate": 1.1466865985064072e-06, "loss": 0.0007421199698001146, "num_tokens": 189029524.0, "reward": 2.408740520477295, "reward_std": 0.5060040950775146, "rewards/code_complexity_reward/mean": 0.9332031011581421, "rewards/code_complexity_reward/std": 0.07675598561763763, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1254, "step_time": 35.288286635652184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 296.0, "completions/max_terminated_length": 296.0, "completions/mean_length": 128.7265625, "completions/mean_terminated_length": 128.7265625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23901086254045367, "epoch": 0.7150997150997151, "frac_reward_zero_std": 0.578125, "grad_norm": 0.05626853555440903, "kl": 0.14932605961803347, "learning_rate": 1.1425070559590214e-06, "loss": 0.0007464293739758432, "num_tokens": 189164512.0, "reward": 2.3144044876098633, "reward_std": 0.4842139184474945, "rewards/code_complexity_reward/mean": 0.9286133050918579, "rewards/code_complexity_reward/std": 0.086724653840065, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1255, "step_time": 35.87000576965511 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 129.4609375, "completions/mean_terminated_length": 128.7123260498047, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23454201011918485, "epoch": 0.7156695156695156, "frac_reward_zero_std": 0.6875, "grad_norm": 0.048927947878837585, "kl": 0.13821528980042785, "learning_rate": 1.1383328871055216e-06, "loss": 0.0006911064265295863, "num_tokens": 189298740.0, "reward": 2.3302247524261475, "reward_std": 0.48920416831970215, "rewards/code_complexity_reward/mean": 0.9306640625, "rewards/code_complexity_reward/std": 0.10115355998277664, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1256, "step_time": 47.992097621783614 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 394.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 129.775390625, "completions/mean_terminated_length": 129.775390625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23349074111320078, "epoch": 0.7162393162393162, "frac_reward_zero_std": 0.640625, "grad_norm": 0.0551271066069603, "kl": 0.14594119263347238, "learning_rate": 1.1341641084695327e-06, "loss": 0.0007296163821592927, "num_tokens": 189433449.0, "reward": 2.334765672683716, "reward_std": 0.4776126444339752, "rewards/code_complexity_reward/mean": 0.9349609017372131, "rewards/code_complexity_reward/std": 0.07493244111537933, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1257, "step_time": 41.410400327295065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 331.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 134.345703125, "completions/mean_terminated_length": 134.345703125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23256071261130273, "epoch": 0.7168091168091169, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06440875679254532, "kl": 0.15214401728007942, "learning_rate": 1.1300007365533432e-06, "loss": 0.0007605827413499355, "num_tokens": 189570682.0, "reward": 2.2616701126098633, "reward_std": 0.5003147721290588, "rewards/code_complexity_reward/mean": 0.9109374284744263, "rewards/code_complexity_reward/std": 0.1390402466058731, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1258, "step_time": 35.832824372686446 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 134.900390625, "completions/mean_terminated_length": 134.900390625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23827890958637, "epoch": 0.7173789173789173, "frac_reward_zero_std": 0.546875, "grad_norm": 0.059443164616823196, "kl": 0.1580548668280244, "learning_rate": 1.1258427878378377e-06, "loss": 0.0007903377991169691, "num_tokens": 189710679.0, "reward": 2.2859864234924316, "reward_std": 0.48233485221862793, "rewards/code_complexity_reward/mean": 0.9188476204872131, "rewards/code_complexity_reward/std": 0.10677279531955719, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1259, "step_time": 55.45539976563305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 351.0, "completions/max_terminated_length": 351.0, "completions/mean_length": 132.365234375, "completions/mean_terminated_length": 132.365234375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.244261693675071, "epoch": 0.717948717948718, "frac_reward_zero_std": 0.53125, "grad_norm": 0.055410031229257584, "kl": 0.15176334988791496, "learning_rate": 1.1216902787824366e-06, "loss": 0.0007591029861941934, "num_tokens": 189847666.0, "reward": 2.3553712368011475, "reward_std": 0.49947667121887207, "rewards/code_complexity_reward/mean": 0.9280273914337158, "rewards/code_complexity_reward/std": 0.09549856930971146, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09921874850988388, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 1260, "step_time": 38.145065248943865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 133.712890625, "completions/mean_terminated_length": 133.712890625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23033095616847277, "epoch": 0.7185185185185186, "frac_reward_zero_std": 0.40625, "grad_norm": 0.06332636624574661, "kl": 0.15602470410522074, "learning_rate": 1.117543225825022e-06, "loss": 0.0007800331804901361, "num_tokens": 189985431.0, "reward": 2.407958984375, "reward_std": 0.5051800012588501, "rewards/code_complexity_reward/mean": 0.930468738079071, "rewards/code_complexity_reward/std": 0.07750622928142548, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1261, "step_time": 67.91797418985516 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 426.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 131.798828125, "completions/mean_terminated_length": 131.798828125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22406239272095263, "epoch": 0.719088319088319, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06592466682195663, "kl": 0.13380079600028694, "learning_rate": 1.113401645381883e-06, "loss": 0.0006689990404993296, "num_tokens": 190119568.0, "reward": 2.37548828125, "reward_std": 0.537716269493103, "rewards/code_complexity_reward/mean": 0.9131835699081421, "rewards/code_complexity_reward/std": 0.12589025497436523, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1262, "step_time": 41.769652978517115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 271.0, "completions/max_terminated_length": 271.0, "completions/mean_length": 127.44921875, "completions/mean_terminated_length": 127.44921875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22322961688041687, "epoch": 0.7196581196581197, "frac_reward_zero_std": 0.625, "grad_norm": 0.06156415119767189, "kl": 0.1433166868519038, "learning_rate": 1.1092655538476405e-06, "loss": 0.0007160267559811473, "num_tokens": 190252422.0, "reward": 2.4489259719848633, "reward_std": 0.5028625130653381, "rewards/code_complexity_reward/mean": 0.930957019329071, "rewards/code_complexity_reward/std": 0.06428640335798264, "rewards/code_execution_reward/mean": 0.419921875, "rewards/code_execution_reward/std": 0.4940285086631775, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1263, "step_time": 36.849568472243845 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 134.310546875, "completions/mean_terminated_length": 134.310546875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.25311321043409407, "epoch": 0.7202279202279203, "frac_reward_zero_std": 0.46875, "grad_norm": 0.07354837656021118, "kl": 0.16081696236506104, "learning_rate": 1.105134967595191e-06, "loss": 0.0008039043168537319, "num_tokens": 190389757.0, "reward": 2.2373046875, "reward_std": 0.43414103984832764, "rewards/code_complexity_reward/mean": 0.9322265982627869, "rewards/code_complexity_reward/std": 0.08751026540994644, "rewards/code_execution_reward/mean": 0.208984375, "rewards/code_execution_reward/std": 0.40698084235191345, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1264, "step_time": 50.59394927788526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 330.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 134.333984375, "completions/mean_terminated_length": 134.333984375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23269164562225342, "epoch": 0.7207977207977208, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05374142900109291, "kl": 0.14744073583278805, "learning_rate": 1.1010099029756372e-06, "loss": 0.0007372424006462097, "num_tokens": 190531104.0, "reward": 2.3219239711761475, "reward_std": 0.474124014377594, "rewards/code_complexity_reward/mean": 0.924511730670929, "rewards/code_complexity_reward/std": 0.07782359421253204, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1265, "step_time": 37.22771355044097 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 124.1328125, "completions/mean_terminated_length": 124.1328125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2366110635921359, "epoch": 0.7213675213675214, "frac_reward_zero_std": 0.484375, "grad_norm": 0.07356170564889908, "kl": 0.15935628733132035, "learning_rate": 1.096890376318223e-06, "loss": 0.0007966895354911685, "num_tokens": 190660796.0, "reward": 2.4283692836761475, "reward_std": 0.49918439984321594, "rewards/code_complexity_reward/mean": 0.93701171875, "rewards/code_complexity_reward/std": 0.05103868246078491, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1266, "step_time": 84.25486082397401 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 288.0, "completions/max_terminated_length": 288.0, "completions/mean_length": 130.974609375, "completions/mean_terminated_length": 130.974609375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23988445638678968, "epoch": 0.721937321937322, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06631213426589966, "kl": 0.15501159999985248, "learning_rate": 1.092776403930272e-06, "loss": 0.0007748439675197005, "num_tokens": 190796247.0, "reward": 2.3969240188598633, "reward_std": 0.5115510821342468, "rewards/code_complexity_reward/mean": 0.92431640625, "rewards/code_complexity_reward/std": 0.08685369044542313, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1267, "step_time": 45.587725137360394 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 132.2578125, "completions/mean_terminated_length": 132.2578125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23483170964755118, "epoch": 0.7225071225071225, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06592388451099396, "kl": 0.15540462627541274, "learning_rate": 1.088668002097118e-06, "loss": 0.0007768657524138689, "num_tokens": 190930067.0, "reward": 2.2740724086761475, "reward_std": 0.5066416263580322, "rewards/code_complexity_reward/mean": 0.91748046875, "rewards/code_complexity_reward/std": 0.13859830796718597, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1268, "step_time": 35.19462421908975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 138.76953125, "completions/mean_terminated_length": 138.76953125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23440198344178498, "epoch": 0.7230769230769231, "frac_reward_zero_std": 0.484375, "grad_norm": 0.0546913705766201, "kl": 0.15290451666805893, "learning_rate": 1.0845651870820478e-06, "loss": 0.0007644651923328638, "num_tokens": 191069453.0, "reward": 2.2755372524261475, "reward_std": 0.4860514998435974, "rewards/code_complexity_reward/mean": 0.923828125, "rewards/code_complexity_reward/std": 0.11951710283756256, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1269, "step_time": 36.01589448098093 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 132.4609375, "completions/mean_terminated_length": 132.4609375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2353372722864151, "epoch": 0.7236467236467237, "frac_reward_zero_std": 0.515625, "grad_norm": 0.05829854682087898, "kl": 0.14628731599077582, "learning_rate": 1.0804679751262287e-06, "loss": 0.000731223146431148, "num_tokens": 191208569.0, "reward": 2.3899903297424316, "reward_std": 0.525018572807312, "rewards/code_complexity_reward/mean": 0.9271484017372131, "rewards/code_complexity_reward/std": 0.1046627014875412, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1270, "step_time": 49.55155619978905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 461.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 137.1640625, "completions/mean_terminated_length": 137.1640625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.25013814424164593, "epoch": 0.7242165242165243, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06550800800323486, "kl": 0.14682611159514636, "learning_rate": 1.0763763824486497e-06, "loss": 0.000734104251023382, "num_tokens": 191349997.0, "reward": 2.295459270477295, "reward_std": 0.5428755879402161, "rewards/code_complexity_reward/mean": 0.9053710699081421, "rewards/code_complexity_reward/std": 0.15613813698291779, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1271, "step_time": 54.31582154985517 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 463.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 135.15625, "completions/mean_terminated_length": 135.15625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2202536843251437, "epoch": 0.7247863247863248, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05843733623623848, "kl": 0.13816323585342616, "learning_rate": 1.0722904252460546e-06, "loss": 0.0006906662601977587, "num_tokens": 191486029.0, "reward": 2.381396532058716, "reward_std": 0.5282451510429382, "rewards/code_complexity_reward/mean": 0.9175781011581421, "rewards/code_complexity_reward/std": 0.11520250886678696, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1272, "step_time": 44.733535774983466 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 327.0, "completions/max_terminated_length": 327.0, "completions/mean_length": 134.9375, "completions/mean_terminated_length": 134.9375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2420067919883877, "epoch": 0.7253561253561254, "frac_reward_zero_std": 0.484375, "grad_norm": 0.07597747445106506, "kl": 0.14215202257037163, "learning_rate": 1.0682101196928812e-06, "loss": 0.000710474734660238, "num_tokens": 191627349.0, "reward": 2.2703614234924316, "reward_std": 0.49012428522109985, "rewards/code_complexity_reward/mean": 0.9159179925918579, "rewards/code_complexity_reward/std": 0.12032058089971542, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 1273, "step_time": 39.844207445159554 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 467.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 139.03515625, "completions/mean_terminated_length": 139.03515625, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.23348388075828552, "epoch": 0.725925925925926, "frac_reward_zero_std": 0.421875, "grad_norm": 0.06337945908308029, "kl": 0.14148926397319883, "learning_rate": 1.0641354819411923e-06, "loss": 0.0007074011955410242, "num_tokens": 191767879.0, "reward": 2.391845703125, "reward_std": 0.5305197834968567, "rewards/code_complexity_reward/mean": 0.9192382097244263, "rewards/code_complexity_reward/std": 0.10761606693267822, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1274, "step_time": 53.52196319215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 138.21875, "completions/mean_terminated_length": 137.48727416992188, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2406386893708259, "epoch": 0.7264957264957265, "frac_reward_zero_std": 0.46875, "grad_norm": 0.07002361118793488, "kl": 0.15634714206680655, "learning_rate": 1.060066528120616e-06, "loss": 0.0007819035090506077, "num_tokens": 191906031.0, "reward": 2.30615234375, "reward_std": 0.49859264492988586, "rewards/code_complexity_reward/mean": 0.9214843511581421, "rewards/code_complexity_reward/std": 0.1199614405632019, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1275, "step_time": 55.61973014380783 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 400.0, "completions/max_terminated_length": 400.0, "completions/mean_length": 134.83984375, "completions/mean_terminated_length": 134.83984375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22857428924180567, "epoch": 0.7270655270655271, "frac_reward_zero_std": 0.453125, "grad_norm": 0.06482529640197754, "kl": 0.15309007570613176, "learning_rate": 1.0560032743382814e-06, "loss": 0.000765350298024714, "num_tokens": 192043261.0, "reward": 2.2982420921325684, "reward_std": 0.48140251636505127, "rewards/code_complexity_reward/mean": 0.921875, "rewards/code_complexity_reward/std": 0.10251246392726898, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1276, "step_time": 41.04124493710697 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 129.265625, "completions/mean_terminated_length": 129.265625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23846861929632723, "epoch": 0.7276353276353277, "frac_reward_zero_std": 0.515625, "grad_norm": 0.06554228812456131, "kl": 0.15222975541837513, "learning_rate": 1.0519457366787505e-06, "loss": 0.0007610602770000696, "num_tokens": 192176965.0, "reward": 2.3725099563598633, "reward_std": 0.5051035284996033, "rewards/code_complexity_reward/mean": 0.92333984375, "rewards/code_complexity_reward/std": 0.08723395317792892, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1277, "step_time": 42.807916451245546 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 135.9296875, "completions/mean_terminated_length": 135.9296875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.23891224153339863, "epoch": 0.7282051282051282, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06604985147714615, "kl": 0.1490611732006073, "learning_rate": 1.0478939312039616e-06, "loss": 0.0007454443257302046, "num_tokens": 192314601.0, "reward": 2.294238328933716, "reward_std": 0.5080573558807373, "rewards/code_complexity_reward/mean": 0.9120116829872131, "rewards/code_complexity_reward/std": 0.12900057435035706, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1278, "step_time": 49.11759447399527 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 457.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 130.93359375, "completions/mean_terminated_length": 130.93359375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23303427593782544, "epoch": 0.7287749287749288, "frac_reward_zero_std": 0.5, "grad_norm": 0.0699029266834259, "kl": 0.1570177118992433, "learning_rate": 1.043847873953158e-06, "loss": 0.0007848691311664879, "num_tokens": 192449311.0, "reward": 2.2894043922424316, "reward_std": 0.4962543547153473, "rewards/code_complexity_reward/mean": 0.9217773675918579, "rewards/code_complexity_reward/std": 0.11657776683568954, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 1279, "step_time": 63.341452192515135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 132.130859375, "completions/mean_terminated_length": 132.130859375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23247361672110856, "epoch": 0.7293447293447294, "frac_reward_zero_std": 0.421875, "grad_norm": 0.0699739158153534, "kl": 0.15144172206055373, "learning_rate": 1.0398075809428321e-06, "loss": 0.0007576147327199578, "num_tokens": 192583690.0, "reward": 2.2933106422424316, "reward_std": 0.49433231353759766, "rewards/code_complexity_reward/mean": 0.9222656488418579, "rewards/code_complexity_reward/std": 0.1189991757273674, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1280, "step_time": 65.9736758004874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 134.07421875, "completions/mean_terminated_length": 134.07421875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2358765066601336, "epoch": 0.7299145299145299, "frac_reward_zero_std": 0.4375, "grad_norm": 0.060171663761138916, "kl": 0.16015500109642744, "learning_rate": 1.035773068166655e-06, "loss": 0.0008009662851691246, "num_tokens": 192722088.0, "reward": 2.32080078125, "reward_std": 0.5038326382637024, "rewards/code_complexity_reward/mean": 0.9151366949081421, "rewards/code_complexity_reward/std": 0.11399342119693756, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1281, "step_time": 45.15372377540916 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 443.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 131.728515625, "completions/mean_terminated_length": 131.728515625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23121631797403097, "epoch": 0.7304843304843305, "frac_reward_zero_std": 0.46875, "grad_norm": 0.07843738049268723, "kl": 0.1545756411505863, "learning_rate": 1.0317443515954202e-06, "loss": 0.0007730390643700957, "num_tokens": 192855429.0, "reward": 2.4141602516174316, "reward_std": 0.535354495048523, "rewards/code_complexity_reward/mean": 0.920605480670929, "rewards/code_complexity_reward/std": 0.1145920380949974, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1282, "step_time": 44.384709616191685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 321.0, "completions/mean_length": 130.08203125, "completions/mean_terminated_length": 129.3346405029297, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24017062853090465, "epoch": 0.7310541310541311, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06741420179605484, "kl": 0.15036879759281874, "learning_rate": 1.0277214471769728e-06, "loss": 0.000751382380258292, "num_tokens": 192991023.0, "reward": 2.3074707984924316, "reward_std": 0.4928913712501526, "rewards/code_complexity_reward/mean": 0.9234374761581421, "rewards/code_complexity_reward/std": 0.10509291291236877, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1283, "step_time": 58.15149530488998 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 315.0, "completions/max_terminated_length": 315.0, "completions/mean_length": 126.515625, "completions/mean_terminated_length": 126.515625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23798795137554407, "epoch": 0.7316239316239316, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06227855011820793, "kl": 0.14266136451624334, "learning_rate": 1.023704370836152e-06, "loss": 0.0007132139289751649, "num_tokens": 193121167.0, "reward": 2.3531250953674316, "reward_std": 0.5213846564292908, "rewards/code_complexity_reward/mean": 0.9219726324081421, "rewards/code_complexity_reward/std": 0.11895095556974411, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1284, "step_time": 36.16743385232985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 434.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 134.591796875, "completions/mean_terminated_length": 134.591796875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2452322884928435, "epoch": 0.7321937321937322, "frac_reward_zero_std": 0.46875, "grad_norm": 0.07311592996120453, "kl": 0.1447454678127542, "learning_rate": 1.0196931384747277e-06, "loss": 0.0007236610981635749, "num_tokens": 193259598.0, "reward": 2.3502440452575684, "reward_std": 0.5426309704780579, "rewards/code_complexity_reward/mean": 0.9137694835662842, "rewards/code_complexity_reward/std": 0.13845261931419373, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1285, "step_time": 44.395946885459125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 367.0, "completions/max_terminated_length": 367.0, "completions/mean_length": 135.890625, "completions/mean_terminated_length": 135.890625, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.23394879209809005, "epoch": 0.7327635327635328, "frac_reward_zero_std": 0.390625, "grad_norm": 0.0636376217007637, "kl": 0.1539342087926343, "learning_rate": 1.015687765971333e-06, "loss": 0.0007695470121689141, "num_tokens": 193401294.0, "reward": 2.3504884243011475, "reward_std": 0.490578830242157, "rewards/code_complexity_reward/mean": 0.9308593273162842, "rewards/code_complexity_reward/std": 0.0863187313079834, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1286, "step_time": 50.78738074749708 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 379.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 136.193359375, "completions/mean_terminated_length": 136.193359375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24569380446337163, "epoch": 0.7333333333333333, "frac_reward_zero_std": 0.453125, "grad_norm": 0.07854178547859192, "kl": 0.151545490603894, "learning_rate": 1.011688269181408e-06, "loss": 0.0007574677001684904, "num_tokens": 193542297.0, "reward": 2.2723634243011475, "reward_std": 0.5366648435592651, "rewards/code_complexity_reward/mean": 0.904492199420929, "rewards/code_complexity_reward/std": 0.1615704894065857, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1287, "step_time": 44.51069973129779 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 137.8359375, "completions/mean_terminated_length": 137.1037139892578, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21876720152795315, "epoch": 0.7339031339031339, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06466319411993027, "kl": 0.1482310490682721, "learning_rate": 1.0076946639371303e-06, "loss": 0.0007411977858282626, "num_tokens": 193680173.0, "reward": 2.3275880813598633, "reward_std": 0.5089769959449768, "rewards/code_complexity_reward/mean": 0.912402331829071, "rewards/code_complexity_reward/std": 0.1119454875588417, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 1288, "step_time": 57.78727673366666 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 133.1328125, "completions/mean_terminated_length": 132.39138793945312, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24792560562491417, "epoch": 0.7344729344729345, "frac_reward_zero_std": 0.421875, "grad_norm": 0.07552231103181839, "kl": 0.14648731844499707, "learning_rate": 1.0037069660473585e-06, "loss": 0.0007322101737372577, "num_tokens": 193822001.0, "reward": 2.3895020484924316, "reward_std": 0.546594500541687, "rewards/code_complexity_reward/mean": 0.9203125238418579, "rewards/code_complexity_reward/std": 0.12612506747245789, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 1289, "step_time": 50.56491082627326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 130.109375, "completions/mean_terminated_length": 130.109375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2366251212079078, "epoch": 0.7350427350427351, "frac_reward_zero_std": 0.5625, "grad_norm": 0.06117605045437813, "kl": 0.14894597069360316, "learning_rate": 9.997251912975639e-07, "loss": 0.0007448046235367656, "num_tokens": 193956801.0, "reward": 2.349902391433716, "reward_std": 0.4921944737434387, "rewards/code_complexity_reward/mean": 0.9354492425918579, "rewards/code_complexity_reward/std": 0.08548350632190704, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1290, "step_time": 46.94292411673814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 333.0, "completions/max_terminated_length": 333.0, "completions/mean_length": 130.763671875, "completions/mean_terminated_length": 130.763671875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23707980778999627, "epoch": 0.7356125356125356, "frac_reward_zero_std": 0.46875, "grad_norm": 0.0686858668923378, "kl": 0.15584323834627867, "learning_rate": 9.957493554497737e-07, "loss": 0.0007790606468915939, "num_tokens": 194091432.0, "reward": 2.2393555641174316, "reward_std": 0.4741445779800415, "rewards/code_complexity_reward/mean": 0.9196288585662842, "rewards/code_complexity_reward/std": 0.12640775740146637, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1291, "step_time": 36.29117288906127 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 299.0, "completions/max_terminated_length": 299.0, "completions/mean_length": 134.00390625, "completions/mean_terminated_length": 134.00390625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23098536557517946, "epoch": 0.7361823361823362, "frac_reward_zero_std": 0.5625, "grad_norm": 0.05965288728475571, "kl": 0.15326421544887125, "learning_rate": 9.917794742425026e-07, "loss": 0.0007660469855181873, "num_tokens": 194226642.0, "reward": 2.2599122524261475, "reward_std": 0.48308032751083374, "rewards/code_complexity_reward/mean": 0.9210937023162842, "rewards/code_complexity_reward/std": 0.12615172564983368, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1292, "step_time": 34.429113685153425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 130.82421875, "completions/mean_terminated_length": 130.82421875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23155253962613642, "epoch": 0.7367521367521368, "frac_reward_zero_std": 0.46875, "grad_norm": 0.0647348091006279, "kl": 0.16003701312001795, "learning_rate": 9.878155633906966e-07, "loss": 0.0007999228546395898, "num_tokens": 194364944.0, "reward": 2.326171875, "reward_std": 0.5335890650749207, "rewards/code_complexity_reward/mean": 0.912402331829071, "rewards/code_complexity_reward/std": 0.13798056542873383, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1293, "step_time": 58.271515055559576 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 125.875, "completions/mean_terminated_length": 124.36079406738281, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22191343130543828, "epoch": 0.7373219373219373, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06254924088716507, "kl": 0.16551764926407486, "learning_rate": 9.83857638585665e-07, "loss": 0.0008276531007140875, "num_tokens": 194495256.0, "reward": 2.348877191543579, "reward_std": 0.5224953889846802, "rewards/code_complexity_reward/mean": 0.916796863079071, "rewards/code_complexity_reward/std": 0.11489420384168625, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1294, "step_time": 87.30349608603865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 132.08984375, "completions/mean_terminated_length": 131.34637451171875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.22463893168605864, "epoch": 0.7378917378917379, "frac_reward_zero_std": 0.53125, "grad_norm": 0.09762970358133316, "kl": 0.1549294872675091, "learning_rate": 9.799057154950236e-07, "loss": 0.0007741625304333866, "num_tokens": 194630550.0, "reward": 2.1925294399261475, "reward_std": 0.4319709837436676, "rewards/code_complexity_reward/mean": 0.923828125, "rewards/code_complexity_reward/std": 0.11471309512853622, "rewards/code_execution_reward/mean": 0.17578125, "rewards/code_execution_reward/std": 0.3810062110424042, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1295, "step_time": 49.230848737992346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 302.0, "completions/max_terminated_length": 302.0, "completions/mean_length": 126.849609375, "completions/mean_terminated_length": 126.849609375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22778704180382192, "epoch": 0.7384615384615385, "frac_reward_zero_std": 0.5, "grad_norm": 0.0693117082118988, "kl": 0.15996531769633293, "learning_rate": 9.75959809762629e-07, "loss": 0.0007993113249540329, "num_tokens": 194761785.0, "reward": 2.40869140625, "reward_std": 0.524831235408783, "rewards/code_complexity_reward/mean": 0.9278320074081421, "rewards/code_complexity_reward/std": 0.10445921868085861, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1296, "step_time": 43.95249981340021 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 447.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 132.130859375, "completions/mean_terminated_length": 132.130859375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23496441403403878, "epoch": 0.739031339031339, "frac_reward_zero_std": 0.484375, "grad_norm": 0.07334158569574356, "kl": 0.151590789668262, "learning_rate": 9.720199370085158e-07, "loss": 0.0007578155491501093, "num_tokens": 194900532.0, "reward": 2.3087403774261475, "reward_std": 0.5076523423194885, "rewards/code_complexity_reward/mean": 0.9206054210662842, "rewards/code_complexity_reward/std": 0.12017690390348434, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1297, "step_time": 43.4410909852013 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 127.46875, "completions/mean_terminated_length": 126.71623992919922, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22981160902418196, "epoch": 0.7396011396011396, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06706592440605164, "kl": 0.1632897830568254, "learning_rate": 9.680861128288418e-07, "loss": 0.0008166123880073428, "num_tokens": 195034196.0, "reward": 2.2824220657348633, "reward_std": 0.49161872267723083, "rewards/code_complexity_reward/mean": 0.92724609375, "rewards/code_complexity_reward/std": 0.11925581842660904, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1298, "step_time": 48.01777061447501 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 366.0, "completions/mean_length": 133.41015625, "completions/mean_terminated_length": 132.66928100585938, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2399075785651803, "epoch": 0.7401709401709402, "frac_reward_zero_std": 0.515625, "grad_norm": 0.06081007421016693, "kl": 0.15362877526786178, "learning_rate": 9.641583527958158e-07, "loss": 0.0007680384442210197, "num_tokens": 195173118.0, "reward": 2.3297364711761475, "reward_std": 0.5078356862068176, "rewards/code_complexity_reward/mean": 0.92431640625, "rewards/code_complexity_reward/std": 0.11221624910831451, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1299, "step_time": 49.65447225049138 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 131.421875, "completions/mean_terminated_length": 130.67710876464844, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23125919396989048, "epoch": 0.7407407407407407, "frac_reward_zero_std": 0.359375, "grad_norm": 0.06744491308927536, "kl": 0.15724328346550465, "learning_rate": 9.602366724576455e-07, "loss": 0.0007862140191718936, "num_tokens": 195308390.0, "reward": 2.3252930641174316, "reward_std": 0.5111261010169983, "rewards/code_complexity_reward/mean": 0.9181640148162842, "rewards/code_complexity_reward/std": 0.11300964653491974, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1300, "step_time": 51.37759054917842 }, { "epoch": 0.7407407407407407, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 185.89, "eval_completions/max_terminated_length": 185.89, "eval_completions/mean_length": 131.235, "eval_completions/mean_terminated_length": 131.235, "eval_completions/min_length": 94.19, "eval_completions/min_terminated_length": 94.19, "eval_entropy": 0.22799469608813525, "eval_frac_reward_zero_std": 0.46, "eval_kl": 0.1494658175483346, "eval_loss": 0.0007473800797015429, "eval_num_tokens": 195308390.0, "eval_reward": 2.3182813715934754, "eval_reward_std": 0.22074986731633545, "eval_rewards/code_complexity_reward/mean": 0.920312482714653, "eval_rewards/code_complexity_reward/std": 0.04210827825590968, "eval_rewards/code_execution_reward/mean": 0.30625, "eval_rewards/code_execution_reward/std": 0.17292112439870835, "eval_rewards/code_syntax_reward/mean": 0.491875, "eval_rewards/code_syntax_reward/std": 0.01904443174600601, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.49984375, "eval_rewards/xmlcount_reward_func/std": 0.0004419417306780815, "eval_runtime": 858.6602, "eval_samples_per_second": 0.116, "eval_steps_per_second": 0.015, "step": 1300 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 130.884765625, "completions/mean_terminated_length": 130.13894653320312, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22776395338587463, "epoch": 0.7413105413105413, "frac_reward_zero_std": 0.4375, "grad_norm": 0.0628451257944107, "kl": 0.14273853413760662, "learning_rate": 9.56321087338469e-07, "loss": 0.0007136210915632546, "num_tokens": 195442603.0, "reward": 2.3417482376098633, "reward_std": 0.5447908043861389, "rewards/code_complexity_reward/mean": 0.9089843034744263, "rewards/code_complexity_reward/std": 0.13991661369800568, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1301, "step_time": 48.804088548757136 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 266.0, "completions/max_terminated_length": 266.0, "completions/mean_length": 130.904296875, "completions/mean_terminated_length": 130.904296875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2189727497752756, "epoch": 0.7418803418803419, "frac_reward_zero_std": 0.40625, "grad_norm": 0.10049878060817719, "kl": 0.1735290220240131, "learning_rate": 9.52411612938299e-07, "loss": 0.0008673090487718582, "num_tokens": 195578010.0, "reward": 2.2863283157348633, "reward_std": 0.5052838325500488, "rewards/code_complexity_reward/mean": 0.9148436784744263, "rewards/code_complexity_reward/std": 0.1336466372013092, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1302, "step_time": 33.163185399957 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 281.0, "completions/max_terminated_length": 281.0, "completions/mean_length": 127.884765625, "completions/mean_terminated_length": 127.884765625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2275636864360422, "epoch": 0.7424501424501424, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06667820364236832, "kl": 0.1449334358330816, "learning_rate": 9.485082647329555e-07, "loss": 0.0007243315340019763, "num_tokens": 195712135.0, "reward": 2.392822265625, "reward_std": 0.5116629004478455, "rewards/code_complexity_reward/mean": 0.930957019329071, "rewards/code_complexity_reward/std": 0.09607595950365067, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1303, "step_time": 33.82122263405472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 389.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 131.64453125, "completions/mean_terminated_length": 131.64453125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2308393211569637, "epoch": 0.743019943019943, "frac_reward_zero_std": 0.421875, "grad_norm": 0.08683138340711594, "kl": 0.15921021823305637, "learning_rate": 9.4461105817401e-07, "loss": 0.000795461586676538, "num_tokens": 195847753.0, "reward": 2.361377000808716, "reward_std": 0.5155400037765503, "rewards/code_complexity_reward/mean": 0.9249023199081421, "rewards/code_complexity_reward/std": 0.11243607103824615, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1304, "step_time": 40.38919993303716 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 423.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 131.90625, "completions/mean_terminated_length": 131.90625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2373979389667511, "epoch": 0.7435897435897436, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06807995587587357, "kl": 0.14265040308237076, "learning_rate": 9.407200086887228e-07, "loss": 0.000713012763299048, "num_tokens": 195984905.0, "reward": 2.407958984375, "reward_std": 0.5199530124664307, "rewards/code_complexity_reward/mean": 0.92529296875, "rewards/code_complexity_reward/std": 0.09842155873775482, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1305, "step_time": 46.02753511350602 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 130.05078125, "completions/mean_terminated_length": 130.05078125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.24137857602909207, "epoch": 0.7441595441595441, "frac_reward_zero_std": 0.4375, "grad_norm": 0.07361232489347458, "kl": 0.16065120964776725, "learning_rate": 9.36835131679977e-07, "loss": 0.0008033670019358397, "num_tokens": 196119435.0, "reward": 2.3421878814697266, "reward_std": 0.5194339752197266, "rewards/code_complexity_reward/mean": 0.9189453125, "rewards/code_complexity_reward/std": 0.1281452178955078, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1306, "step_time": 37.14458568114787 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 127.669921875, "completions/mean_terminated_length": 127.669921875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.24419515277259052, "epoch": 0.7447293447293447, "frac_reward_zero_std": 0.421875, "grad_norm": 0.07127337902784348, "kl": 0.15646162710618228, "learning_rate": 9.329564425262272e-07, "loss": 0.000782164977863431, "num_tokens": 196251754.0, "reward": 2.4333009719848633, "reward_std": 0.5077572464942932, "rewards/code_complexity_reward/mean": 0.932910144329071, "rewards/code_complexity_reward/std": 0.06735081970691681, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1307, "step_time": 52.443253512494266 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 353.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 133.134765625, "completions/mean_terminated_length": 133.134765625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2397970852907747, "epoch": 0.7452991452991453, "frac_reward_zero_std": 0.5, "grad_norm": 0.06162083148956299, "kl": 0.15371890517417341, "learning_rate": 9.290839565814286e-07, "loss": 0.0007686017197556794, "num_tokens": 196387223.0, "reward": 2.3200197219848633, "reward_std": 0.5107380151748657, "rewards/code_complexity_reward/mean": 0.924023449420929, "rewards/code_complexity_reward/std": 0.11935501545667648, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1308, "step_time": 65.15763980988413 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 277.0, "completions/max_terminated_length": 277.0, "completions/mean_length": 124.5, "completions/mean_terminated_length": 124.5, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2389489591587335, "epoch": 0.7458689458689459, "frac_reward_zero_std": 0.484375, "grad_norm": 0.06675752997398376, "kl": 0.1547224969835952, "learning_rate": 9.252176891749829e-07, "loss": 0.0007733363891020417, "num_tokens": 196520063.0, "reward": 2.336230516433716, "reward_std": 0.5053537487983704, "rewards/code_complexity_reward/mean": 0.9227538704872131, "rewards/code_complexity_reward/std": 0.1125006303191185, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1309, "step_time": 41.88691252004355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 136.173828125, "completions/mean_terminated_length": 136.173828125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23623366793617606, "epoch": 0.7464387464387464, "frac_reward_zero_std": 0.359375, "grad_norm": 0.06278769671916962, "kl": 0.1492636731127277, "learning_rate": 9.21357655611674e-07, "loss": 0.0007463564397767186, "num_tokens": 196657760.0, "reward": 2.4015138149261475, "reward_std": 0.5285062789916992, "rewards/code_complexity_reward/mean": 0.9248046875, "rewards/code_complexity_reward/std": 0.10795175284147263, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1310, "step_time": 43.03768609277904 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 472.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 132.052734375, "completions/mean_terminated_length": 132.052734375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23263846756890416, "epoch": 0.747008547008547, "frac_reward_zero_std": 0.40625, "grad_norm": 0.07427442818880081, "kl": 0.15233154967427254, "learning_rate": 9.175038711716119e-07, "loss": 0.0007611840264871716, "num_tokens": 196795339.0, "reward": 2.3761720657348633, "reward_std": 0.5242538452148438, "rewards/code_complexity_reward/mean": 0.9216796159744263, "rewards/code_complexity_reward/std": 0.11282145231962204, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1311, "step_time": 53.23598570283502 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 131.236328125, "completions/mean_terminated_length": 131.236328125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23907381133176386, "epoch": 0.7475783475783476, "frac_reward_zero_std": 0.375, "grad_norm": 0.0742715522646904, "kl": 0.14040851732715964, "learning_rate": 9.136563511101651e-07, "loss": 0.0007015554583631456, "num_tokens": 196934812.0, "reward": 2.3104004859924316, "reward_std": 0.49401310086250305, "rewards/code_complexity_reward/mean": 0.924609363079071, "rewards/code_complexity_reward/std": 0.10699526965618134, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1312, "step_time": 37.82766311150044 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 133.458984375, "completions/mean_terminated_length": 132.71820068359375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2353745847940445, "epoch": 0.7481481481481481, "frac_reward_zero_std": 0.578125, "grad_norm": 0.061170339584350586, "kl": 0.15443786955438554, "learning_rate": 9.098151106579081e-07, "loss": 0.0007722804439254105, "num_tokens": 197073007.0, "reward": 2.2631349563598633, "reward_std": 0.4861733913421631, "rewards/code_complexity_reward/mean": 0.920214831829071, "rewards/code_complexity_reward/std": 0.1259661316871643, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1313, "step_time": 53.84032853972167 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 324.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 131.05078125, "completions/mean_terminated_length": 131.05078125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22700667544268072, "epoch": 0.7487179487179487, "frac_reward_zero_std": 0.421875, "grad_norm": 0.08233053237199783, "kl": 0.15764506824780256, "learning_rate": 9.059801650205538e-07, "loss": 0.0007876467425376177, "num_tokens": 197207625.0, "reward": 2.422900438308716, "reward_std": 0.5293667912483215, "rewards/code_complexity_reward/mean": 0.9266601800918579, "rewards/code_complexity_reward/std": 0.10560225695371628, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1314, "step_time": 36.96443772036582 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 329.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 124.40234375, "completions/mean_terminated_length": 124.40234375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2182834674604237, "epoch": 0.7492877492877493, "frac_reward_zero_std": 0.5, "grad_norm": 0.07180368900299072, "kl": 0.158855099696666, "learning_rate": 9.021515293788996e-07, "loss": 0.0007941153598949313, "num_tokens": 197336871.0, "reward": 2.419726848602295, "reward_std": 0.5130980610847473, "rewards/code_complexity_reward/mean": 0.9291015267372131, "rewards/code_complexity_reward/std": 0.08732137829065323, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1315, "step_time": 37.15532579924911 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 288.0, "completions/max_terminated_length": 288.0, "completions/mean_length": 126.591796875, "completions/mean_terminated_length": 126.591796875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2393814497627318, "epoch": 0.7498575498575498, "frac_reward_zero_std": 0.421875, "grad_norm": 0.0740618109703064, "kl": 0.1478769819950685, "learning_rate": 8.983292188887641e-07, "loss": 0.0007391985272988677, "num_tokens": 197473054.0, "reward": 2.4012207984924316, "reward_std": 0.5272465944290161, "rewards/code_complexity_reward/mean": 0.9235351085662842, "rewards/code_complexity_reward/std": 0.11342299729585648, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1316, "step_time": 55.41614061407745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 126.4921875, "completions/mean_terminated_length": 125.7377700805664, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22532562352716923, "epoch": 0.7504273504273504, "frac_reward_zero_std": 0.5, "grad_norm": 0.06363329291343689, "kl": 0.15422486851457506, "learning_rate": 8.945132486809252e-07, "loss": 0.0007707830518484116, "num_tokens": 197604034.0, "reward": 2.393115282058716, "reward_std": 0.5105565786361694, "rewards/code_complexity_reward/mean": 0.9281249642372131, "rewards/code_complexity_reward/std": 0.08797571063041687, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1317, "step_time": 60.57317417021841 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 129.990234375, "completions/mean_terminated_length": 129.990234375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23772483714856207, "epoch": 0.750997150997151, "frac_reward_zero_std": 0.46875, "grad_norm": 0.06456857919692993, "kl": 0.1638486674055457, "learning_rate": 8.907036338610658e-07, "loss": 0.000819153618067503, "num_tokens": 197740797.0, "reward": 2.3603515625, "reward_std": 0.5124480128288269, "rewards/code_complexity_reward/mean": 0.9292968511581421, "rewards/code_complexity_reward/std": 0.11265821009874344, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1318, "step_time": 61.896131405606866 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 270.0, "completions/max_terminated_length": 270.0, "completions/mean_length": 123.439453125, "completions/mean_terminated_length": 123.439453125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2229906755965203, "epoch": 0.7515669515669515, "frac_reward_zero_std": 0.484375, "grad_norm": 0.07665048539638519, "kl": 0.15580035746097565, "learning_rate": 8.869003895097078e-07, "loss": 0.000778660352807492, "num_tokens": 197871630.0, "reward": 2.2918946743011475, "reward_std": 0.47666460275650024, "rewards/code_complexity_reward/mean": 0.93212890625, "rewards/code_complexity_reward/std": 0.10467524081468582, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1319, "step_time": 36.597522204741836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 127.599609375, "completions/mean_terminated_length": 127.599609375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23963913973420858, "epoch": 0.7521367521367521, "frac_reward_zero_std": 0.390625, "grad_norm": 0.07634620368480682, "kl": 0.15079453983344138, "learning_rate": 8.83103530682158e-07, "loss": 0.0007538426434621215, "num_tokens": 198008449.0, "reward": 2.3522462844848633, "reward_std": 0.514619767665863, "rewards/code_complexity_reward/mean": 0.925097644329071, "rewards/code_complexity_reward/std": 0.11265341937541962, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1320, "step_time": 66.61148410104215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 306.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 129.8828125, "completions/mean_terminated_length": 129.8828125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23990117060020566, "epoch": 0.7527065527065527, "frac_reward_zero_std": 0.40625, "grad_norm": 0.07409483939409256, "kl": 0.1511731781065464, "learning_rate": 8.793130724084436e-07, "loss": 0.0007555817719548941, "num_tokens": 198144757.0, "reward": 2.2982420921325684, "reward_std": 0.47832348942756653, "rewards/code_complexity_reward/mean": 0.92578125, "rewards/code_complexity_reward/std": 0.09757015854120255, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1321, "step_time": 35.58239844813943 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 261.0, "completions/max_terminated_length": 261.0, "completions/mean_length": 125.306640625, "completions/mean_terminated_length": 125.306640625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23127251327969134, "epoch": 0.7532763532763532, "frac_reward_zero_std": 0.390625, "grad_norm": 0.08139171451330185, "kl": 0.15844469529110938, "learning_rate": 8.755290296932561e-07, "loss": 0.0007920832140371203, "num_tokens": 198277626.0, "reward": 2.3607423305511475, "reward_std": 0.5211376547813416, "rewards/code_complexity_reward/mean": 0.930371105670929, "rewards/code_complexity_reward/std": 0.11956566572189331, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 1322, "step_time": 32.60388107225299 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 137.9375, "completions/mean_terminated_length": 137.20547485351562, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23260622681118548, "epoch": 0.7538461538461538, "frac_reward_zero_std": 0.4375, "grad_norm": 0.07820121198892593, "kl": 0.14400446589570493, "learning_rate": 8.717514175158895e-07, "loss": 0.0007198089151643217, "num_tokens": 198418562.0, "reward": 2.322021722793579, "reward_std": 0.5259458422660828, "rewards/code_complexity_reward/mean": 0.9107421636581421, "rewards/code_complexity_reward/std": 0.13613994419574738, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1323, "step_time": 49.16400844510645 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 426.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 133.34765625, "completions/mean_terminated_length": 133.34765625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2387918105814606, "epoch": 0.7544159544159544, "frac_reward_zero_std": 0.453125, "grad_norm": 0.07454630732536316, "kl": 0.1460516331717372, "learning_rate": 8.67980250830184e-07, "loss": 0.0007299709250219166, "num_tokens": 198556404.0, "reward": 2.287402629852295, "reward_std": 0.4729911684989929, "rewards/code_complexity_reward/mean": 0.9295898675918579, "rewards/code_complexity_reward/std": 0.10528327524662018, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1324, "step_time": 45.456603275612 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 435.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 125.638671875, "completions/mean_terminated_length": 125.638671875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23773227236233652, "epoch": 0.7549857549857549, "frac_reward_zero_std": 0.4375, "grad_norm": 0.08423313498497009, "kl": 0.1634178552776575, "learning_rate": 8.642155445644648e-07, "loss": 0.00081680528819561, "num_tokens": 198688811.0, "reward": 2.2958009243011475, "reward_std": 0.4861053228378296, "rewards/code_complexity_reward/mean": 0.92626953125, "rewards/code_complexity_reward/std": 0.10556135326623917, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1325, "step_time": 46.34683719556779 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 137.376953125, "completions/mean_terminated_length": 137.376953125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2332092234864831, "epoch": 0.7555555555555555, "frac_reward_zero_std": 0.390625, "grad_norm": 0.06823813170194626, "kl": 0.14797681185882539, "learning_rate": 8.604573136214814e-07, "loss": 0.0007397118024528027, "num_tokens": 198828236.0, "reward": 2.400146484375, "reward_std": 0.49641692638397217, "rewards/code_complexity_reward/mean": 0.9332031011581421, "rewards/code_complexity_reward/std": 0.06636415421962738, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1326, "step_time": 44.353794309310615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 503.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 125.09765625, "completions/mean_terminated_length": 125.09765625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23108892678283155, "epoch": 0.7561253561253561, "frac_reward_zero_std": 0.4375, "grad_norm": 0.08200597763061523, "kl": 0.15746669610962272, "learning_rate": 8.567055728783532e-07, "loss": 0.0007871586130931973, "num_tokens": 198960254.0, "reward": 2.354736566543579, "reward_std": 0.4973789155483246, "rewards/code_complexity_reward/mean": 0.9307616949081421, "rewards/code_complexity_reward/std": 0.08839736878871918, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1327, "step_time": 48.06764630321413 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 483.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 135.37890625, "completions/mean_terminated_length": 135.37890625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21393248927779496, "epoch": 0.7566951566951567, "frac_reward_zero_std": 0.484375, "grad_norm": 0.07167475670576096, "kl": 0.15309226792305708, "learning_rate": 8.52960337186505e-07, "loss": 0.0007654668297618628, "num_tokens": 199096384.0, "reward": 2.3625001907348633, "reward_std": 0.47748735547065735, "rewards/code_complexity_reward/mean": 0.931445300579071, "rewards/code_complexity_reward/std": 0.0509926974773407, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1328, "step_time": 45.96122302953154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 480.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 130.123046875, "completions/mean_terminated_length": 130.123046875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22804241720587015, "epoch": 0.7572649572649572, "frac_reward_zero_std": 0.46875, "grad_norm": 0.0742696225643158, "kl": 0.1517558548366651, "learning_rate": 8.492216213716139e-07, "loss": 0.000758426554966718, "num_tokens": 199231839.0, "reward": 2.356250047683716, "reward_std": 0.5041898488998413, "rewards/code_complexity_reward/mean": 0.9232421517372131, "rewards/code_complexity_reward/std": 0.10235407203435898, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1329, "step_time": 46.21193824056536 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 138.296875, "completions/mean_terminated_length": 138.296875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2325224478263408, "epoch": 0.7578347578347578, "frac_reward_zero_std": 0.34375, "grad_norm": 0.07816971838474274, "kl": 0.15045095223467797, "learning_rate": 8.454894402335448e-07, "loss": 0.0007518336642533541, "num_tokens": 199373743.0, "reward": 2.2353029251098633, "reward_std": 0.46531352400779724, "rewards/code_complexity_reward/mean": 0.9187499284744263, "rewards/code_complexity_reward/std": 0.12037936598062515, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1330, "step_time": 43.70113446190953 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 132.8515625, "completions/mean_terminated_length": 132.8515625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22375725046731532, "epoch": 0.7584045584045584, "frac_reward_zero_std": 0.34375, "grad_norm": 0.09485563635826111, "kl": 0.15284318884368986, "learning_rate": 8.417638085462979e-07, "loss": 0.0007640032563358545, "num_tokens": 199510251.0, "reward": 2.3096680641174316, "reward_std": 0.48201438784599304, "rewards/code_complexity_reward/mean": 0.9333007335662842, "rewards/code_complexity_reward/std": 0.0969669371843338, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1331, "step_time": 46.004981994628906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 421.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 130.9140625, "completions/mean_terminated_length": 130.9140625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23700680513866246, "epoch": 0.7589743589743589, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06990832090377808, "kl": 0.18087161472067237, "learning_rate": 8.38044741057944e-07, "loss": 0.0009041114244610071, "num_tokens": 199645543.0, "reward": 2.3456544876098633, "reward_std": 0.49365779757499695, "rewards/code_complexity_reward/mean": 0.9245116710662842, "rewards/code_complexity_reward/std": 0.08940855413675308, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1332, "step_time": 58.267636543139815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 124.3515625, "completions/mean_terminated_length": 124.3515625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23821721971035004, "epoch": 0.7595441595441595, "frac_reward_zero_std": 0.328125, "grad_norm": 0.11049162596464157, "kl": 0.1500009475275874, "learning_rate": 8.343322524905726e-07, "loss": 0.0007497046026401222, "num_tokens": 199775291.0, "reward": 2.372607707977295, "reward_std": 0.4880373179912567, "rewards/code_complexity_reward/mean": 0.9345703125, "rewards/code_complexity_reward/std": 0.06632815301418304, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.018299104645848274, "step": 1333, "step_time": 46.65229552797973 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 483.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 130.763671875, "completions/mean_terminated_length": 130.763671875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23290874296799302, "epoch": 0.7601139601139602, "frac_reward_zero_std": 0.4375, "grad_norm": 0.06917916238307953, "kl": 0.15576475753914565, "learning_rate": 8.306263575402276e-07, "loss": 0.0007790968520566821, "num_tokens": 199910810.0, "reward": 2.4183595180511475, "reward_std": 0.5046892762184143, "rewards/code_complexity_reward/mean": 0.9326171875, "rewards/code_complexity_reward/std": 0.07757696509361267, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1334, "step_time": 52.630873404443264 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 122.23828125, "completions/mean_terminated_length": 122.23828125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23668726533651352, "epoch": 0.7606837606837606, "frac_reward_zero_std": 0.484375, "grad_norm": 0.07986224442720413, "kl": 0.21430755965411663, "learning_rate": 8.26927070876852e-07, "loss": 0.0010708037298172712, "num_tokens": 200039604.0, "reward": 2.399169921875, "reward_std": 0.4980495572090149, "rewards/code_complexity_reward/mean": 0.9351562261581421, "rewards/code_complexity_reward/std": 0.07897567749023438, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1335, "step_time": 37.820849916897714 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 417.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 130.5703125, "completions/mean_terminated_length": 130.5703125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24976519215852022, "epoch": 0.7612535612535613, "frac_reward_zero_std": 0.4375, "grad_norm": 0.07038361579179764, "kl": 0.16254559869412333, "learning_rate": 8.232344071442316e-07, "loss": 0.0008125635795295238, "num_tokens": 200176888.0, "reward": 2.2165040969848633, "reward_std": 0.4058115780353546, "rewards/code_complexity_reward/mean": 0.9358398914337158, "rewards/code_complexity_reward/std": 0.08090533316135406, "rewards/code_execution_reward/mean": 0.18359375, "rewards/code_execution_reward/std": 0.3875311613082886, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1336, "step_time": 43.67174978926778 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 128.494140625, "completions/mean_terminated_length": 128.494140625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22129136882722378, "epoch": 0.7618233618233619, "frac_reward_zero_std": 0.453125, "grad_norm": 0.07392911612987518, "kl": 0.19999030127655715, "learning_rate": 8.195483809599325e-07, "loss": 0.0009991888655349612, "num_tokens": 200311605.0, "reward": 2.378222942352295, "reward_std": 0.478292852640152, "rewards/code_complexity_reward/mean": 0.938281238079071, "rewards/code_complexity_reward/std": 0.051106877624988556, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1337, "step_time": 37.014609115198255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 457.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 121.25, "completions/mean_terminated_length": 121.25, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23066318803466856, "epoch": 0.7623931623931623, "frac_reward_zero_std": 0.40625, "grad_norm": 0.09273458272218704, "kl": 0.16183583857491612, "learning_rate": 8.158690069152483e-07, "loss": 0.0008089643088169396, "num_tokens": 200441973.0, "reward": 2.525195360183716, "reward_std": 0.5099498629570007, "rewards/code_complexity_reward/mean": 0.9359374642372131, "rewards/code_complexity_reward/std": 0.05231902748346329, "rewards/code_execution_reward/mean": 0.490234375, "rewards/code_execution_reward/std": 0.5003935098648071, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1338, "step_time": 54.9046966470778 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 130.78125, "completions/mean_terminated_length": 130.78125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23481521173380315, "epoch": 0.762962962962963, "frac_reward_zero_std": 0.4375, "grad_norm": 0.08536630868911743, "kl": 0.1519405940780416, "learning_rate": 8.12196299575137e-07, "loss": 0.0007593645714223385, "num_tokens": 200579405.0, "reward": 2.35205078125, "reward_std": 0.5067698955535889, "rewards/code_complexity_reward/mean": 0.92724609375, "rewards/code_complexity_reward/std": 0.10452030599117279, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 1339, "step_time": 35.16014220472425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 350.0, "completions/max_terminated_length": 350.0, "completions/mean_length": 131.35546875, "completions/mean_terminated_length": 131.35546875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24471340025775135, "epoch": 0.7635327635327636, "frac_reward_zero_std": 0.421875, "grad_norm": 0.08137970417737961, "kl": 0.15577681129798293, "learning_rate": 8.085302734781697e-07, "loss": 0.0007788217626512051, "num_tokens": 200717075.0, "reward": 2.2461915016174316, "reward_std": 0.4983437955379486, "rewards/code_complexity_reward/mean": 0.9157226085662842, "rewards/code_complexity_reward/std": 0.14477820694446564, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1340, "step_time": 47.34769091755152 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 127.220703125, "completions/mean_terminated_length": 127.220703125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23622635984793305, "epoch": 0.764102564102564, "frac_reward_zero_std": 0.390625, "grad_norm": 0.0702262744307518, "kl": 0.14795518957544118, "learning_rate": 8.048709431364654e-07, "loss": 0.00073994230479002, "num_tokens": 200853100.0, "reward": 2.4149904251098633, "reward_std": 0.5419362187385559, "rewards/code_complexity_reward/mean": 0.9296875, "rewards/code_complexity_reward/std": 0.12762890756130219, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1341, "step_time": 38.919562003575265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 128.865234375, "completions/mean_terminated_length": 128.865234375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23354358691722155, "epoch": 0.7646723646723647, "frac_reward_zero_std": 0.40625, "grad_norm": 0.07909784466028214, "kl": 0.15697992220520973, "learning_rate": 8.012183230356416e-07, "loss": 0.0007847717497497797, "num_tokens": 200986303.0, "reward": 2.3703126907348633, "reward_std": 0.5010824799537659, "rewards/code_complexity_reward/mean": 0.9255859851837158, "rewards/code_complexity_reward/std": 0.08769525587558746, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1342, "step_time": 49.78450806066394 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 127.220703125, "completions/mean_terminated_length": 127.220703125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24717209837399423, "epoch": 0.7652421652421653, "frac_reward_zero_std": 0.484375, "grad_norm": 0.12295660376548767, "kl": 0.16468342917505652, "learning_rate": 7.975724276347494e-07, "loss": 0.0008227381040342152, "num_tokens": 201122776.0, "reward": 2.3193359375, "reward_std": 0.5024082064628601, "rewards/code_complexity_reward/mean": 0.9214843511581421, "rewards/code_complexity_reward/std": 0.11372256278991699, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1343, "step_time": 39.1010800562799 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 296.0, "completions/max_terminated_length": 296.0, "completions/mean_length": 129.076171875, "completions/mean_terminated_length": 129.076171875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22601170628331602, "epoch": 0.7658119658119659, "frac_reward_zero_std": 0.34375, "grad_norm": 0.08534343540668488, "kl": 0.16188892140053213, "learning_rate": 7.939332713662229e-07, "loss": 0.0008092124480754137, "num_tokens": 201258647.0, "reward": 2.3477540016174316, "reward_std": 0.4829908311367035, "rewards/code_complexity_reward/mean": 0.934765636920929, "rewards/code_complexity_reward/std": 0.07896309345960617, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1344, "step_time": 41.81485740561038 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 132.97265625, "completions/mean_terminated_length": 132.97265625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2246989356353879, "epoch": 0.7663817663817664, "frac_reward_zero_std": 0.390625, "grad_norm": 0.08842414617538452, "kl": 0.20862933818716556, "learning_rate": 7.903008686358182e-07, "loss": 0.0010420875623822212, "num_tokens": 201392321.0, "reward": 2.3514161109924316, "reward_std": 0.5063145160675049, "rewards/code_complexity_reward/mean": 0.9237304329872131, "rewards/code_complexity_reward/std": 0.10753577202558517, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1345, "step_time": 39.27358408924192 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 131.3671875, "completions/mean_terminated_length": 131.3671875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.24532073200680315, "epoch": 0.766951566951567, "frac_reward_zero_std": 0.421875, "grad_norm": 0.10301309078931808, "kl": 0.17132453224621713, "learning_rate": 7.866752338225564e-07, "loss": 0.0008564341114833951, "num_tokens": 201528373.0, "reward": 2.2845215797424316, "reward_std": 0.5085505843162537, "rewards/code_complexity_reward/mean": 0.9188476800918579, "rewards/code_complexity_reward/std": 0.1361018568277359, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.024042511358857155, "step": 1346, "step_time": 48.62246728781611 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 124.96484375, "completions/mean_terminated_length": 124.96484375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23821373586542904, "epoch": 0.7675213675213676, "frac_reward_zero_std": 0.421875, "grad_norm": 0.09757494181394577, "kl": 0.15495079581160098, "learning_rate": 7.830563812786681e-07, "loss": 0.0007747051422484219, "num_tokens": 201662667.0, "reward": 2.319384813308716, "reward_std": 0.49501076340675354, "rewards/code_complexity_reward/mean": 0.9237304329872131, "rewards/code_complexity_reward/std": 0.11449860781431198, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1347, "step_time": 41.96925927139819 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 130.90625, "completions/mean_terminated_length": 130.16046142578125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2440546047873795, "epoch": 0.7680911680911681, "frac_reward_zero_std": 0.21875, "grad_norm": 0.09098180383443832, "kl": 0.1689040125347674, "learning_rate": 7.794443253295347e-07, "loss": 0.000844151945784688, "num_tokens": 201799547.0, "reward": 2.3707520961761475, "reward_std": 0.54317307472229, "rewards/code_complexity_reward/mean": 0.9201171398162842, "rewards/code_complexity_reward/std": 0.13554120063781738, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1348, "step_time": 48.245107382535934 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 417.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 131.345703125, "completions/mean_terminated_length": 131.345703125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.22982045263051987, "epoch": 0.7686609686609687, "frac_reward_zero_std": 0.390625, "grad_norm": 0.10664571076631546, "kl": 0.15776527486741543, "learning_rate": 7.758390802736359e-07, "loss": 0.0007885504746809602, "num_tokens": 201937220.0, "reward": 2.3168458938598633, "reward_std": 0.48432692885398865, "rewards/code_complexity_reward/mean": 0.9269530773162842, "rewards/code_complexity_reward/std": 0.09496159106492996, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1349, "step_time": 58.65876358933747 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 307.0, "completions/max_terminated_length": 307.0, "completions/mean_length": 129.5703125, "completions/mean_terminated_length": 129.5703125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23556642164476216, "epoch": 0.7692307692307693, "frac_reward_zero_std": 0.46875, "grad_norm": 0.07464877516031265, "kl": 0.16600739862769842, "learning_rate": 7.722406603824873e-07, "loss": 0.0008300664485432208, "num_tokens": 202071504.0, "reward": 2.282519817352295, "reward_std": 0.4756107032299042, "rewards/code_complexity_reward/mean": 0.9247069954872131, "rewards/code_complexity_reward/std": 0.10694985091686249, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1350, "step_time": 34.76487946230918 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 134.056640625, "completions/mean_terminated_length": 134.056640625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23949935426935554, "epoch": 0.7698005698005698, "frac_reward_zero_std": 0.375, "grad_norm": 0.07278486341238022, "kl": 0.167155513423495, "learning_rate": 7.686490799005888e-07, "loss": 0.00083546107634902, "num_tokens": 202209445.0, "reward": 2.3086915016174316, "reward_std": 0.5063758492469788, "rewards/code_complexity_reward/mean": 0.9245116710662842, "rewards/code_complexity_reward/std": 0.12713828682899475, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1351, "step_time": 37.7034911904484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 316.0, "completions/max_terminated_length": 316.0, "completions/mean_length": 125.412109375, "completions/mean_terminated_length": 125.412109375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23403228842653334, "epoch": 0.7703703703703704, "frac_reward_zero_std": 0.34375, "grad_norm": 0.10595783591270447, "kl": 0.1559851833153516, "learning_rate": 7.650643530453644e-07, "loss": 0.0007795093115419149, "num_tokens": 202341768.0, "reward": 2.423877000808716, "reward_std": 0.4966348707675934, "rewards/code_complexity_reward/mean": 0.9393554925918579, "rewards/code_complexity_reward/std": 0.06664283573627472, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1352, "step_time": 36.201515541411936 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 444.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 129.853515625, "completions/mean_terminated_length": 129.853515625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23711298941634595, "epoch": 0.770940170940171, "frac_reward_zero_std": 0.28125, "grad_norm": 0.10694598406553268, "kl": 0.156663624686189, "learning_rate": 7.614864940071098e-07, "loss": 0.0007829307578504086, "num_tokens": 202476493.0, "reward": 2.339794874191284, "reward_std": 0.4909898638725281, "rewards/code_complexity_reward/mean": 0.931445300579071, "rewards/code_complexity_reward/std": 0.09059198945760727, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1353, "step_time": 49.47958576399833 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 359.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 127.15625, "completions/mean_terminated_length": 127.15625, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.24342956161126494, "epoch": 0.7715099715099715, "frac_reward_zero_std": 0.328125, "grad_norm": 0.0974690318107605, "kl": 0.1599023153539747, "learning_rate": 7.579155169489322e-07, "loss": 0.0007991913007572293, "num_tokens": 202609525.0, "reward": 2.326220989227295, "reward_std": 0.4952140748500824, "rewards/code_complexity_reward/mean": 0.9317382574081421, "rewards/code_complexity_reward/std": 0.10595528036355972, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1354, "step_time": 40.00415446329862 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 368.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 125.1875, "completions/mean_terminated_length": 125.1875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22890132083557546, "epoch": 0.7720797720797721, "frac_reward_zero_std": 0.375, "grad_norm": 0.09106729924678802, "kl": 0.16394543170463294, "learning_rate": 7.54351436006697e-07, "loss": 0.0008194723050110042, "num_tokens": 202741701.0, "reward": 2.435302972793579, "reward_std": 0.5240311622619629, "rewards/code_complexity_reward/mean": 0.9294922351837158, "rewards/code_complexity_reward/std": 0.09033171832561493, "rewards/code_execution_reward/mean": 0.41015625, "rewards/code_execution_reward/std": 0.49234291911125183, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1355, "step_time": 49.18550601042807 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 351.0, "completions/max_terminated_length": 351.0, "completions/mean_length": 126.669921875, "completions/mean_terminated_length": 126.669921875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24288675654679537, "epoch": 0.7726495726495727, "frac_reward_zero_std": 0.328125, "grad_norm": 0.10693381726741791, "kl": 0.15118331904523075, "learning_rate": 7.507942652889721e-07, "loss": 0.0007558264769613743, "num_tokens": 202878476.0, "reward": 2.384716749191284, "reward_std": 0.5104122161865234, "rewards/code_complexity_reward/mean": 0.9324219226837158, "rewards/code_complexity_reward/std": 0.09843984991312027, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1356, "step_time": 39.73765183798969 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 444.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 135.947265625, "completions/mean_terminated_length": 135.947265625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2488708165474236, "epoch": 0.7732193732193732, "frac_reward_zero_std": 0.4375, "grad_norm": 0.08376769721508026, "kl": 0.14690287900157273, "learning_rate": 7.472440188769684e-07, "loss": 0.0007343083852902055, "num_tokens": 203019217.0, "reward": 2.2754883766174316, "reward_std": 0.49226874113082886, "rewards/code_complexity_reward/mean": 0.922558605670929, "rewards/code_complexity_reward/std": 0.12803609669208527, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1357, "step_time": 51.31666350364685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 444.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 130.546875, "completions/mean_terminated_length": 130.546875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.24095629435032606, "epoch": 0.7737891737891738, "frac_reward_zero_std": 0.296875, "grad_norm": 0.09483648836612701, "kl": 0.1590696140192449, "learning_rate": 7.437007108244901e-07, "loss": 0.0007953262538649142, "num_tokens": 203155769.0, "reward": 2.291796922683716, "reward_std": 0.48169124126434326, "rewards/code_complexity_reward/mean": 0.9300780892372131, "rewards/code_complexity_reward/std": 0.11012176424264908, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1358, "step_time": 45.16880874801427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 366.0, "completions/max_terminated_length": 366.0, "completions/mean_length": 127.947265625, "completions/mean_terminated_length": 127.947265625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24188018776476383, "epoch": 0.7743589743589744, "frac_reward_zero_std": 0.328125, "grad_norm": 0.09466645866632462, "kl": 0.17340810992754996, "learning_rate": 7.401643551578721e-07, "loss": 0.0008669847738929093, "num_tokens": 203290678.0, "reward": 2.343066692352295, "reward_std": 0.48833581805229187, "rewards/code_complexity_reward/mean": 0.933398425579071, "rewards/code_complexity_reward/std": 0.09455723315477371, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1359, "step_time": 47.36482864059508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 477.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 136.08984375, "completions/mean_terminated_length": 136.08984375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.22959541459567845, "epoch": 0.7749287749287749, "frac_reward_zero_std": 0.390625, "grad_norm": 0.07944485545158386, "kl": 0.14713408472016454, "learning_rate": 7.366349658759306e-07, "loss": 0.0007355231791734695, "num_tokens": 203430156.0, "reward": 2.3521971702575684, "reward_std": 0.49702274799346924, "rewards/code_complexity_reward/mean": 0.9264647960662842, "rewards/code_complexity_reward/std": 0.09466080367565155, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1360, "step_time": 56.04449054412544 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 128.828125, "completions/mean_terminated_length": 128.07827758789062, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23091342556290329, "epoch": 0.7754985754985755, "frac_reward_zero_std": 0.296875, "grad_norm": 0.11252162605524063, "kl": 0.17830100236460567, "learning_rate": 7.331125569499026e-07, "loss": 0.0008914158097468317, "num_tokens": 203562020.0, "reward": 2.4397950172424316, "reward_std": 0.5217195749282837, "rewards/code_complexity_reward/mean": 0.935742199420929, "rewards/code_complexity_reward/std": 0.08958392590284348, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1361, "step_time": 53.08860755991191 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 422.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 133.169921875, "completions/mean_terminated_length": 133.169921875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23233612230978906, "epoch": 0.7760683760683761, "frac_reward_zero_std": 0.28125, "grad_norm": 0.11555681377649307, "kl": 0.14894732320681214, "learning_rate": 7.295971423233961e-07, "loss": 0.0007442560745403171, "num_tokens": 203697619.0, "reward": 2.3637208938598633, "reward_std": 0.5204604864120483, "rewards/code_complexity_reward/mean": 0.92333984375, "rewards/code_complexity_reward/std": 0.11823546886444092, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1362, "step_time": 53.365480767562985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 131.873046875, "completions/mean_terminated_length": 131.873046875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.22926082368940115, "epoch": 0.7766381766381767, "frac_reward_zero_std": 0.28125, "grad_norm": 0.12756876647472382, "kl": 0.1442476405063644, "learning_rate": 7.260887359123286e-07, "loss": 0.0007210713811218739, "num_tokens": 203835410.0, "reward": 2.370410442352295, "reward_std": 0.4986168444156647, "rewards/code_complexity_reward/mean": 0.9344726800918579, "rewards/code_complexity_reward/std": 0.09048745781183243, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1363, "step_time": 41.67700550891459 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 377.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 134.8515625, "completions/mean_terminated_length": 134.8515625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2291897472459823, "epoch": 0.7772079772079772, "frac_reward_zero_std": 0.265625, "grad_norm": 0.1066921129822731, "kl": 0.17629298416431993, "learning_rate": 7.225873516048781e-07, "loss": 0.0008812793530523777, "num_tokens": 203972302.0, "reward": 2.3502931594848633, "reward_std": 0.5059866309165955, "rewards/code_complexity_reward/mean": 0.931933581829071, "rewards/code_complexity_reward/std": 0.10772859305143356, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1364, "step_time": 55.39672186318785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 351.0, "completions/max_terminated_length": 351.0, "completions/mean_length": 130.44921875, "completions/mean_terminated_length": 130.44921875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24479752383194864, "epoch": 0.7777777777777778, "frac_reward_zero_std": 0.34375, "grad_norm": 0.1193513348698616, "kl": 0.1585262274602428, "learning_rate": 7.190930032614249e-07, "loss": 0.0007918608607724309, "num_tokens": 204107340.0, "reward": 2.280810832977295, "reward_std": 0.4727083742618561, "rewards/code_complexity_reward/mean": 0.9369140863418579, "rewards/code_complexity_reward/std": 0.1070629432797432, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1365, "step_time": 46.42843665368855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 269.0, "completions/max_terminated_length": 269.0, "completions/mean_length": 123.359375, "completions/mean_terminated_length": 123.359375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23529532505199313, "epoch": 0.7783475783475784, "frac_reward_zero_std": 0.3125, "grad_norm": 0.10425852239131927, "kl": 0.1644717528251931, "learning_rate": 7.156057047144945e-07, "loss": 0.0008220277377404273, "num_tokens": 204239428.0, "reward": 2.3880860805511475, "reward_std": 0.5004544258117676, "rewards/code_complexity_reward/mean": 0.9365234375, "rewards/code_complexity_reward/std": 0.08970502018928528, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1366, "step_time": 34.844109300523996 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 134.50390625, "completions/mean_terminated_length": 133.76516723632812, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23895900277420878, "epoch": 0.7789173789173789, "frac_reward_zero_std": 0.15625, "grad_norm": 0.15800222754478455, "kl": 0.1778161956463009, "learning_rate": 7.121254697687094e-07, "loss": 0.0008886682335287333, "num_tokens": 204377622.0, "reward": 2.326416015625, "reward_std": 0.5078871250152588, "rewards/code_complexity_reward/mean": 0.9314453601837158, "rewards/code_complexity_reward/std": 0.1203283965587616, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1367, "step_time": 67.2530844276771 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 125.78125, "completions/mean_terminated_length": 125.78125, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.23736951244063675, "epoch": 0.7794871794871795, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1505139172077179, "kl": 0.16379393124952912, "learning_rate": 7.086523122007264e-07, "loss": 0.000818649132270366, "num_tokens": 204510478.0, "reward": 2.2821779251098633, "reward_std": 0.4609835743904114, "rewards/code_complexity_reward/mean": 0.9423828125, "rewards/code_complexity_reward/std": 0.09077069908380508, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1368, "step_time": 36.03529995586723 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 450.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 126.861328125, "completions/mean_terminated_length": 126.861328125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2481027354951948, "epoch": 0.7800569800569801, "frac_reward_zero_std": 0.265625, "grad_norm": 0.1124832034111023, "kl": 0.15716123499441892, "learning_rate": 7.051862457591901e-07, "loss": 0.0007855295552872121, "num_tokens": 204649335.0, "reward": 2.432373046875, "reward_std": 0.5411022305488586, "rewards/code_complexity_reward/mean": 0.9234374761581421, "rewards/code_complexity_reward/std": 0.12743709981441498, "rewards/code_execution_reward/mean": 0.41796875, "rewards/code_execution_reward/std": 0.4937073290348053, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 1369, "step_time": 44.721963888034225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 372.0, "completions/max_terminated_length": 372.0, "completions/mean_length": 130.99609375, "completions/mean_terminated_length": 130.99609375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.25147180794738233, "epoch": 0.7806267806267806, "frac_reward_zero_std": 0.234375, "grad_norm": 0.12124606966972351, "kl": 0.15692265599500388, "learning_rate": 7.017272841646713e-07, "loss": 0.0007845730287954211, "num_tokens": 204783669.0, "reward": 2.353808641433716, "reward_std": 0.5319789052009583, "rewards/code_complexity_reward/mean": 0.9271484613418579, "rewards/code_complexity_reward/std": 0.1308383047580719, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1370, "step_time": 48.2721552234143 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 440.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 126.6484375, "completions/mean_terminated_length": 126.6484375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2338775061070919, "epoch": 0.7811965811965812, "frac_reward_zero_std": 0.25, "grad_norm": 0.09410248696804047, "kl": 0.17168120841961354, "learning_rate": 6.98275441109619e-07, "loss": 0.0008577851112931967, "num_tokens": 204916505.0, "reward": 2.4040040969848633, "reward_std": 0.49759557843208313, "rewards/code_complexity_reward/mean": 0.9407227039337158, "rewards/code_complexity_reward/std": 0.06914961338043213, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1371, "step_time": 45.215542173944414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 128.50390625, "completions/mean_terminated_length": 127.75342559814453, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24150552391074598, "epoch": 0.7817663817663818, "frac_reward_zero_std": 0.203125, "grad_norm": 0.10776101052761078, "kl": 0.17467690049670637, "learning_rate": 6.948307302583005e-07, "loss": 0.0008737185853533447, "num_tokens": 205050035.0, "reward": 2.3857421875, "reward_std": 0.5063436031341553, "rewards/code_complexity_reward/mean": 0.938769519329071, "rewards/code_complexity_reward/std": 0.0957811176776886, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1372, "step_time": 48.10905635450035 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 379.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 126.80078125, "completions/mean_terminated_length": 126.80078125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23698057187721133, "epoch": 0.7823361823361823, "frac_reward_zero_std": 0.265625, "grad_norm": 0.13457998633384705, "kl": 0.1619843103690073, "learning_rate": 6.913931652467514e-07, "loss": 0.0008093587821349502, "num_tokens": 205183389.0, "reward": 2.3121094703674316, "reward_std": 0.49421292543411255, "rewards/code_complexity_reward/mean": 0.928906261920929, "rewards/code_complexity_reward/std": 0.11327872425317764, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1373, "step_time": 46.72902894113213 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 299.0, "completions/max_terminated_length": 299.0, "completions/mean_length": 119.935546875, "completions/mean_terminated_length": 119.935546875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24568913108669221, "epoch": 0.7829059829059829, "frac_reward_zero_std": 0.34375, "grad_norm": 0.14581969380378723, "kl": 0.18519306101370603, "learning_rate": 6.879627596827187e-07, "loss": 0.0009257474448531866, "num_tokens": 205312036.0, "reward": 2.3270509243011475, "reward_std": 0.4700373411178589, "rewards/code_complexity_reward/mean": 0.94580078125, "rewards/code_complexity_reward/std": 0.06836748868227005, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1374, "step_time": 35.82590982504189 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 376.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 126.548828125, "completions/mean_terminated_length": 126.548828125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23394376947544515, "epoch": 0.7834757834757835, "frac_reward_zero_std": 0.328125, "grad_norm": 0.09699789434671402, "kl": 0.16560083150397986, "learning_rate": 6.84539527145611e-07, "loss": 0.0008282714406959713, "num_tokens": 205447549.0, "reward": 2.43408203125, "reward_std": 0.5103292465209961, "rewards/code_complexity_reward/mean": 0.9444335699081421, "rewards/code_complexity_reward/std": 0.08032024651765823, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1375, "step_time": 44.30190247949213 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 455.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 132.2109375, "completions/mean_terminated_length": 132.2109375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2504870081320405, "epoch": 0.784045584045584, "frac_reward_zero_std": 0.265625, "grad_norm": 0.11742883920669556, "kl": 0.17132669035345316, "learning_rate": 6.811234811864406e-07, "loss": 0.0008562306757085025, "num_tokens": 205586081.0, "reward": 2.4132814407348633, "reward_std": 0.5142484307289124, "rewards/code_complexity_reward/mean": 0.941210925579071, "rewards/code_complexity_reward/std": 0.10150963813066483, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1376, "step_time": 52.459297084249556 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 123.8046875, "completions/mean_terminated_length": 123.8046875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2455221228301525, "epoch": 0.7846153846153846, "frac_reward_zero_std": 0.203125, "grad_norm": 0.14500971138477325, "kl": 0.17660518735647202, "learning_rate": 6.777146353277705e-07, "loss": 0.000883155211340636, "num_tokens": 205719917.0, "reward": 2.4274415969848633, "reward_std": 0.5267863273620605, "rewards/code_complexity_reward/mean": 0.9403319954872131, "rewards/code_complexity_reward/std": 0.11160317808389664, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 1377, "step_time": 40.39629040006548 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 331.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 125.845703125, "completions/mean_terminated_length": 125.845703125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23156906245276332, "epoch": 0.7851851851851852, "frac_reward_zero_std": 0.25, "grad_norm": 0.16249093413352966, "kl": 0.20397758181206882, "learning_rate": 6.743130030636647e-07, "loss": 0.0010193688794970512, "num_tokens": 205853774.0, "reward": 2.388916015625, "reward_std": 0.5541741847991943, "rewards/code_complexity_reward/mean": 0.929003894329071, "rewards/code_complexity_reward/std": 0.1431186944246292, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1378, "step_time": 37.527023267000914 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 305.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 127.767578125, "completions/mean_terminated_length": 127.767578125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2515443339943886, "epoch": 0.7857549857549857, "frac_reward_zero_std": 0.25, "grad_norm": 0.12960302829742432, "kl": 0.1765378408599645, "learning_rate": 6.709185978596277e-07, "loss": 0.000882613705471158, "num_tokens": 205986455.0, "reward": 2.3349609375, "reward_std": 0.5094683766365051, "rewards/code_complexity_reward/mean": 0.9390624761581421, "rewards/code_complexity_reward/std": 0.11576977372169495, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1379, "step_time": 34.97214168123901 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 351.0, "completions/mean_length": 125.880859375, "completions/mean_terminated_length": 125.125244140625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2416248070076108, "epoch": 0.7863247863247863, "frac_reward_zero_std": 0.3125, "grad_norm": 0.12116542458534241, "kl": 0.17825332866050303, "learning_rate": 6.675314331525598e-07, "loss": 0.0008908403106033802, "num_tokens": 206120130.0, "reward": 2.34228515625, "reward_std": 0.5397248268127441, "rewards/code_complexity_reward/mean": 0.924609363079071, "rewards/code_complexity_reward/std": 0.14781461656093597, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1380, "step_time": 48.75426159892231 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 118.6640625, "completions/mean_terminated_length": 118.6640625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24729960155673325, "epoch": 0.7868945868945869, "frac_reward_zero_std": 0.203125, "grad_norm": 0.12815448641777039, "kl": 0.16992846166249365, "learning_rate": 6.641515223506958e-07, "loss": 0.0008492381894029677, "num_tokens": 206249654.0, "reward": 2.356494188308716, "reward_std": 0.5120400190353394, "rewards/code_complexity_reward/mean": 0.9415038824081421, "rewards/code_complexity_reward/std": 0.11608032137155533, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1381, "step_time": 35.11665871180594 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 129.884765625, "completions/mean_terminated_length": 129.884765625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23347092443145812, "epoch": 0.7874643874643875, "frac_reward_zero_std": 0.296875, "grad_norm": 0.11574411392211914, "kl": 0.16855268855579197, "learning_rate": 6.607788788335584e-07, "loss": 0.000842598092276603, "num_tokens": 206384563.0, "reward": 2.3705077171325684, "reward_std": 0.4958702325820923, "rewards/code_complexity_reward/mean": 0.9413085579872131, "rewards/code_complexity_reward/std": 0.09075485914945602, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1382, "step_time": 60.68724521622062 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 494.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 125.458984375, "completions/mean_terminated_length": 125.458984375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2373220433946699, "epoch": 0.788034188034188, "frac_reward_zero_std": 0.171875, "grad_norm": 0.1619648039340973, "kl": 0.1620894728694111, "learning_rate": 6.574135159519001e-07, "loss": 0.0008102376013994217, "num_tokens": 206519622.0, "reward": 2.4100098609924316, "reward_std": 0.5316027402877808, "rewards/code_complexity_reward/mean": 0.9369140863418579, "rewards/code_complexity_reward/std": 0.11655797809362411, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 1383, "step_time": 48.04237446933985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 335.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 122.609375, "completions/mean_terminated_length": 122.609375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2558944299817085, "epoch": 0.7886039886039886, "frac_reward_zero_std": 0.265625, "grad_norm": 0.14828693866729736, "kl": 0.19113931711763144, "learning_rate": 6.54055447027655e-07, "loss": 0.000955477706156671, "num_tokens": 206652358.0, "reward": 2.3096680641174316, "reward_std": 0.4680410325527191, "rewards/code_complexity_reward/mean": 0.9445312023162842, "rewards/code_complexity_reward/std": 0.09097953140735626, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1384, "step_time": 38.10325957927853 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 300.0, "completions/max_terminated_length": 300.0, "completions/mean_length": 128.787109375, "completions/mean_terminated_length": 128.787109375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24644636968150735, "epoch": 0.7891737891737892, "frac_reward_zero_std": 0.15625, "grad_norm": 0.16130320727825165, "kl": 0.1701688232133165, "learning_rate": 6.50704685353882e-07, "loss": 0.0008508928585797548, "num_tokens": 206784241.0, "reward": 2.3272461891174316, "reward_std": 0.4756649434566498, "rewards/code_complexity_reward/mean": 0.9450194835662842, "rewards/code_complexity_reward/std": 0.0834646001458168, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1385, "step_time": 33.89709156099707 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 123.109375, "completions/mean_terminated_length": 123.109375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2477635128889233, "epoch": 0.7897435897435897, "frac_reward_zero_std": 0.125, "grad_norm": 0.1575814038515091, "kl": 0.1683810637332499, "learning_rate": 6.473612441947139e-07, "loss": 0.000841453205794096, "num_tokens": 206916905.0, "reward": 2.4122071266174316, "reward_std": 0.5400721430778503, "rewards/code_complexity_reward/mean": 0.939746081829071, "rewards/code_complexity_reward/std": 0.13174031674861908, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 1386, "step_time": 45.336782679893076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 124.625, "completions/mean_terminated_length": 124.625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22708607488311827, "epoch": 0.7903133903133903, "frac_reward_zero_std": 0.21875, "grad_norm": 0.12962578237056732, "kl": 0.1714634143281728, "learning_rate": 6.440251367853065e-07, "loss": 0.000856948783621192, "num_tokens": 207048889.0, "reward": 2.435546875, "reward_std": 0.5160402059555054, "rewards/code_complexity_reward/mean": 0.9439452886581421, "rewards/code_complexity_reward/std": 0.08353996276855469, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1387, "step_time": 43.04962013568729 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 129.572265625, "completions/mean_terminated_length": 129.572265625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24873107811436057, "epoch": 0.7908831908831909, "frac_reward_zero_std": 0.203125, "grad_norm": 0.13078418374061584, "kl": 0.17314086749684066, "learning_rate": 6.406963763317827e-07, "loss": 0.0008652870892547071, "num_tokens": 207185646.0, "reward": 2.3197264671325684, "reward_std": 0.5434916019439697, "rewards/code_complexity_reward/mean": 0.9267578125, "rewards/code_complexity_reward/std": 0.16141144931316376, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1388, "step_time": 53.752974493429065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 124.94921875, "completions/mean_terminated_length": 124.94921875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23606185032986104, "epoch": 0.7914529914529914, "frac_reward_zero_std": 0.265625, "grad_norm": 0.15744668245315552, "kl": 0.18683443823829293, "learning_rate": 6.373749760111847e-07, "loss": 0.0009339408134110272, "num_tokens": 207320068.0, "reward": 2.3353514671325684, "reward_std": 0.48951858282089233, "rewards/code_complexity_reward/mean": 0.9384765625, "rewards/code_complexity_reward/std": 0.10157287120819092, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 1389, "step_time": 48.30557696707547 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 338.0, "completions/max_terminated_length": 338.0, "completions/mean_length": 126.783203125, "completions/mean_terminated_length": 126.783203125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23686812072992325, "epoch": 0.792022792022792, "frac_reward_zero_std": 0.1875, "grad_norm": 0.11450877785682678, "kl": 0.1822750607971102, "learning_rate": 6.340609489714158e-07, "loss": 0.000911697163246572, "num_tokens": 207452813.0, "reward": 2.3468751907348633, "reward_std": 0.5188803672790527, "rewards/code_complexity_reward/mean": 0.938281238079071, "rewards/code_complexity_reward/std": 0.12461045384407043, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1390, "step_time": 46.27340752445161 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 136.59375, "completions/mean_terminated_length": 135.85910034179688, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.25444304570555687, "epoch": 0.7925925925925926, "frac_reward_zero_std": 0.203125, "grad_norm": 0.14370916783809662, "kl": 0.1670980271883309, "learning_rate": 6.307543083311956e-07, "loss": 0.0008348668343387544, "num_tokens": 207592597.0, "reward": 2.279589891433716, "reward_std": 0.49643635749816895, "rewards/code_complexity_reward/mean": 0.9302734136581421, "rewards/code_complexity_reward/std": 0.12856091558933258, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1391, "step_time": 48.60567694623023 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 282.0, "completions/max_terminated_length": 282.0, "completions/mean_length": 128.15625, "completions/mean_terminated_length": 128.15625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23220128379762173, "epoch": 0.7931623931623931, "frac_reward_zero_std": 0.265625, "grad_norm": 0.10715832561254501, "kl": 0.18119661381933838, "learning_rate": 6.274550671800008e-07, "loss": 0.0009057644056156278, "num_tokens": 207726853.0, "reward": 2.3738770484924316, "reward_std": 0.5183921456336975, "rewards/code_complexity_reward/mean": 0.9416015148162842, "rewards/code_complexity_reward/std": 0.11589764058589935, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 1392, "step_time": 51.972446873784065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 401.0, "completions/max_terminated_length": 401.0, "completions/mean_length": 124.404296875, "completions/mean_terminated_length": 124.404296875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24265906331129372, "epoch": 0.7937321937321937, "frac_reward_zero_std": 0.15625, "grad_norm": 0.12554103136062622, "kl": 0.18092764168977737, "learning_rate": 6.241632385780205e-07, "loss": 0.0009040176519192755, "num_tokens": 207861092.0, "reward": 2.3023927211761475, "reward_std": 0.4765595495700836, "rewards/code_complexity_reward/mean": 0.9440429210662842, "rewards/code_complexity_reward/std": 0.10611749440431595, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1393, "step_time": 42.46260242443532 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 133.90234375, "completions/mean_terminated_length": 133.90234375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24883330636657774, "epoch": 0.7943019943019943, "frac_reward_zero_std": 0.21875, "grad_norm": 0.12498205900192261, "kl": 0.16903148859273642, "learning_rate": 6.208788355560971e-07, "loss": 0.0008449675515294075, "num_tokens": 207997146.0, "reward": 2.2293457984924316, "reward_std": 0.41808950901031494, "rewards/code_complexity_reward/mean": 0.938183605670929, "rewards/code_complexity_reward/std": 0.09163668006658554, "rewards/code_execution_reward/mean": 0.1953125, "rewards/code_execution_reward/std": 0.3968288004398346, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1394, "step_time": 38.799842909909785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 130.28125, "completions/mean_terminated_length": 129.53424072265625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2436610832810402, "epoch": 0.7948717948717948, "frac_reward_zero_std": 0.140625, "grad_norm": 0.1698130965232849, "kl": 0.1952692198101431, "learning_rate": 6.17601871115682e-07, "loss": 0.0009757521911524236, "num_tokens": 208134818.0, "reward": 2.3427248001098633, "reward_std": 0.5279766321182251, "rewards/code_complexity_reward/mean": 0.9343749284744263, "rewards/code_complexity_reward/std": 0.13999581336975098, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1395, "step_time": 48.78012890089303 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 134.111328125, "completions/mean_terminated_length": 134.111328125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23381877434439957, "epoch": 0.7954415954415954, "frac_reward_zero_std": 0.234375, "grad_norm": 0.14029929041862488, "kl": 0.1608183440985158, "learning_rate": 6.14332358228778e-07, "loss": 0.0008038539672270417, "num_tokens": 208275459.0, "reward": 2.3837890625, "reward_std": 0.503452718257904, "rewards/code_complexity_reward/mean": 0.9400390386581421, "rewards/code_complexity_reward/std": 0.09398962557315826, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1396, "step_time": 45.93909010011703 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 126.1796875, "completions/mean_terminated_length": 126.1796875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24368947907350957, "epoch": 0.796011396011396, "frac_reward_zero_std": 0.21875, "grad_norm": 0.17555493116378784, "kl": 0.18235262972302735, "learning_rate": 6.110703098378929e-07, "loss": 0.0009115057182498276, "num_tokens": 208407695.0, "reward": 2.3373048305511475, "reward_std": 0.49846628308296204, "rewards/code_complexity_reward/mean": 0.9443359375, "rewards/code_complexity_reward/std": 0.10937980562448502, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1397, "step_time": 47.70972139574587 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 416.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 128.552734375, "completions/mean_terminated_length": 128.552734375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23899316950701177, "epoch": 0.7965811965811965, "frac_reward_zero_std": 0.140625, "grad_norm": 0.11912088096141815, "kl": 0.17989498295355588, "learning_rate": 6.078157388559838e-07, "loss": 0.0008994050440378487, "num_tokens": 208542930.0, "reward": 2.33935546875, "reward_std": 0.517302393913269, "rewards/code_complexity_reward/mean": 0.9346679449081421, "rewards/code_complexity_reward/std": 0.12738636136054993, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1398, "step_time": 41.952698377892375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 305.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 129.068359375, "completions/mean_terminated_length": 129.068359375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24725812044925988, "epoch": 0.7971509971509971, "frac_reward_zero_std": 0.203125, "grad_norm": 0.13210144639015198, "kl": 0.23499450297094882, "learning_rate": 6.04568658166409e-07, "loss": 0.0011743338545784354, "num_tokens": 208679293.0, "reward": 2.2365236282348633, "reward_std": 0.47574639320373535, "rewards/code_complexity_reward/mean": 0.930468738079071, "rewards/code_complexity_reward/std": 0.14317265152931213, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1399, "step_time": 42.323733270168304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 301.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 123.046875, "completions/mean_terminated_length": 123.046875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23867104784585536, "epoch": 0.7977207977207977, "frac_reward_zero_std": 0.1875, "grad_norm": 0.136953666806221, "kl": 0.174345841165632, "learning_rate": 6.013290806228775e-07, "loss": 0.0008717669988982379, "num_tokens": 208809821.0, "reward": 2.4286623001098633, "reward_std": 0.5133920907974243, "rewards/code_complexity_reward/mean": 0.9453125, "rewards/code_complexity_reward/std": 0.08857150375843048, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1400, "step_time": 35.13003207091242 }, { "epoch": 0.7977207977207977, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.00125, "eval_completions/max_length": 171.57, "eval_completions/max_terminated_length": 171.44, "eval_completions/mean_length": 124.735, "eval_completions/mean_terminated_length": 124.44785736083985, "eval_completions/min_length": 91.16, "eval_completions/min_terminated_length": 91.16, "eval_entropy": 0.24000942170619965, "eval_frac_reward_zero_std": 0.14, "eval_kl": 0.1888293530791998, "eval_loss": 0.002719376003369689, "eval_num_tokens": 208809821.0, "eval_reward": 2.3414375245571137, "eval_reward_std": 0.25052721675485373, "eval_rewards/code_complexity_reward/mean": 0.944062493443489, "eval_rewards/code_complexity_reward/std": 0.060391366705298426, "eval_rewards/code_execution_reward/mean": 0.30625, "eval_rewards/code_execution_reward/std": 0.1868927437067032, "eval_rewards/code_syntax_reward/mean": 0.491875, "eval_rewards/code_syntax_reward/std": 0.02175998643040657, "eval_rewards/reasoning_present_reward_func/mean": 0.0998750015348196, "eval_rewards/reasoning_present_reward_func/std": 0.000353553406894207, "eval_rewards/xmlcount_reward_func/mean": 0.499375, "eval_rewards/xmlcount_reward_func/std": 0.001767766922712326, "eval_runtime": 792.1798, "eval_samples_per_second": 0.126, "eval_steps_per_second": 0.016, "step": 1400 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 135.25, "completions/mean_terminated_length": 134.51272583007812, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23588887741789222, "epoch": 0.7982905982905983, "frac_reward_zero_std": 0.1875, "grad_norm": 0.13469301164150238, "kl": 0.18771613016724586, "learning_rate": 5.98097019049394e-07, "loss": 0.0009381215204484761, "num_tokens": 208950461.0, "reward": 2.365478515625, "reward_std": 0.5430542826652527, "rewards/code_complexity_reward/mean": 0.9268554449081421, "rewards/code_complexity_reward/std": 0.14806802570819855, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1401, "step_time": 63.03617244865745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 341.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 126.013671875, "completions/mean_terminated_length": 126.013671875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2556341914460063, "epoch": 0.7988603988603988, "frac_reward_zero_std": 0.171875, "grad_norm": 0.14428795874118805, "kl": 0.2185524026863277, "learning_rate": 5.948724862402138e-07, "loss": 0.0010917489416897297, "num_tokens": 209086964.0, "reward": 2.280810594558716, "reward_std": 0.4995439946651459, "rewards/code_complexity_reward/mean": 0.9419921636581421, "rewards/code_complexity_reward/std": 0.14492164552211761, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1402, "step_time": 37.300300226546824 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 124.984375, "completions/mean_terminated_length": 124.22700500488281, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.239622687920928, "epoch": 0.7994301994301994, "frac_reward_zero_std": 0.140625, "grad_norm": 0.14381136000156403, "kl": 0.19126286334358156, "learning_rate": 5.916554949597872e-07, "loss": 0.0009553870186209679, "num_tokens": 209220484.0, "reward": 2.419189453125, "reward_std": 0.5306175351142883, "rewards/code_complexity_reward/mean": 0.9473632574081421, "rewards/code_complexity_reward/std": 0.11686494946479797, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1403, "step_time": 47.424778369255364 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 487.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 130.654296875, "completions/mean_terminated_length": 130.654296875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2578676366247237, "epoch": 0.8, "frac_reward_zero_std": 0.140625, "grad_norm": 0.12195640802383423, "kl": 0.1867094491608441, "learning_rate": 5.884460579427117e-07, "loss": 0.0009335750946775079, "num_tokens": 209355707.0, "reward": 2.2596678733825684, "reward_std": 0.5215129852294922, "rewards/code_complexity_reward/mean": 0.93017578125, "rewards/code_complexity_reward/std": 0.16952958703041077, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1404, "step_time": 63.74823062866926 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 473.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 121.115234375, "completions/mean_terminated_length": 121.115234375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2331022631842643, "epoch": 0.8005698005698005, "frac_reward_zero_std": 0.15625, "grad_norm": 0.20623500645160675, "kl": 0.21156394435092807, "learning_rate": 5.852441878936821e-07, "loss": 0.0010576082859188318, "num_tokens": 209485030.0, "reward": 2.390429735183716, "reward_std": 0.49608662724494934, "rewards/code_complexity_reward/mean": 0.9574218392372131, "rewards/code_complexity_reward/std": 0.08236709237098694, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1405, "step_time": 49.9834602130577 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 122.033203125, "completions/mean_terminated_length": 121.27005767822266, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2497465005144477, "epoch": 0.8011396011396011, "frac_reward_zero_std": 0.1875, "grad_norm": 0.1386420875787735, "kl": 0.1905490409117192, "learning_rate": 5.820498974874364e-07, "loss": 0.000952486414462328, "num_tokens": 209615999.0, "reward": 2.2916994094848633, "reward_std": 0.5046694278717041, "rewards/code_complexity_reward/mean": 0.94287109375, "rewards/code_complexity_reward/std": 0.14004108309745789, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 1406, "step_time": 50.48064946569502 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 124.05078125, "completions/mean_terminated_length": 124.05078125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2460538949817419, "epoch": 0.8017094017094017, "frac_reward_zero_std": 0.296875, "grad_norm": 0.13470439612865448, "kl": 0.19705666485242546, "learning_rate": 5.788631993687116e-07, "loss": 0.0009851076174527407, "num_tokens": 209748177.0, "reward": 2.2452149391174316, "reward_std": 0.48006394505500793, "rewards/code_complexity_reward/mean": 0.934277355670929, "rewards/code_complexity_reward/std": 0.1416352093219757, "rewards/code_execution_reward/mean": 0.220703125, "rewards/code_execution_reward/std": 0.4151262938976288, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1407, "step_time": 40.530794966965914 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 128.974609375, "completions/mean_terminated_length": 128.974609375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2439562191721052, "epoch": 0.8022792022792022, "frac_reward_zero_std": 0.171875, "grad_norm": 0.13987907767295837, "kl": 0.19882008619606495, "learning_rate": 5.756841061521873e-07, "loss": 0.000994008150883019, "num_tokens": 209881452.0, "reward": 2.2740724086761475, "reward_std": 0.4867599904537201, "rewards/code_complexity_reward/mean": 0.9352538585662842, "rewards/code_complexity_reward/std": 0.13305214047431946, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1408, "step_time": 33.860126953572035 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 329.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 125.6484375, "completions/mean_terminated_length": 125.6484375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23441996332257986, "epoch": 0.8028490028490028, "frac_reward_zero_std": 0.25, "grad_norm": 0.11617230623960495, "kl": 0.2017365712672472, "learning_rate": 5.725126304224396e-07, "loss": 0.0010085590183734894, "num_tokens": 210013648.0, "reward": 2.395800828933716, "reward_std": 0.559727132320404, "rewards/code_complexity_reward/mean": 0.9383789300918579, "rewards/code_complexity_reward/std": 0.1516275405883789, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1409, "step_time": 36.74479923862964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 451.0, "completions/mean_length": 125.666015625, "completions/mean_terminated_length": 124.90998077392578, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.25750140217132866, "epoch": 0.8034188034188035, "frac_reward_zero_std": 0.0625, "grad_norm": 0.16673393547534943, "kl": 0.19691187096759677, "learning_rate": 5.693487847338918e-07, "loss": 0.0009839350823312998, "num_tokens": 210145469.0, "reward": 2.322998046875, "reward_std": 0.4989246428012848, "rewards/code_complexity_reward/mean": 0.944628894329071, "rewards/code_complexity_reward/std": 0.11806271970272064, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.024042511358857155, "step": 1410, "step_time": 48.989634059369564 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 123.9609375, "completions/mean_terminated_length": 123.9609375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2455924868118018, "epoch": 0.803988603988604, "frac_reward_zero_std": 0.21875, "grad_norm": 0.13595233857631683, "kl": 0.22363557177595794, "learning_rate": 5.661925816107625e-07, "loss": 0.0011183323804289103, "num_tokens": 210277177.0, "reward": 2.3219237327575684, "reward_std": 0.49206429719924927, "rewards/code_complexity_reward/mean": 0.9518554210662842, "rewards/code_complexity_reward/std": 0.11750619113445282, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1411, "step_time": 47.14461905416101 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 234.0, "completions/max_terminated_length": 234.0, "completions/mean_length": 126.091796875, "completions/mean_terminated_length": 126.091796875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2370060202665627, "epoch": 0.8045584045584045, "frac_reward_zero_std": 0.21875, "grad_norm": 0.1184019073843956, "kl": 0.1756566094700247, "learning_rate": 5.630440335470156e-07, "loss": 0.000878864258993417, "num_tokens": 210409720.0, "reward": 2.42822265625, "reward_std": 0.5291681289672852, "rewards/code_complexity_reward/mean": 0.9454101920127869, "rewards/code_complexity_reward/std": 0.11034294217824936, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1412, "step_time": 57.477369382977486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 243.0, "completions/max_terminated_length": 243.0, "completions/mean_length": 119.59765625, "completions/mean_terminated_length": 119.59765625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.25227220077067614, "epoch": 0.8051282051282052, "frac_reward_zero_std": 0.109375, "grad_norm": 0.20397330820560455, "kl": 0.2250297055579722, "learning_rate": 5.599031530063142e-07, "loss": 0.001126056769862771, "num_tokens": 210540850.0, "reward": 2.370410203933716, "reward_std": 0.5017273426055908, "rewards/code_complexity_reward/mean": 0.9530273675918579, "rewards/code_complexity_reward/std": 0.10367803275585175, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1413, "step_time": 32.740839172154665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 330.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 123.595703125, "completions/mean_terminated_length": 123.595703125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24627217883244157, "epoch": 0.8056980056980056, "frac_reward_zero_std": 0.1875, "grad_norm": 0.14976805448532104, "kl": 0.19262204715050757, "learning_rate": 5.567699524219677e-07, "loss": 0.0009630921995267272, "num_tokens": 210673691.0, "reward": 2.3557615280151367, "reward_std": 0.544084906578064, "rewards/code_complexity_reward/mean": 0.9413086175918579, "rewards/code_complexity_reward/std": 0.151952862739563, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1414, "step_time": 36.594447313807905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 429.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 125.001953125, "completions/mean_terminated_length": 125.001953125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23817671393044293, "epoch": 0.8062678062678063, "frac_reward_zero_std": 0.109375, "grad_norm": 0.190715029835701, "kl": 0.20291037391871214, "learning_rate": 5.536444441968849e-07, "loss": 0.0010145124979317188, "num_tokens": 210806924.0, "reward": 2.30712890625, "reward_std": 0.4761256277561188, "rewards/code_complexity_reward/mean": 0.9532226324081421, "rewards/code_complexity_reward/std": 0.11180062592029572, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1415, "step_time": 52.70430488232523 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 328.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 122.802734375, "completions/mean_terminated_length": 122.802734375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2406777839642018, "epoch": 0.8068376068376069, "frac_reward_zero_std": 0.171875, "grad_norm": 0.127985879778862, "kl": 0.2023830998223275, "learning_rate": 5.505266407035245e-07, "loss": 0.0010113287717103958, "num_tokens": 210938479.0, "reward": 2.3499512672424316, "reward_std": 0.497981458902359, "rewards/code_complexity_reward/mean": 0.9484374523162842, "rewards/code_complexity_reward/std": 0.10807649791240692, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1416, "step_time": 49.908627431839705 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 473.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 124.30859375, "completions/mean_terminated_length": 124.30859375, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.2486771484836936, "epoch": 0.8074074074074075, "frac_reward_zero_std": 0.109375, "grad_norm": 0.13950319588184357, "kl": 0.19443286443129182, "learning_rate": 5.474165542838439e-07, "loss": 0.0009717894718050957, "num_tokens": 211071765.0, "reward": 2.343017578125, "reward_std": 0.5244569778442383, "rewards/code_complexity_reward/mean": 0.9458984136581421, "rewards/code_complexity_reward/std": 0.1310531347990036, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 1417, "step_time": 45.283612503670156 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 431.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 124.0703125, "completions/mean_terminated_length": 124.0703125, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.24811373371630907, "epoch": 0.807977207977208, "frac_reward_zero_std": 0.09375, "grad_norm": 0.15384525060653687, "kl": 0.20677303685806692, "learning_rate": 5.443141972492543e-07, "loss": 0.0010340646840631962, "num_tokens": 211203593.0, "reward": 2.42626953125, "reward_std": 0.5432456135749817, "rewards/code_complexity_reward/mean": 0.9512695670127869, "rewards/code_complexity_reward/std": 0.12814860045909882, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1418, "step_time": 50.232987981289625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 122.169921875, "completions/mean_terminated_length": 121.40704345703125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24319396866485476, "epoch": 0.8085470085470086, "frac_reward_zero_std": 0.1875, "grad_norm": 0.16808128356933594, "kl": 0.18889880715869367, "learning_rate": 5.412195818805674e-07, "loss": 0.0009444261086173356, "num_tokens": 211332328.0, "reward": 2.2716310024261475, "reward_std": 0.4595657289028168, "rewards/code_complexity_reward/mean": 0.9482421875, "rewards/code_complexity_reward/std": 0.1144494041800499, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1419, "step_time": 55.865154908038676 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 124.072265625, "completions/mean_terminated_length": 124.072265625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24569486360996962, "epoch": 0.8091168091168092, "frac_reward_zero_std": 0.203125, "grad_norm": 0.1697133332490921, "kl": 0.19724522496107966, "learning_rate": 5.381327204279518e-07, "loss": 0.00098621123470366, "num_tokens": 211462653.0, "reward": 2.3753905296325684, "reward_std": 0.5397504568099976, "rewards/code_complexity_reward/mean": 0.9453125, "rewards/code_complexity_reward/std": 0.1396721750497818, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1420, "step_time": 38.55875185225159 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 272.0, "completions/max_terminated_length": 272.0, "completions/mean_length": 123.8984375, "completions/mean_terminated_length": 123.8984375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23682583589106798, "epoch": 0.8096866096866097, "frac_reward_zero_std": 0.171875, "grad_norm": 0.1810099482536316, "kl": 0.19206822360865772, "learning_rate": 5.350536251108801e-07, "loss": 0.0009600984631106257, "num_tokens": 211592177.0, "reward": 2.3262696266174316, "reward_std": 0.4859960377216339, "rewards/code_complexity_reward/mean": 0.950878918170929, "rewards/code_complexity_reward/std": 0.11078888177871704, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1421, "step_time": 36.12865463178605 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 422.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 124.8125, "completions/mean_terminated_length": 124.8125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.2496068465989083, "epoch": 0.8102564102564103, "frac_reward_zero_std": 0.140625, "grad_norm": 0.1860157549381256, "kl": 0.19507814815733582, "learning_rate": 5.319823081180823e-07, "loss": 0.0009756421786732972, "num_tokens": 211723337.0, "reward": 2.338427782058716, "reward_std": 0.5360480546951294, "rewards/code_complexity_reward/mean": 0.9468750357627869, "rewards/code_complexity_reward/std": 0.1569960117340088, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1422, "step_time": 51.789462925866246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 115.21875, "completions/mean_terminated_length": 115.21875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2317929973360151, "epoch": 0.8108262108262109, "frac_reward_zero_std": 0.15625, "grad_norm": 0.15672120451927185, "kl": 0.21519617130979896, "learning_rate": 5.289187816074989e-07, "loss": 0.0010760558070614934, "num_tokens": 211850393.0, "reward": 2.423095703125, "reward_std": 0.5344264507293701, "rewards/code_complexity_reward/mean": 0.9551757574081421, "rewards/code_complexity_reward/std": 0.1194310113787651, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1423, "step_time": 39.25883524585515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 123.00390625, "completions/mean_terminated_length": 123.00390625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.25218939781188965, "epoch": 0.8113960113960114, "frac_reward_zero_std": 0.171875, "grad_norm": 0.1687704622745514, "kl": 0.19122923631221056, "learning_rate": 5.258630577062304e-07, "loss": 0.0009562023915350437, "num_tokens": 211982683.0, "reward": 2.284374952316284, "reward_std": 0.4950043857097626, "rewards/code_complexity_reward/mean": 0.9460937976837158, "rewards/code_complexity_reward/std": 0.13938072323799133, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1424, "step_time": 37.20437093731016 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 120.84375, "completions/mean_terminated_length": 120.84375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24331675725989044, "epoch": 0.811965811965812, "frac_reward_zero_std": 0.203125, "grad_norm": 0.16810372471809387, "kl": 0.18940588075201958, "learning_rate": 5.228151485104899e-07, "loss": 0.0009471513913013041, "num_tokens": 212111603.0, "reward": 2.3287110328674316, "reward_std": 0.5374851822853088, "rewards/code_complexity_reward/mean": 0.9425780773162842, "rewards/code_complexity_reward/std": 0.156820148229599, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1425, "step_time": 57.69561199005693 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 306.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 123.10546875, "completions/mean_terminated_length": 123.10546875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.25664157373830676, "epoch": 0.8125356125356126, "frac_reward_zero_std": 0.21875, "grad_norm": 0.13729530572891235, "kl": 0.2000620283652097, "learning_rate": 5.197750660855566e-07, "loss": 0.0010001494083553553, "num_tokens": 212242665.0, "reward": 2.34765625, "reward_std": 0.5187429785728455, "rewards/code_complexity_reward/mean": 0.9419921636581421, "rewards/code_complexity_reward/std": 0.1329859346151352, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1426, "step_time": 34.913421243429184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 117.044921875, "completions/mean_terminated_length": 117.044921875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24380387691780925, "epoch": 0.8131054131054131, "frac_reward_zero_std": 0.125, "grad_norm": 0.14462625980377197, "kl": 0.2098227613605559, "learning_rate": 5.167428224657278e-07, "loss": 0.001049440586939454, "num_tokens": 212369624.0, "reward": 2.4200196266174316, "reward_std": 0.5442140102386475, "rewards/code_complexity_reward/mean": 0.9596679210662842, "rewards/code_complexity_reward/std": 0.13267269730567932, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1427, "step_time": 48.353150894865394 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 314.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 115.17578125, "completions/mean_terminated_length": 115.17578125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24781676172278821, "epoch": 0.8136752136752137, "frac_reward_zero_std": 0.15625, "grad_norm": 0.19077147543430328, "kl": 0.19408431730698794, "learning_rate": 5.137184296542682e-07, "loss": 0.0009704558760859072, "num_tokens": 212495018.0, "reward": 2.376415967941284, "reward_std": 0.5002853274345398, "rewards/code_complexity_reward/mean": 0.96533203125, "rewards/code_complexity_reward/std": 0.10113972425460815, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1428, "step_time": 45.0546589596197 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 377.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 126.8203125, "completions/mean_terminated_length": 126.8203125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23493769485503435, "epoch": 0.8142450142450143, "frac_reward_zero_std": 0.171875, "grad_norm": 0.1506134420633316, "kl": 0.1916435060556978, "learning_rate": 5.107018996233676e-07, "loss": 0.000957886571995914, "num_tokens": 212627734.0, "reward": 2.327929735183716, "reward_std": 0.4748789072036743, "rewards/code_complexity_reward/mean": 0.9583984613418579, "rewards/code_complexity_reward/std": 0.09435005486011505, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1429, "step_time": 39.289882014505565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 368.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 118.197265625, "completions/mean_terminated_length": 118.197265625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2346681896597147, "epoch": 0.8148148148148148, "frac_reward_zero_std": 0.140625, "grad_norm": 0.16372986137866974, "kl": 0.19914359506219625, "learning_rate": 5.076932443140875e-07, "loss": 0.0009960804600268602, "num_tokens": 212755859.0, "reward": 2.414306640625, "reward_std": 0.5190421342849731, "rewards/code_complexity_reward/mean": 0.958300769329071, "rewards/code_complexity_reward/std": 0.10404934734106064, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1430, "step_time": 50.19744755420834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 127.6640625, "completions/mean_terminated_length": 127.6640625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23664751136675477, "epoch": 0.8153846153846154, "frac_reward_zero_std": 0.15625, "grad_norm": 0.16666273772716522, "kl": 0.19404153944924474, "learning_rate": 5.0469247563632e-07, "loss": 0.0009703486575745046, "num_tokens": 212893615.0, "reward": 2.339404344558716, "reward_std": 0.5059443712234497, "rewards/code_complexity_reward/mean": 0.9478515982627869, "rewards/code_complexity_reward/std": 0.12044142186641693, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1431, "step_time": 46.47859122324735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 327.0, "completions/mean_length": 130.15234375, "completions/mean_terminated_length": 129.40509033203125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2524991526734084, "epoch": 0.815954415954416, "frac_reward_zero_std": 0.15625, "grad_norm": 0.15916170179843903, "kl": 0.20357289956882596, "learning_rate": 5.016996054687354e-07, "loss": 0.0010184157872572541, "num_tokens": 213032845.0, "reward": 2.260693311691284, "reward_std": 0.48985716700553894, "rewards/code_complexity_reward/mean": 0.9421875476837158, "rewards/code_complexity_reward/std": 0.15220975875854492, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1432, "step_time": 54.98472947161645 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 278.0, "completions/max_terminated_length": 278.0, "completions/mean_length": 117.63671875, "completions/mean_terminated_length": 117.63671875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22436794661916792, "epoch": 0.8165242165242165, "frac_reward_zero_std": 0.1875, "grad_norm": 0.11331093311309814, "kl": 0.1962982825934887, "learning_rate": 4.987146456587389e-07, "loss": 0.0009819408878684044, "num_tokens": 213159723.0, "reward": 2.3437013626098633, "reward_std": 0.5209652185440063, "rewards/code_complexity_reward/mean": 0.94921875, "rewards/code_complexity_reward/std": 0.1393631547689438, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1433, "step_time": 51.747258976101875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 372.0, "completions/max_terminated_length": 372.0, "completions/mean_length": 121.748046875, "completions/mean_terminated_length": 121.748046875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24863627483136952, "epoch": 0.8170940170940171, "frac_reward_zero_std": 0.15625, "grad_norm": 0.16719825565814972, "kl": 0.2438754099421203, "learning_rate": 4.957376080224213e-07, "loss": 0.0012198705226182938, "num_tokens": 213293474.0, "reward": 2.2737793922424316, "reward_std": 0.46931496262550354, "rewards/code_complexity_reward/mean": 0.9632812738418579, "rewards/code_complexity_reward/std": 0.12551021575927734, "rewards/code_execution_reward/mean": 0.21875, "rewards/code_execution_reward/std": 0.41380295157432556, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1434, "step_time": 46.62986506614834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 324.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 121.3828125, "completions/mean_terminated_length": 121.3828125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2395266741514206, "epoch": 0.8176638176638177, "frac_reward_zero_std": 0.140625, "grad_norm": 0.13349328935146332, "kl": 0.20469013252295554, "learning_rate": 4.927685043445129e-07, "loss": 0.001023858436383307, "num_tokens": 213425534.0, "reward": 2.45361328125, "reward_std": 0.5184762477874756, "rewards/code_complexity_reward/mean": 0.9620116949081421, "rewards/code_complexity_reward/std": 0.10373294353485107, "rewards/code_execution_reward/mean": 0.396484375, "rewards/code_execution_reward/std": 0.4896455705165863, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1435, "step_time": 36.32184597942978 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 297.0, "completions/max_terminated_length": 297.0, "completions/mean_length": 122.23046875, "completions/mean_terminated_length": 122.23046875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24916316522285342, "epoch": 0.8182336182336183, "frac_reward_zero_std": 0.140625, "grad_norm": 0.1750546097755432, "kl": 0.24000333715230227, "learning_rate": 4.898073463783389e-07, "loss": 0.001199391670525074, "num_tokens": 213556108.0, "reward": 2.4137697219848633, "reward_std": 0.5133928656578064, "rewards/code_complexity_reward/mean": 0.961230456829071, "rewards/code_complexity_reward/std": 0.10158167034387589, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1436, "step_time": 35.080908932723105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 124.46875, "completions/mean_terminated_length": 123.71037292480469, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2438616536092013, "epoch": 0.8188034188034188, "frac_reward_zero_std": 0.109375, "grad_norm": 0.14192138612270355, "kl": 0.19248339417390525, "learning_rate": 4.868541458457682e-07, "loss": 0.0009622344514355063, "num_tokens": 213688692.0, "reward": 2.3344240188598633, "reward_std": 0.49795109033584595, "rewards/code_complexity_reward/mean": 0.9572266340255737, "rewards/code_complexity_reward/std": 0.11848077178001404, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1437, "step_time": 85.9187478441745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 120.33203125, "completions/mean_terminated_length": 119.56555938720703, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23348751291632652, "epoch": 0.8193732193732194, "frac_reward_zero_std": 0.1875, "grad_norm": 0.1372142732143402, "kl": 0.20847795181907713, "learning_rate": 4.839089144371728e-07, "loss": 0.0010422116611152887, "num_tokens": 213818982.0, "reward": 2.4197754859924316, "reward_std": 0.5410617589950562, "rewards/code_complexity_reward/mean": 0.9537109136581421, "rewards/code_complexity_reward/std": 0.12926064431667328, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1438, "step_time": 53.434172104112804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 321.0, "completions/max_terminated_length": 321.0, "completions/mean_length": 121.22265625, "completions/mean_terminated_length": 121.22265625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24701545806601644, "epoch": 0.81994301994302, "frac_reward_zero_std": 0.125, "grad_norm": 0.1572091281414032, "kl": 0.19893110380508006, "learning_rate": 4.80971663811376e-07, "loss": 0.0009946620557457209, "num_tokens": 213951912.0, "reward": 2.3873534202575684, "reward_std": 0.5248030424118042, "rewards/code_complexity_reward/mean": 0.95166015625, "rewards/code_complexity_reward/std": 0.12148050218820572, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 1439, "step_time": 36.83607648778707 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 121.451171875, "completions/mean_terminated_length": 121.451171875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2474160350393504, "epoch": 0.8205128205128205, "frac_reward_zero_std": 0.125, "grad_norm": 0.14379993081092834, "kl": 0.20948326005600393, "learning_rate": 4.780424055956114e-07, "loss": 0.0010478574549779296, "num_tokens": 214086751.0, "reward": 2.2769532203674316, "reward_std": 0.47688862681388855, "rewards/code_complexity_reward/mean": 0.9572266340255737, "rewards/code_complexity_reward/std": 0.13387395441532135, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1440, "step_time": 43.61315849516541 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 122.89453125, "completions/mean_terminated_length": 122.13307189941406, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.25884617515839636, "epoch": 0.8210826210826211, "frac_reward_zero_std": 0.171875, "grad_norm": 0.15095391869544983, "kl": 0.21066149207763374, "learning_rate": 4.7512115138547143e-07, "loss": 0.0010538387577980757, "num_tokens": 214219217.0, "reward": 2.29931640625, "reward_std": 0.5133463740348816, "rewards/code_complexity_reward/mean": 0.9500000476837158, "rewards/code_complexity_reward/std": 0.15183161199092865, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1441, "step_time": 66.2305535133928 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 285.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 118.01171875, "completions/mean_terminated_length": 118.01171875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23158159153535962, "epoch": 0.8216524216524217, "frac_reward_zero_std": 0.1875, "grad_norm": 0.14157669246196747, "kl": 0.203330609947443, "learning_rate": 4.7220791274486755e-07, "loss": 0.0010167601285502315, "num_tokens": 214346823.0, "reward": 2.40478515625, "reward_std": 0.5319549441337585, "rewards/code_complexity_reward/mean": 0.9522461295127869, "rewards/code_complexity_reward/std": 0.13304293155670166, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1442, "step_time": 33.949915329925716 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 268.0, "completions/max_terminated_length": 268.0, "completions/mean_length": 119.123046875, "completions/mean_terminated_length": 119.123046875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.24147746595554054, "epoch": 0.8222222222222222, "frac_reward_zero_std": 0.15625, "grad_norm": 0.16505062580108643, "kl": 0.20548898470588028, "learning_rate": 4.693027012059778e-07, "loss": 0.001027461839839816, "num_tokens": 214475982.0, "reward": 2.320849657058716, "reward_std": 0.5177109241485596, "rewards/code_complexity_reward/mean": 0.9486328363418579, "rewards/code_complexity_reward/std": 0.14674580097198486, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1443, "step_time": 43.324637508019805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 132.33203125, "completions/mean_terminated_length": 132.33203125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24405625462532043, "epoch": 0.8227920227920228, "frac_reward_zero_std": 0.140625, "grad_norm": 0.12055884301662445, "kl": 0.1912419251166284, "learning_rate": 4.664055282692076e-07, "loss": 0.0009566039079800248, "num_tokens": 214613568.0, "reward": 2.3165040016174316, "reward_std": 0.5265235304832458, "rewards/code_complexity_reward/mean": 0.945019543170929, "rewards/code_complexity_reward/std": 0.15205571055412292, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1444, "step_time": 58.4261262845248 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 126.0078125, "completions/mean_terminated_length": 126.0078125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2579957526177168, "epoch": 0.8233618233618234, "frac_reward_zero_std": 0.21875, "grad_norm": 0.13586163520812988, "kl": 0.1983035383746028, "learning_rate": 4.635164054031391e-07, "loss": 0.0009914443362504244, "num_tokens": 214747812.0, "reward": 2.2393555641174316, "reward_std": 0.4787857234477997, "rewards/code_complexity_reward/mean": 0.9459960460662842, "rewards/code_complexity_reward/std": 0.15253432095050812, "rewards/code_execution_reward/mean": 0.205078125, "rewards/code_execution_reward/std": 0.4041535556316376, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1445, "step_time": 46.073916419409215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 462.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 120.30078125, "completions/mean_terminated_length": 120.30078125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2299506519921124, "epoch": 0.8239316239316239, "frac_reward_zero_std": 0.1875, "grad_norm": 0.117156021296978, "kl": 0.2009225251385942, "learning_rate": 4.606353440444897e-07, "loss": 0.0010043231304734945, "num_tokens": 214879886.0, "reward": 2.478759765625, "reward_std": 0.5389562845230103, "rewards/code_complexity_reward/mean": 0.9573242664337158, "rewards/code_complexity_reward/std": 0.1119130402803421, "rewards/code_execution_reward/mean": 0.427734375, "rewards/code_execution_reward/std": 0.4952339828014374, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1446, "step_time": 47.41960246022791 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 110.84375, "completions/mean_terminated_length": 110.84375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2438540831208229, "epoch": 0.8245014245014245, "frac_reward_zero_std": 0.1875, "grad_norm": 0.15816980600357056, "kl": 0.2669631780590862, "learning_rate": 4.5776235559806313e-07, "loss": 0.001335792476311326, "num_tokens": 215006934.0, "reward": 2.3916993141174316, "reward_std": 0.5105007290840149, "rewards/code_complexity_reward/mean": 0.965527355670929, "rewards/code_complexity_reward/std": 0.11129075288772583, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1447, "step_time": 48.30592873971909 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 285.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 117.470703125, "completions/mean_terminated_length": 117.470703125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24199969577603042, "epoch": 0.8250712250712251, "frac_reward_zero_std": 0.203125, "grad_norm": 0.1025867834687233, "kl": 0.22049139020964503, "learning_rate": 4.5489745143670715e-07, "loss": 0.0011026081629097462, "num_tokens": 215135327.0, "reward": 2.3602538108825684, "reward_std": 0.5169105529785156, "rewards/code_complexity_reward/mean": 0.952343761920929, "rewards/code_complexity_reward/std": 0.12974600493907928, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1448, "step_time": 33.975055888295174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 283.0, "completions/max_terminated_length": 283.0, "completions/mean_length": 116.4453125, "completions/mean_terminated_length": 116.4453125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2433408119250089, "epoch": 0.8256410256410256, "frac_reward_zero_std": 0.140625, "grad_norm": 0.1320139467716217, "kl": 0.22436312306672335, "learning_rate": 4.5204064290126806e-07, "loss": 0.0011221907334402204, "num_tokens": 215261291.0, "reward": 2.43603515625, "reward_std": 0.49301907420158386, "rewards/code_complexity_reward/mean": 0.9747070074081421, "rewards/code_complexity_reward/std": 0.06846022605895996, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1449, "step_time": 34.17912975046784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 351.0, "completions/max_terminated_length": 351.0, "completions/mean_length": 123.095703125, "completions/mean_terminated_length": 123.095703125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24314567679539323, "epoch": 0.8262108262108262, "frac_reward_zero_std": 0.21875, "grad_norm": 0.17125564813613892, "kl": 0.20632390910759568, "learning_rate": 4.4919194130054356e-07, "loss": 0.0010323114693164825, "num_tokens": 215392180.0, "reward": 2.3193845748901367, "reward_std": 0.496150940656662, "rewards/code_complexity_reward/mean": 0.9581054449081421, "rewards/code_complexity_reward/std": 0.12960773706436157, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1450, "step_time": 37.615564273670316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 465.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 118.12109375, "completions/mean_terminated_length": 118.12109375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24924031482078135, "epoch": 0.8267806267806268, "frac_reward_zero_std": 0.109375, "grad_norm": 0.1683889925479889, "kl": 0.21273994585499167, "learning_rate": 4.463513579112422e-07, "loss": 0.0010644226567819715, "num_tokens": 215520066.0, "reward": 2.367968797683716, "reward_std": 0.48942771553993225, "rewards/code_complexity_reward/mean": 0.9652343988418579, "rewards/code_complexity_reward/std": 0.09437374770641327, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1451, "step_time": 45.264624645002186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 300.0, "completions/max_terminated_length": 300.0, "completions/mean_length": 121.169921875, "completions/mean_terminated_length": 121.169921875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2371737735811621, "epoch": 0.8273504273504273, "frac_reward_zero_std": 0.140625, "grad_norm": 0.15336936712265015, "kl": 0.2031182232312858, "learning_rate": 4.4351890397793335e-07, "loss": 0.0010160787496715784, "num_tokens": 215652481.0, "reward": 2.3682126998901367, "reward_std": 0.5378361344337463, "rewards/code_complexity_reward/mean": 0.9530273079872131, "rewards/code_complexity_reward/std": 0.14673756062984467, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1452, "step_time": 38.088052281178534 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 281.0, "completions/max_terminated_length": 281.0, "completions/mean_length": 118.9765625, "completions/mean_terminated_length": 118.9765625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2477715250570327, "epoch": 0.8279202279202279, "frac_reward_zero_std": 0.1875, "grad_norm": 0.1394660919904709, "kl": 0.21315450756810606, "learning_rate": 4.4069459071300834e-07, "loss": 0.0010663443244993687, "num_tokens": 215780669.0, "reward": 2.349609375, "reward_std": 0.5383436679840088, "rewards/code_complexity_reward/mean": 0.9546874761581421, "rewards/code_complexity_reward/std": 0.15224188566207886, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1453, "step_time": 33.97266622353345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 116.390625, "completions/mean_terminated_length": 115.61643981933594, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2433261915575713, "epoch": 0.8284900284900285, "frac_reward_zero_std": 0.203125, "grad_norm": 0.1836201548576355, "kl": 0.2180402260273695, "learning_rate": 4.3787842929663096e-07, "loss": 0.0010904248338192701, "num_tokens": 215907381.0, "reward": 2.3696775436401367, "reward_std": 0.5433189272880554, "rewards/code_complexity_reward/mean": 0.9574218988418579, "rewards/code_complexity_reward/std": 0.1517142504453659, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1454, "step_time": 48.43966364022344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 345.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 118.9140625, "completions/mean_terminated_length": 118.9140625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24670959101058543, "epoch": 0.8290598290598291, "frac_reward_zero_std": 0.171875, "grad_norm": 0.14327837526798248, "kl": 0.21589801623485982, "learning_rate": 4.3507043087669733e-07, "loss": 0.0010798097355291247, "num_tokens": 216037113.0, "reward": 2.2853517532348633, "reward_std": 0.49045711755752563, "rewards/code_complexity_reward/mean": 0.956835925579071, "rewards/code_complexity_reward/std": 0.1409783661365509, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1455, "step_time": 44.47772340942174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 269.0, "completions/max_terminated_length": 269.0, "completions/mean_length": 118.41015625, "completions/mean_terminated_length": 118.41015625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24309772672131658, "epoch": 0.8296296296296296, "frac_reward_zero_std": 0.1875, "grad_norm": 0.16213321685791016, "kl": 0.22510260296985507, "learning_rate": 4.322706065687895e-07, "loss": 0.0011260609608143568, "num_tokens": 216165915.0, "reward": 2.3363282680511475, "reward_std": 0.4859659969806671, "rewards/code_complexity_reward/mean": 0.9638671875, "rewards/code_complexity_reward/std": 0.10705580562353134, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1456, "step_time": 33.838088602758944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 122.083984375, "completions/mean_terminated_length": 122.083984375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24148042569868267, "epoch": 0.8301994301994302, "frac_reward_zero_std": 0.25, "grad_norm": 0.14992469549179077, "kl": 0.21859983494505286, "learning_rate": 4.2947896745613086e-07, "loss": 0.0010932954028248787, "num_tokens": 216294182.0, "reward": 2.3965821266174316, "reward_std": 0.4975703954696655, "rewards/code_complexity_reward/mean": 0.966503918170929, "rewards/code_complexity_reward/std": 0.0966467335820198, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1457, "step_time": 37.96386128757149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 317.0, "completions/max_terminated_length": 317.0, "completions/mean_length": 122.708984375, "completions/mean_terminated_length": 122.708984375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2362561512272805, "epoch": 0.8307692307692308, "frac_reward_zero_std": 0.1875, "grad_norm": 0.13679637014865875, "kl": 0.19771784730255604, "learning_rate": 4.2669552458954546e-07, "loss": 0.000988912070170045, "num_tokens": 216428681.0, "reward": 2.38623046875, "reward_std": 0.5227208733558655, "rewards/code_complexity_reward/mean": 0.955078125, "rewards/code_complexity_reward/std": 0.12830215692520142, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1458, "step_time": 44.064127133227885 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 119.951171875, "completions/mean_terminated_length": 119.951171875, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.2470145768020302, "epoch": 0.8313390313390313, "frac_reward_zero_std": 0.234375, "grad_norm": 0.15516319870948792, "kl": 0.21021677181124687, "learning_rate": 4.239202889874103e-07, "loss": 0.0010520153446123004, "num_tokens": 216559296.0, "reward": 2.39892578125, "reward_std": 0.5082430839538574, "rewards/code_complexity_reward/mean": 0.9678710699081421, "rewards/code_complexity_reward/std": 0.1025027334690094, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1459, "step_time": 34.50033118482679 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 113.921875, "completions/mean_terminated_length": 113.921875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24988156068138778, "epoch": 0.8319088319088319, "frac_reward_zero_std": 0.171875, "grad_norm": 0.12447261810302734, "kl": 0.2230022712610662, "learning_rate": 4.2115327163561425e-07, "loss": 0.0011157577391713858, "num_tokens": 216684352.0, "reward": 2.4207029342651367, "reward_std": 0.552036464214325, "rewards/code_complexity_reward/mean": 0.9571288824081421, "rewards/code_complexity_reward/std": 0.1470617651939392, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1460, "step_time": 54.95064941607416 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 118.341796875, "completions/mean_terminated_length": 118.341796875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2251836566720158, "epoch": 0.8324786324786325, "frac_reward_zero_std": 0.15625, "grad_norm": 0.4804333746433258, "kl": 0.64043033355847, "learning_rate": 4.183944834875128e-07, "loss": 0.003205346642062068, "num_tokens": 216813191.0, "reward": 2.4461913108825684, "reward_std": 0.5493459105491638, "rewards/code_complexity_reward/mean": 0.96630859375, "rewards/code_complexity_reward/std": 0.13301651179790497, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1461, "step_time": 41.73647580947727 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 118.173828125, "completions/mean_terminated_length": 118.173828125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24211826478131115, "epoch": 0.833048433048433, "frac_reward_zero_std": 0.1875, "grad_norm": 0.1877371072769165, "kl": 0.2365606368985027, "learning_rate": 4.156439354638889e-07, "loss": 0.0011829833965748549, "num_tokens": 216941360.0, "reward": 2.3724608421325684, "reward_std": 0.5835152864456177, "rewards/code_complexity_reward/mean": 0.93994140625, "rewards/code_complexity_reward/std": 0.1835755854845047, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 1462, "step_time": 39.832264938391745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 114.212890625, "completions/mean_terminated_length": 114.212890625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23982156091369689, "epoch": 0.8336182336182336, "frac_reward_zero_std": 0.1875, "grad_norm": 0.14186391234397888, "kl": 0.22212846507318318, "learning_rate": 4.129016384529025e-07, "loss": 0.0011114201042801142, "num_tokens": 217066173.0, "reward": 2.436279296875, "reward_std": 0.5196467041969299, "rewards/code_complexity_reward/mean": 0.9703124761581421, "rewards/code_complexity_reward/std": 0.10154145210981369, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1463, "step_time": 35.17147097270936 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 115.408203125, "completions/mean_terminated_length": 115.408203125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24227468017488718, "epoch": 0.8341880341880342, "frac_reward_zero_std": 0.25, "grad_norm": 0.16248583793640137, "kl": 0.22528019966557622, "learning_rate": 4.1016760331005547e-07, "loss": 0.0011274211574345827, "num_tokens": 217191670.0, "reward": 2.3550782203674316, "reward_std": 0.46164384484291077, "rewards/code_complexity_reward/mean": 0.976757824420929, "rewards/code_complexity_reward/std": 0.05575347691774368, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1464, "step_time": 37.91445657517761 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 122.095703125, "completions/mean_terminated_length": 122.095703125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.25214234087616205, "epoch": 0.8347578347578347, "frac_reward_zero_std": 0.078125, "grad_norm": 0.15642869472503662, "kl": 0.20700101950205863, "learning_rate": 4.0744184085814126e-07, "loss": 0.0010355847189202905, "num_tokens": 217325463.0, "reward": 2.3070313930511475, "reward_std": 0.4547524154186249, "rewards/code_complexity_reward/mean": 0.9658203125, "rewards/code_complexity_reward/std": 0.08476807922124863, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1465, "step_time": 38.749779257923365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 117.162109375, "completions/mean_terminated_length": 117.162109375, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.24344153678976, "epoch": 0.8353276353276353, "frac_reward_zero_std": 0.171875, "grad_norm": 0.13993032276630402, "kl": 0.23081386438570917, "learning_rate": 4.047243618872079e-07, "loss": 0.0011549426708370447, "num_tokens": 217454586.0, "reward": 2.3890624046325684, "reward_std": 0.5007113814353943, "rewards/code_complexity_reward/mean": 0.970703125, "rewards/code_complexity_reward/std": 0.09847712516784668, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1466, "step_time": 44.71846341434866 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 446.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 115.396484375, "completions/mean_terminated_length": 115.396484375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24229464400559664, "epoch": 0.8358974358974359, "frac_reward_zero_std": 0.15625, "grad_norm": 0.1701904684305191, "kl": 0.21676951297558844, "learning_rate": 4.0201517715451276e-07, "loss": 0.0010846755467355251, "num_tokens": 217581005.0, "reward": 2.406494140625, "reward_std": 0.508468508720398, "rewards/code_complexity_reward/mean": 0.9698241949081421, "rewards/code_complexity_reward/std": 0.10247699171304703, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1467, "step_time": 46.12625970505178 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 119.140625, "completions/mean_terminated_length": 119.140625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23747320845723152, "epoch": 0.8364672364672364, "frac_reward_zero_std": 0.21875, "grad_norm": 0.12329281121492386, "kl": 0.22420198586769402, "learning_rate": 3.993142973844782e-07, "loss": 0.0011212778044864535, "num_tokens": 217711141.0, "reward": 2.3165526390075684, "reward_std": 0.49871718883514404, "rewards/code_complexity_reward/mean": 0.961132824420929, "rewards/code_complexity_reward/std": 0.12845265865325928, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1468, "step_time": 36.75724045652896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 123.87890625, "completions/mean_terminated_length": 123.11936950683594, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24640002637170255, "epoch": 0.837037037037037, "frac_reward_zero_std": 0.265625, "grad_norm": 0.11776763200759888, "kl": 0.21007572254166007, "learning_rate": 3.9662173326865365e-07, "loss": 0.0010508723789826035, "num_tokens": 217843039.0, "reward": 2.308349609375, "reward_std": 0.4699791967868805, "rewards/code_complexity_reward/mean": 0.9673827886581421, "rewards/code_complexity_reward/std": 0.10384292155504227, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1469, "step_time": 55.3188175233081 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 416.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 116.1875, "completions/mean_terminated_length": 116.1875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.23411160381510854, "epoch": 0.8376068376068376, "frac_reward_zero_std": 0.296875, "grad_norm": 0.12360279262065887, "kl": 0.2234401721507311, "learning_rate": 3.93937495465668e-07, "loss": 0.0011180725414305925, "num_tokens": 217971215.0, "reward": 2.4013185501098633, "reward_std": 0.49841180443763733, "rewards/code_complexity_reward/mean": 0.97216796875, "rewards/code_complexity_reward/std": 0.08697902411222458, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.025282343849539757, "step": 1470, "step_time": 50.804699359461665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 302.0, "completions/max_terminated_length": 302.0, "completions/mean_length": 111.943359375, "completions/mean_terminated_length": 111.943359375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.25128353759646416, "epoch": 0.8381766381766381, "frac_reward_zero_std": 0.21875, "grad_norm": 0.14821982383728027, "kl": 0.23157198331318796, "learning_rate": 3.9126159460119273e-07, "loss": 0.0011585162719711661, "num_tokens": 218096090.0, "reward": 2.2916994094848633, "reward_std": 0.4773586094379425, "rewards/code_complexity_reward/mean": 0.9670898914337158, "rewards/code_complexity_reward/std": 0.1266995072364807, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1471, "step_time": 86.40272031910717 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 306.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 118.466796875, "completions/mean_terminated_length": 118.466796875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2406397615559399, "epoch": 0.8387464387464387, "frac_reward_zero_std": 0.109375, "grad_norm": 0.1509789228439331, "kl": 0.21173807885497808, "learning_rate": 3.885940412678948e-07, "loss": 0.0010591903701424599, "num_tokens": 218224833.0, "reward": 2.3236327171325684, "reward_std": 0.5000959038734436, "rewards/code_complexity_reward/mean": 0.96484375, "rewards/code_complexity_reward/std": 0.13287568092346191, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1472, "step_time": 35.33431780990213 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 114.6328125, "completions/mean_terminated_length": 114.6328125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24782538786530495, "epoch": 0.8393162393162393, "frac_reward_zero_std": 0.109375, "grad_norm": 0.1828615516424179, "kl": 0.22844797396101058, "learning_rate": 3.8593484602539894e-07, "loss": 0.0011432826286181808, "num_tokens": 218354517.0, "reward": 2.2867674827575684, "reward_std": 0.5025395154953003, "rewards/code_complexity_reward/mean": 0.958691418170929, "rewards/code_complexity_reward/std": 0.1523708552122116, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1473, "step_time": 44.4473479995504 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 314.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 111.083984375, "completions/mean_terminated_length": 111.083984375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23090757778845727, "epoch": 0.8398860398860399, "frac_reward_zero_std": 0.21875, "grad_norm": 0.1517908126115799, "kl": 0.21569812181405723, "learning_rate": 3.832840194002424e-07, "loss": 0.001078711124137044, "num_tokens": 218481944.0, "reward": 2.480419874191284, "reward_std": 0.5199716687202454, "rewards/code_complexity_reward/mean": 0.97265625, "rewards/code_complexity_reward/std": 0.09196897596120834, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1474, "step_time": 47.373873512260616 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 314.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 120.166015625, "completions/mean_terminated_length": 120.166015625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2627433920279145, "epoch": 0.8404558404558404, "frac_reward_zero_std": 0.25, "grad_norm": 0.1372746229171753, "kl": 0.2369740316644311, "learning_rate": 3.806415718858358e-07, "loss": 0.0011847192654386163, "num_tokens": 218613453.0, "reward": 2.2621092796325684, "reward_std": 0.48990777134895325, "rewards/code_complexity_reward/mean": 0.95703125, "rewards/code_complexity_reward/std": 0.15295323729515076, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1475, "step_time": 36.20156306400895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 270.0, "completions/max_terminated_length": 270.0, "completions/mean_length": 118.013671875, "completions/mean_terminated_length": 118.013671875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23243912192992866, "epoch": 0.841025641025641, "frac_reward_zero_std": 0.140625, "grad_norm": 0.18374310433864594, "kl": 0.27981971064582467, "learning_rate": 3.780075139424222e-07, "loss": 0.0013984990073367953, "num_tokens": 218739932.0, "reward": 2.410400390625, "reward_std": 0.5334096550941467, "rewards/code_complexity_reward/mean": 0.962207019329071, "rewards/code_complexity_reward/std": 0.13338294625282288, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1476, "step_time": 37.334852955304086 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 277.0, "completions/max_terminated_length": 277.0, "completions/mean_length": 116.55078125, "completions/mean_terminated_length": 116.55078125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.25398632464930415, "epoch": 0.8415954415954416, "frac_reward_zero_std": 0.21875, "grad_norm": 0.11907719820737839, "kl": 0.21752184233628213, "learning_rate": 3.753818559970307e-07, "loss": 0.0010886971140280366, "num_tokens": 218868358.0, "reward": 2.306201219558716, "reward_std": 0.4727318584918976, "rewards/code_complexity_reward/mean": 0.9693359136581421, "rewards/code_complexity_reward/std": 0.11979740113019943, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1477, "step_time": 43.42684883438051 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 296.0, "completions/max_terminated_length": 296.0, "completions/mean_length": 115.962890625, "completions/mean_terminated_length": 115.962890625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24568841373547912, "epoch": 0.8421652421652421, "frac_reward_zero_std": 0.203125, "grad_norm": 0.0960652083158493, "kl": 0.23167413659393787, "learning_rate": 3.727646084434422e-07, "loss": 0.001158472616225481, "num_tokens": 218993771.0, "reward": 2.369140625, "reward_std": 0.5399271845817566, "rewards/code_complexity_reward/mean": 0.9576172232627869, "rewards/code_complexity_reward/std": 0.1472865641117096, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1478, "step_time": 42.460333357565105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 120.544921875, "completions/mean_terminated_length": 120.544921875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2537126336246729, "epoch": 0.8427350427350427, "frac_reward_zero_std": 0.1875, "grad_norm": 0.14699821174144745, "kl": 0.23037734883837402, "learning_rate": 3.7015578164214136e-07, "loss": 0.0011524406727403402, "num_tokens": 219124754.0, "reward": 2.315380811691284, "reward_std": 0.481673926115036, "rewards/code_complexity_reward/mean": 0.9626953601837158, "rewards/code_complexity_reward/std": 0.1188453957438469, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1479, "step_time": 53.33051959145814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 475.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 120.060546875, "completions/mean_terminated_length": 120.060546875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24727423349395394, "epoch": 0.8433048433048433, "frac_reward_zero_std": 0.25, "grad_norm": 0.12248016893863678, "kl": 0.24007524666376412, "learning_rate": 3.675553859202827e-07, "loss": 0.0012005427852272987, "num_tokens": 219254025.0, "reward": 2.321582078933716, "reward_std": 0.47209614515304565, "rewards/code_complexity_reward/mean": 0.9637695550918579, "rewards/code_complexity_reward/std": 0.10141295939683914, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1480, "step_time": 53.630220513790846 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 256.0, "completions/max_terminated_length": 256.0, "completions/mean_length": 118.0390625, "completions/mean_terminated_length": 118.0390625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24289900925941765, "epoch": 0.8438746438746438, "frac_reward_zero_std": 0.203125, "grad_norm": 0.12608040869235992, "kl": 0.21760072279721498, "learning_rate": 3.649634315716419e-07, "loss": 0.001088402234017849, "num_tokens": 219383173.0, "reward": 2.353564500808716, "reward_std": 0.5132760405540466, "rewards/code_complexity_reward/mean": 0.9629883170127869, "rewards/code_complexity_reward/std": 0.13374866545200348, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 1481, "step_time": 39.72870243340731 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 269.0, "completions/mean_length": 114.734375, "completions/mean_terminated_length": 113.17647552490234, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24766844091936946, "epoch": 0.8444444444444444, "frac_reward_zero_std": 0.171875, "grad_norm": 0.1565515398979187, "kl": 0.21911910152994096, "learning_rate": 3.62379928856583e-07, "loss": 0.0010956699261441827, "num_tokens": 219510357.0, "reward": 2.3145506381988525, "reward_std": 0.48688843846321106, "rewards/code_complexity_reward/mean": 0.964062511920929, "rewards/code_complexity_reward/std": 0.11881467700004578, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1482, "step_time": 59.22075697220862 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 417.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 123.798828125, "completions/mean_terminated_length": 123.798828125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.25582740316167474, "epoch": 0.845014245014245, "frac_reward_zero_std": 0.15625, "grad_norm": 0.17080655694007874, "kl": 0.21881006099283695, "learning_rate": 3.5980488800201025e-07, "loss": 0.0010947970440611243, "num_tokens": 219644734.0, "reward": 2.3236327171325684, "reward_std": 0.49924659729003906, "rewards/code_complexity_reward/mean": 0.9588867425918579, "rewards/code_complexity_reward/std": 0.13257989287376404, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1483, "step_time": 59.90522724483162 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 314.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 114.2109375, "completions/mean_terminated_length": 114.2109375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24092518375255167, "epoch": 0.8455840455840455, "frac_reward_zero_std": 0.15625, "grad_norm": 0.13330940902233124, "kl": 0.22472477331757545, "learning_rate": 3.572383192013346e-07, "loss": 0.0011240628082305193, "num_tokens": 219772074.0, "reward": 2.391064405441284, "reward_std": 0.5937331318855286, "rewards/code_complexity_reward/mean": 0.94580078125, "rewards/code_complexity_reward/std": 0.1843084692955017, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1484, "step_time": 52.0906742811203 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 328.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 114.47265625, "completions/mean_terminated_length": 114.47265625, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.24244855344295502, "epoch": 0.8461538461538461, "frac_reward_zero_std": 0.203125, "grad_norm": 0.1272633671760559, "kl": 0.23291791882365942, "learning_rate": 3.546802326144275e-07, "loss": 0.0011647030478343368, "num_tokens": 219898220.0, "reward": 2.399609327316284, "reward_std": 0.5403031706809998, "rewards/code_complexity_reward/mean": 0.965624988079071, "rewards/code_complexity_reward/std": 0.1401006132364273, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1485, "step_time": 67.98359210416675 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 118.107421875, "completions/mean_terminated_length": 118.107421875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24920673412270844, "epoch": 0.8467236467236468, "frac_reward_zero_std": 0.21875, "grad_norm": 0.14549675583839417, "kl": 0.22671966557390988, "learning_rate": 3.5213063836758377e-07, "loss": 0.0011342416983097792, "num_tokens": 220033235.0, "reward": 2.3052244186401367, "reward_std": 0.4717647135257721, "rewards/code_complexity_reward/mean": 0.9722656607627869, "rewards/code_complexity_reward/std": 0.11767623573541641, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1486, "step_time": 42.92561831884086 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 258.0, "completions/max_terminated_length": 258.0, "completions/mean_length": 103.8515625, "completions/mean_terminated_length": 103.8515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23551468504592776, "epoch": 0.8472934472934472, "frac_reward_zero_std": 0.21875, "grad_norm": 0.1300806850194931, "kl": 0.26538359094411135, "learning_rate": 3.495895465534824e-07, "loss": 0.001327680191025138, "num_tokens": 220154287.0, "reward": 2.517773389816284, "reward_std": 0.5345559120178223, "rewards/code_complexity_reward/mean": 0.977343738079071, "rewards/code_complexity_reward/std": 0.10152639448642731, "rewards/code_execution_reward/mean": 0.4453125, "rewards/code_execution_reward/std": 0.49748632311820984, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1487, "step_time": 34.33271881379187 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 397.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 118.05078125, "completions/mean_terminated_length": 118.05078125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24680659407749772, "epoch": 0.8478632478632478, "frac_reward_zero_std": 0.234375, "grad_norm": 0.11212898045778275, "kl": 0.22599772550165653, "learning_rate": 3.4705696723114274e-07, "loss": 0.0011303642531856894, "num_tokens": 220282169.0, "reward": 2.3626952171325684, "reward_std": 0.5104149580001831, "rewards/code_complexity_reward/mean": 0.9658203125, "rewards/code_complexity_reward/std": 0.12638165056705475, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1488, "step_time": 41.22982823755592 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 326.0, "completions/max_terminated_length": 326.0, "completions/mean_length": 117.931640625, "completions/mean_terminated_length": 117.931640625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.252202172530815, "epoch": 0.8484330484330485, "frac_reward_zero_std": 0.171875, "grad_norm": 0.1662508249282837, "kl": 0.2332209781743586, "learning_rate": 3.4453291042588986e-07, "loss": 0.0011659408919513226, "num_tokens": 220411558.0, "reward": 2.303955078125, "reward_std": 0.481720894575119, "rewards/code_complexity_reward/mean": 0.9661133289337158, "rewards/code_complexity_reward/std": 0.12620894610881805, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1489, "step_time": 36.61098569910973 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 336.0, "completions/max_terminated_length": 336.0, "completions/mean_length": 114.310546875, "completions/mean_terminated_length": 114.310546875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24171348684467375, "epoch": 0.8490028490028491, "frac_reward_zero_std": 0.234375, "grad_norm": 0.10788645595312119, "kl": 0.2302578934468329, "learning_rate": 3.4201738612930914e-07, "loss": 0.0011518788523972034, "num_tokens": 220540261.0, "reward": 2.3560547828674316, "reward_std": 0.514641284942627, "rewards/code_complexity_reward/mean": 0.9718749523162842, "rewards/code_complexity_reward/std": 0.13237999379634857, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1490, "step_time": 37.26017574686557 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 288.0, "completions/mean_length": 116.658203125, "completions/mean_terminated_length": 115.88453674316406, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2399429394863546, "epoch": 0.8495726495726496, "frac_reward_zero_std": 0.296875, "grad_norm": 0.10165175795555115, "kl": 0.22990456083789468, "learning_rate": 3.3951040429921257e-07, "loss": 0.0011500038672238588, "num_tokens": 220666486.0, "reward": 2.357470750808716, "reward_std": 0.4901416301727295, "rewards/code_complexity_reward/mean": 0.9754883050918579, "rewards/code_complexity_reward/std": 0.1033194437623024, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1491, "step_time": 48.08622050099075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 472.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 115.109375, "completions/mean_terminated_length": 115.109375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2408582887146622, "epoch": 0.8501424501424502, "frac_reward_zero_std": 0.21875, "grad_norm": 0.14842823147773743, "kl": 0.22901456826366484, "learning_rate": 3.370119748595935e-07, "loss": 0.0011454543564468622, "num_tokens": 220794462.0, "reward": 2.3466796875, "reward_std": 0.48757675290107727, "rewards/code_complexity_reward/mean": 0.969042956829071, "rewards/code_complexity_reward/std": 0.10760541260242462, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1492, "step_time": 51.625032683834434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 420.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 121.048828125, "completions/mean_terminated_length": 121.048828125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.25062969233840704, "epoch": 0.8507122507122508, "frac_reward_zero_std": 0.203125, "grad_norm": 0.15030384063720703, "kl": 0.22481570369563997, "learning_rate": 3.345221077005928e-07, "loss": 0.001124225091189146, "num_tokens": 220925855.0, "reward": 2.337695360183716, "reward_std": 0.5053480863571167, "rewards/code_complexity_reward/mean": 0.9603515863418579, "rewards/code_complexity_reward/std": 0.12988130748271942, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1493, "step_time": 48.13344890251756 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 394.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 115.044921875, "completions/mean_terminated_length": 115.044921875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.22512010019272566, "epoch": 0.8512820512820513, "frac_reward_zero_std": 0.171875, "grad_norm": 0.1228794977068901, "kl": 0.23550777276977897, "learning_rate": 3.320408126784552e-07, "loss": 0.0011778725311160088, "num_tokens": 221051982.0, "reward": 2.474609375, "reward_std": 0.5100816488265991, "rewards/code_complexity_reward/mean": 0.9768555164337158, "rewards/code_complexity_reward/std": 0.07426399737596512, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1494, "step_time": 40.79293605219573 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 295.0, "completions/max_terminated_length": 295.0, "completions/mean_length": 116.11328125, "completions/mean_terminated_length": 116.11328125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23632872151210904, "epoch": 0.8518518518518519, "frac_reward_zero_std": 0.265625, "grad_norm": 0.12699395418167114, "kl": 0.22582732466980815, "learning_rate": 3.2956809961549373e-07, "loss": 0.0011298356112092733, "num_tokens": 221178296.0, "reward": 2.366015911102295, "reward_std": 0.4863471984863281, "rewards/code_complexity_reward/mean": 0.9750000238418579, "rewards/code_complexity_reward/std": 0.09258200973272324, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1495, "step_time": 54.2511027706787 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 234.0, "completions/max_terminated_length": 234.0, "completions/mean_length": 115.486328125, "completions/mean_terminated_length": 115.486328125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23595088464207947, "epoch": 0.8524216524216525, "frac_reward_zero_std": 0.234375, "grad_norm": 0.20502448081970215, "kl": 0.2158901917282492, "learning_rate": 3.271039783000493e-07, "loss": 0.0010797861032187939, "num_tokens": 221308217.0, "reward": 2.3672852516174316, "reward_std": 0.5126377940177917, "rewards/code_complexity_reward/mean": 0.964550793170929, "rewards/code_complexity_reward/std": 0.12686261534690857, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1496, "step_time": 38.26206820085645 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 419.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 112.9765625, "completions/mean_terminated_length": 112.9765625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2336611996870488, "epoch": 0.852991452991453, "frac_reward_zero_std": 0.234375, "grad_norm": 0.1398289054632187, "kl": 0.2350649661384523, "learning_rate": 3.2464845848645124e-07, "loss": 0.0011758331675082445, "num_tokens": 221435725.0, "reward": 2.3770017623901367, "reward_std": 0.513053297996521, "rewards/code_complexity_reward/mean": 0.9696289300918579, "rewards/code_complexity_reward/std": 0.11915577948093414, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1497, "step_time": 42.14477112609893 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 371.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 118.390625, "completions/mean_terminated_length": 118.390625, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.24322083150036633, "epoch": 0.8535612535612536, "frac_reward_zero_std": 0.203125, "grad_norm": 0.15748849511146545, "kl": 0.22883966052904725, "learning_rate": 3.222015498949793e-07, "loss": 0.0011448442237451673, "num_tokens": 221566253.0, "reward": 2.4120116233825684, "reward_std": 0.4949423670768738, "rewards/code_complexity_reward/mean": 0.97412109375, "rewards/code_complexity_reward/std": 0.09257782995700836, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1498, "step_time": 46.65532266162336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 117.328125, "completions/mean_terminated_length": 117.328125, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.2387225904967636, "epoch": 0.8541310541310542, "frac_reward_zero_std": 0.234375, "grad_norm": 0.12532266974449158, "kl": 0.22931553283706307, "learning_rate": 3.1976326221182576e-07, "loss": 0.0011465143179520965, "num_tokens": 221692837.0, "reward": 2.433398485183716, "reward_std": 0.5066946744918823, "rewards/code_complexity_reward/mean": 0.9740234613418579, "rewards/code_complexity_reward/std": 0.09549041092395782, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1499, "step_time": 44.128580584190786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 350.0, "completions/mean_length": 121.8515625, "completions/mean_terminated_length": 121.08805847167969, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24444237491115928, "epoch": 0.8547008547008547, "frac_reward_zero_std": 0.1875, "grad_norm": 0.1451447457075119, "kl": 0.2138261168729514, "learning_rate": 3.173336050890574e-07, "loss": 0.001069983234629035, "num_tokens": 221823321.0, "reward": 2.3524904251098633, "reward_std": 0.5110518932342529, "rewards/code_complexity_reward/mean": 0.9656250476837158, "rewards/code_complexity_reward/std": 0.12748508155345917, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1500, "step_time": 46.896746108308434 }, { "epoch": 0.8547008547008547, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 160.45, "eval_completions/max_terminated_length": 160.45, "eval_completions/mean_length": 117.04875, "eval_completions/mean_terminated_length": 117.04875, "eval_completions/min_length": 88.51, "eval_completions/min_terminated_length": 88.51, "eval_entropy": 0.23683456428349017, "eval_frac_reward_zero_std": 0.3, "eval_kl": 0.23487672910094262, "eval_loss": -0.0015326904831454158, "eval_num_tokens": 221823321.0, "eval_reward": 2.3485937082767485, "eval_reward_std": 0.23065330347046256, "eval_rewards/code_complexity_reward/mean": 0.9701249992847443, "eval_rewards/code_complexity_reward/std": 0.05085850486531854, "eval_rewards/code_execution_reward/mean": 0.28625, "eval_rewards/code_execution_reward/std": 0.17372595340013505, "eval_rewards/code_syntax_reward/mean": 0.4925, "eval_rewards/code_syntax_reward/std": 0.019992219507694243, "eval_rewards/reasoning_present_reward_func/mean": 0.0998750015348196, "eval_rewards/reasoning_present_reward_func/std": 0.000353553406894207, "eval_rewards/xmlcount_reward_func/mean": 0.49984375, "eval_rewards/xmlcount_reward_func/std": 0.0004419417306780815, "eval_runtime": 748.8472, "eval_samples_per_second": 0.134, "eval_steps_per_second": 0.017, "step": 1500 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 320.0, "completions/max_terminated_length": 320.0, "completions/mean_length": 121.765625, "completions/mean_terminated_length": 121.765625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2678085002116859, "epoch": 0.8552706552706553, "frac_reward_zero_std": 0.25, "grad_norm": 0.1181846484541893, "kl": 0.22093796939589083, "learning_rate": 3.14912588144575e-07, "loss": 0.00110532040707767, "num_tokens": 221955681.0, "reward": 2.27490234375, "reward_std": 0.4782857894897461, "rewards/code_complexity_reward/mean": 0.9639648795127869, "rewards/code_complexity_reward/std": 0.14123201370239258, "rewards/code_execution_reward/mean": 0.220703125, "rewards/code_execution_reward/std": 0.4151262938976288, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1501, "step_time": 44.50747496820986 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 262.0, "completions/max_terminated_length": 262.0, "completions/mean_length": 108.943359375, "completions/mean_terminated_length": 108.943359375, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.2410144319292158, "epoch": 0.8558404558404559, "frac_reward_zero_std": 0.265625, "grad_norm": 0.11808108538389206, "kl": 0.2383680024649948, "learning_rate": 3.125002209620787e-07, "loss": 0.0011924374848604202, "num_tokens": 222079924.0, "reward": 2.395263671875, "reward_std": 0.4990580976009369, "rewards/code_complexity_reward/mean": 0.9783203601837158, "rewards/code_complexity_reward/std": 0.10197995603084564, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1502, "step_time": 38.138971597887576 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 391.0, "completions/max_terminated_length": 391.0, "completions/mean_length": 117.74609375, "completions/mean_terminated_length": 117.74609375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2612687081564218, "epoch": 0.8564102564102564, "frac_reward_zero_std": 0.234375, "grad_norm": 0.10076297074556351, "kl": 0.2352157502900809, "learning_rate": 3.100965130910261e-07, "loss": 0.0011761575005948544, "num_tokens": 222208786.0, "reward": 2.354199171066284, "reward_std": 0.5143793225288391, "rewards/code_complexity_reward/mean": 0.9641602039337158, "rewards/code_complexity_reward/std": 0.13329952955245972, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1503, "step_time": 40.77099897712469 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 309.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 118.27734375, "completions/mean_terminated_length": 118.27734375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2420435610692948, "epoch": 0.856980056980057, "frac_reward_zero_std": 0.15625, "grad_norm": 0.10943735390901566, "kl": 0.2801830095704645, "learning_rate": 3.077014740465986e-07, "loss": 0.001402262132614851, "num_tokens": 222339600.0, "reward": 2.3824706077575684, "reward_std": 0.5154193639755249, "rewards/code_complexity_reward/mean": 0.97021484375, "rewards/code_complexity_reward/std": 0.12577180564403534, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1504, "step_time": 38.16856261342764 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 118.6015625, "completions/mean_terminated_length": 118.6015625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24170946655794978, "epoch": 0.8575498575498576, "frac_reward_zero_std": 0.25, "grad_norm": 0.09212107211351395, "kl": 0.2102734842337668, "learning_rate": 3.053151133096599e-07, "loss": 0.0010518692433834076, "num_tokens": 222471548.0, "reward": 2.3279786109924316, "reward_std": 0.5061434507369995, "rewards/code_complexity_reward/mean": 0.9671875238418579, "rewards/code_complexity_reward/std": 0.13970719277858734, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1505, "step_time": 38.97388267982751 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 345.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 121.263671875, "completions/mean_terminated_length": 121.263671875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.26050226856023073, "epoch": 0.8581196581196581, "frac_reward_zero_std": 0.328125, "grad_norm": 0.0999058410525322, "kl": 0.23704292811453342, "learning_rate": 3.029374403267216e-07, "loss": 0.0011857441859319806, "num_tokens": 222600827.0, "reward": 2.3560547828674316, "reward_std": 0.48822978138923645, "rewards/code_complexity_reward/mean": 0.973828136920929, "rewards/code_complexity_reward/std": 0.10350319743156433, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1506, "step_time": 37.34934572316706 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 114.69140625, "completions/mean_terminated_length": 114.69140625, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.2416015760973096, "epoch": 0.8586894586894587, "frac_reward_zero_std": 0.3125, "grad_norm": 0.11582738906145096, "kl": 0.24638982769101858, "learning_rate": 3.0056846450990387e-07, "loss": 0.0012325821444392204, "num_tokens": 222729717.0, "reward": 2.326855421066284, "reward_std": 0.578822672367096, "rewards/code_complexity_reward/mean": 0.942675769329071, "rewards/code_complexity_reward/std": 0.19793471693992615, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1507, "step_time": 43.23651710525155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 459.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 118.115234375, "completions/mean_terminated_length": 118.115234375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2485833850223571, "epoch": 0.8592592592592593, "frac_reward_zero_std": 0.265625, "grad_norm": 0.10549618303775787, "kl": 0.22529539931565523, "learning_rate": 2.982081952368984e-07, "loss": 0.0011270049726590514, "num_tokens": 222859744.0, "reward": 2.2714357376098633, "reward_std": 0.5327032804489136, "rewards/code_complexity_reward/mean": 0.94921875, "rewards/code_complexity_reward/std": 0.18391633033752441, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1508, "step_time": 44.36533812712878 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 115.6953125, "completions/mean_terminated_length": 115.6953125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2475648787803948, "epoch": 0.8598290598290599, "frac_reward_zero_std": 0.203125, "grad_norm": 0.14279966056346893, "kl": 0.21365183521993458, "learning_rate": 2.958566418509329e-07, "loss": 0.0010687489993870258, "num_tokens": 222985452.0, "reward": 2.3751955032348633, "reward_std": 0.4786641001701355, "rewards/code_complexity_reward/mean": 0.9792969226837158, "rewards/code_complexity_reward/std": 0.08139616996049881, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1509, "step_time": 42.631495920941234 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 295.0, "completions/max_terminated_length": 295.0, "completions/mean_length": 116.72265625, "completions/mean_terminated_length": 116.72265625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24476747517473996, "epoch": 0.8603988603988604, "frac_reward_zero_std": 0.203125, "grad_norm": 0.12494901567697525, "kl": 0.2384337002877146, "learning_rate": 2.935138136607313e-07, "loss": 0.0011929338797926903, "num_tokens": 223114070.0, "reward": 2.288378953933716, "reward_std": 0.5130099654197693, "rewards/code_complexity_reward/mean": 0.9566406607627869, "rewards/code_complexity_reward/std": 0.16375260055065155, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1510, "step_time": 43.61504562944174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 271.0, "completions/max_terminated_length": 271.0, "completions/mean_length": 117.009765625, "completions/mean_terminated_length": 117.009765625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24224050273187459, "epoch": 0.860968660968661, "frac_reward_zero_std": 0.1875, "grad_norm": 0.13217349350452423, "kl": 0.24478501314297318, "learning_rate": 2.911797199404795e-07, "loss": 0.001224704086780548, "num_tokens": 223243563.0, "reward": 2.3506836891174316, "reward_std": 0.5040389895439148, "rewards/code_complexity_reward/mean": 0.969433605670929, "rewards/code_complexity_reward/std": 0.1260504573583603, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1511, "step_time": 35.19848250411451 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 117.529296875, "completions/mean_terminated_length": 117.529296875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24632627028040588, "epoch": 0.8615384615384616, "frac_reward_zero_std": 0.34375, "grad_norm": 0.10490667074918747, "kl": 0.22283956152386963, "learning_rate": 2.888543699297866e-07, "loss": 0.0011145648313686252, "num_tokens": 223372642.0, "reward": 2.329296827316284, "reward_std": 0.4650622308254242, "rewards/code_complexity_reward/mean": 0.9753906726837158, "rewards/code_complexity_reward/std": 0.09415315836668015, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1512, "step_time": 36.24158288538456 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 116.35546875, "completions/mean_terminated_length": 115.58121490478516, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.25174859748221934, "epoch": 0.8621082621082621, "frac_reward_zero_std": 0.265625, "grad_norm": 0.10048829019069672, "kl": 0.2304051520768553, "learning_rate": 2.865377728336513e-07, "loss": 0.0011519542895257473, "num_tokens": 223500144.0, "reward": 2.3252439498901367, "reward_std": 0.4804633855819702, "rewards/code_complexity_reward/mean": 0.9754883050918579, "rewards/code_complexity_reward/std": 0.11050388962030411, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1513, "step_time": 60.68265695963055 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 110.169921875, "completions/mean_terminated_length": 109.38356018066406, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22310464037582278, "epoch": 0.8626780626780627, "frac_reward_zero_std": 0.25, "grad_norm": 0.09364912658929825, "kl": 0.2578521315008402, "learning_rate": 2.8422993782242117e-07, "loss": 0.001289203530177474, "num_tokens": 223625631.0, "reward": 2.460888624191284, "reward_std": 0.5319395661354065, "rewards/code_complexity_reward/mean": 0.9714844226837158, "rewards/code_complexity_reward/std": 0.10670910775661469, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1514, "step_time": 58.49024568218738 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 274.0, "completions/max_terminated_length": 274.0, "completions/mean_length": 112.625, "completions/mean_terminated_length": 112.625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24047818803228438, "epoch": 0.8632478632478633, "frac_reward_zero_std": 0.203125, "grad_norm": 0.10153000801801682, "kl": 0.24861057149246335, "learning_rate": 2.8193087403176027e-07, "loss": 0.0012437893310561776, "num_tokens": 223751031.0, "reward": 2.3466796875, "reward_std": 0.4906849265098572, "rewards/code_complexity_reward/mean": 0.9712890386581421, "rewards/code_complexity_reward/std": 0.11163216084241867, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1515, "step_time": 33.37630357872695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 281.0, "completions/max_terminated_length": 281.0, "completions/mean_length": 112.384765625, "completions/mean_terminated_length": 112.384765625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24970851466059685, "epoch": 0.8638176638176638, "frac_reward_zero_std": 0.375, "grad_norm": 0.10935904085636139, "kl": 0.23805882059969008, "learning_rate": 2.796405905626118e-07, "loss": 0.0011904994025826454, "num_tokens": 223880356.0, "reward": 2.3525876998901367, "reward_std": 0.5303825736045837, "rewards/code_complexity_reward/mean": 0.9657226800918579, "rewards/code_complexity_reward/std": 0.15257689356803894, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1516, "step_time": 34.84594589471817 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 307.0, "completions/max_terminated_length": 307.0, "completions/mean_length": 112.7734375, "completions/mean_terminated_length": 112.7734375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24725092435255647, "epoch": 0.8643874643874644, "frac_reward_zero_std": 0.34375, "grad_norm": 0.0927039161324501, "kl": 0.24569730879738927, "learning_rate": 2.773590964811593e-07, "loss": 0.001229285728186369, "num_tokens": 224007416.0, "reward": 2.4095215797424316, "reward_std": 0.5210428833961487, "rewards/code_complexity_reward/mean": 0.9750000238418579, "rewards/code_complexity_reward/std": 0.11919495463371277, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1517, "step_time": 35.526643798686564 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 123.55078125, "completions/mean_terminated_length": 123.55078125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23863839777186513, "epoch": 0.864957264957265, "frac_reward_zero_std": 0.265625, "grad_norm": 0.12467636913061142, "kl": 0.24577750847674906, "learning_rate": 2.750864008187959e-07, "loss": 0.0012294140178710222, "num_tokens": 224137482.0, "reward": 2.360644578933716, "reward_std": 0.511204183101654, "rewards/code_complexity_reward/mean": 0.9657226800918579, "rewards/code_complexity_reward/std": 0.12837180495262146, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1518, "step_time": 40.26297125779092 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 420.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 115.51953125, "completions/mean_terminated_length": 115.51953125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2547068193089217, "epoch": 0.8655270655270655, "frac_reward_zero_std": 0.3125, "grad_norm": 0.12146870791912079, "kl": 0.2751749709714204, "learning_rate": 2.7282251257208376e-07, "loss": 0.0013752905651926994, "num_tokens": 224266644.0, "reward": 2.386913776397705, "reward_std": 0.4941582977771759, "rewards/code_complexity_reward/mean": 0.978320300579071, "rewards/code_complexity_reward/std": 0.09268083423376083, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1519, "step_time": 42.16505352128297 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 116.7890625, "completions/mean_terminated_length": 116.7890625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23804796300828457, "epoch": 0.8660968660968661, "frac_reward_zero_std": 0.328125, "grad_norm": 0.12047483026981354, "kl": 0.23552872030995786, "learning_rate": 2.705674407027223e-07, "loss": 0.0011780315544456244, "num_tokens": 224398472.0, "reward": 2.345165967941284, "reward_std": 0.4918536841869354, "rewards/code_complexity_reward/mean": 0.9765625, "rewards/code_complexity_reward/std": 0.10982774198055267, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 1520, "step_time": 44.69204773940146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 118.380859375, "completions/mean_terminated_length": 118.380859375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24846225837245584, "epoch": 0.8666666666666667, "frac_reward_zero_std": 0.15625, "grad_norm": 0.11196547746658325, "kl": 0.23045493056997657, "learning_rate": 2.6832119413750886e-07, "loss": 0.0011525468435138464, "num_tokens": 224526387.0, "reward": 2.3727052211761475, "reward_std": 0.5046683549880981, "rewards/code_complexity_reward/mean": 0.97119140625, "rewards/code_complexity_reward/std": 0.11814427375793457, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1521, "step_time": 43.774856217205524 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 269.0, "completions/max_terminated_length": 269.0, "completions/mean_length": 112.65625, "completions/mean_terminated_length": 112.65625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2500302332919091, "epoch": 0.8672364672364672, "frac_reward_zero_std": 0.203125, "grad_norm": 0.1450476199388504, "kl": 0.34663223125971854, "learning_rate": 2.6608378176830734e-07, "loss": 0.001734327059239149, "num_tokens": 224650027.0, "reward": 2.4116697311401367, "reward_std": 0.5146406292915344, "rewards/code_complexity_reward/mean": 0.9781250357627869, "rewards/code_complexity_reward/std": 0.10952664911746979, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1522, "step_time": 40.597109600901604 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 329.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 117.96484375, "completions/mean_terminated_length": 117.96484375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2492500487715006, "epoch": 0.8678062678062678, "frac_reward_zero_std": 0.265625, "grad_norm": 0.10487913340330124, "kl": 0.22357957856729627, "learning_rate": 2.6385521245201053e-07, "loss": 0.0011185683542862535, "num_tokens": 224780329.0, "reward": 2.33935546875, "reward_std": 0.5273618698120117, "rewards/code_complexity_reward/mean": 0.9639648199081421, "rewards/code_complexity_reward/std": 0.151848703622818, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1523, "step_time": 37.23633710667491 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 113.810546875, "completions/mean_terminated_length": 113.03131103515625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23673099605366588, "epoch": 0.8683760683760684, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09539761394262314, "kl": 0.24680209066718817, "learning_rate": 2.616354950105046e-07, "loss": 0.0012346726143732667, "num_tokens": 224903600.0, "reward": 2.3411622047424316, "reward_std": 0.4960078001022339, "rewards/code_complexity_reward/mean": 0.9728515148162842, "rewards/code_complexity_reward/std": 0.12096834182739258, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1524, "step_time": 48.610319428145885 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 113.38671875, "completions/mean_terminated_length": 113.38671875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2535761788021773, "epoch": 0.8689458689458689, "frac_reward_zero_std": 0.3125, "grad_norm": 0.12268434464931488, "kl": 0.23515770537778735, "learning_rate": 2.594246382306356e-07, "loss": 0.0011765402741730213, "num_tokens": 225028662.0, "reward": 2.337353467941284, "reward_std": 0.4975944459438324, "rewards/code_complexity_reward/mean": 0.97021484375, "rewards/code_complexity_reward/std": 0.1260826140642166, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1525, "step_time": 45.24201289471239 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 119.7734375, "completions/mean_terminated_length": 119.00586700439453, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.2521148936357349, "epoch": 0.8695156695156695, "frac_reward_zero_std": 0.15625, "grad_norm": 0.12695039808750153, "kl": 0.23009428498335183, "learning_rate": 2.5722265086417594e-07, "loss": 0.0011512478813529015, "num_tokens": 225157802.0, "reward": 2.345752000808716, "reward_std": 0.5294046998023987, "rewards/code_complexity_reward/mean": 0.9595702886581421, "rewards/code_complexity_reward/std": 0.15348593890666962, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1526, "step_time": 48.469860075972974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 255.0, "completions/mean_length": 113.03515625, "completions/mean_terminated_length": 112.25440216064453, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24552414030767977, "epoch": 0.8700854700854701, "frac_reward_zero_std": 0.234375, "grad_norm": 0.12027189880609512, "kl": 0.24072239105589688, "learning_rate": 2.550295416277862e-07, "loss": 0.0012042011367157102, "num_tokens": 225283676.0, "reward": 2.3448729515075684, "reward_std": 0.4897174835205078, "rewards/code_complexity_reward/mean": 0.970410168170929, "rewards/code_complexity_reward/std": 0.11120416969060898, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 1527, "step_time": 50.16400516219437 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 277.0, "completions/max_terminated_length": 277.0, "completions/mean_length": 117.267578125, "completions/mean_terminated_length": 117.267578125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.25453924015164375, "epoch": 0.8706552706552707, "frac_reward_zero_std": 0.328125, "grad_norm": 0.13497160375118256, "kl": 0.21886823908425868, "learning_rate": 2.528453192029828e-07, "loss": 0.001095048151910305, "num_tokens": 225412869.0, "reward": 2.309814453125, "reward_std": 0.5001052618026733, "rewards/code_complexity_reward/mean": 0.966113269329071, "rewards/code_complexity_reward/std": 0.1399228572845459, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1528, "step_time": 40.53741108160466 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 112.58984375, "completions/mean_terminated_length": 112.58984375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2466233060695231, "epoch": 0.8712250712250712, "frac_reward_zero_std": 0.28125, "grad_norm": 0.1303713619709015, "kl": 0.24855810846202075, "learning_rate": 2.5066999223610443e-07, "loss": 0.0012431945651769638, "num_tokens": 225541155.0, "reward": 2.2895994186401367, "reward_std": 0.4811418652534485, "rewards/code_complexity_reward/mean": 0.9703125357627869, "rewards/code_complexity_reward/std": 0.13489685952663422, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1529, "step_time": 38.0772570008412 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 117.07421875, "completions/mean_terminated_length": 116.3013687133789, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24956439109519124, "epoch": 0.8717948717948718, "frac_reward_zero_std": 0.1875, "grad_norm": 0.14195138216018677, "kl": 0.23980167298577726, "learning_rate": 2.4850356933827506e-07, "loss": 0.0011998366098850965, "num_tokens": 225668537.0, "reward": 2.288330078125, "reward_std": 0.47669440507888794, "rewards/code_complexity_reward/mean": 0.9708007574081421, "rewards/code_complexity_reward/std": 0.13271935284137726, "rewards/code_execution_reward/mean": 0.2265625, "rewards/code_execution_reward/std": 0.4190165400505066, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1530, "step_time": 58.43494268972427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 122.21875, "completions/mean_terminated_length": 121.45597076416016, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23557456629350781, "epoch": 0.8723646723646724, "frac_reward_zero_std": 0.234375, "grad_norm": 0.11344364285469055, "kl": 0.22901567188091576, "learning_rate": 2.4634605908537333e-07, "loss": 0.0011456221109256148, "num_tokens": 225799089.0, "reward": 2.3244142532348633, "reward_std": 0.4828791916370392, "rewards/code_complexity_reward/mean": 0.9709961414337158, "rewards/code_complexity_reward/std": 0.11310241371393204, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1531, "step_time": 55.464938192628324 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 457.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 119.02734375, "completions/mean_terminated_length": 119.02734375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24666665424592793, "epoch": 0.8729344729344729, "frac_reward_zero_std": 0.375, "grad_norm": 0.10000097006559372, "kl": 0.22800974640995264, "learning_rate": 2.4419747001799496e-07, "loss": 0.0011403555981814861, "num_tokens": 225931863.0, "reward": 2.3963377475738525, "reward_std": 0.5046771764755249, "rewards/code_complexity_reward/mean": 0.9774414300918579, "rewards/code_complexity_reward/std": 0.10329169780015945, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1532, "step_time": 54.85863465629518 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 118.103515625, "completions/mean_terminated_length": 115.78192901611328, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.24600683827884495, "epoch": 0.8735042735042735, "frac_reward_zero_std": 0.28125, "grad_norm": 0.13113658130168915, "kl": 0.2381613776087761, "learning_rate": 2.420578106414223e-07, "loss": 0.001191501971334219, "num_tokens": 226060244.0, "reward": 2.276904344558716, "reward_std": 0.4751274585723877, "rewards/code_complexity_reward/mean": 0.9715820550918579, "rewards/code_complexity_reward/std": 0.13303633034229279, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1533, "step_time": 56.69047044031322 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 118.6953125, "completions/mean_terminated_length": 118.6953125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23755299672484398, "epoch": 0.8740740740740741, "frac_reward_zero_std": 0.359375, "grad_norm": 0.11482750624418259, "kl": 0.23534311749972403, "learning_rate": 2.399270894255878e-07, "loss": 0.0011776898754760623, "num_tokens": 226189152.0, "reward": 2.329296827316284, "reward_std": 0.43734776973724365, "rewards/code_complexity_reward/mean": 0.988085925579071, "rewards/code_complexity_reward/std": 0.051157332956790924, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1534, "step_time": 38.58265405986458 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 391.0, "completions/max_terminated_length": 391.0, "completions/mean_length": 116.70703125, "completions/mean_terminated_length": 116.70703125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.252152340952307, "epoch": 0.8746438746438746, "frac_reward_zero_std": 0.34375, "grad_norm": 0.12736110389232635, "kl": 0.23574157408438623, "learning_rate": 2.378053148050438e-07, "loss": 0.001179218990728259, "num_tokens": 226318690.0, "reward": 2.3259763717651367, "reward_std": 0.4989215135574341, "rewards/code_complexity_reward/mean": 0.9671875238418579, "rewards/code_complexity_reward/std": 0.13449731469154358, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1535, "step_time": 40.60936237219721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 284.0, "completions/max_terminated_length": 284.0, "completions/mean_length": 114.94140625, "completions/mean_terminated_length": 114.94140625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2438255858141929, "epoch": 0.8752136752136752, "frac_reward_zero_std": 0.3125, "grad_norm": 0.09335491061210632, "kl": 0.23962674289941788, "learning_rate": 2.3569249517892468e-07, "loss": 0.0011986527824774384, "num_tokens": 226443700.0, "reward": 2.363037109375, "reward_std": 0.5286990404129028, "rewards/code_complexity_reward/mean": 0.9666016101837158, "rewards/code_complexity_reward/std": 0.1401626616716385, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1536, "step_time": 34.25771015044302 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 363.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 120.09375, "completions/mean_terminated_length": 120.09375, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.2414596900343895, "epoch": 0.8757834757834758, "frac_reward_zero_std": 0.265625, "grad_norm": 0.12233467400074005, "kl": 0.22314912057481706, "learning_rate": 2.3358863891091765e-07, "loss": 0.0011160292197018862, "num_tokens": 226576756.0, "reward": 2.4286131858825684, "reward_std": 0.5581775903701782, "rewards/code_complexity_reward/mean": 0.95947265625, "rewards/code_complexity_reward/std": 0.15232418477535248, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1537, "step_time": 42.39285864122212 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 459.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 115.833984375, "completions/mean_terminated_length": 115.833984375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24036665889434516, "epoch": 0.8763532763532763, "frac_reward_zero_std": 0.265625, "grad_norm": 0.12295787036418915, "kl": 0.2289760885760188, "learning_rate": 2.314937543292281e-07, "loss": 0.001145064365118742, "num_tokens": 226703479.0, "reward": 2.3218750953674316, "reward_std": 0.5036752820014954, "rewards/code_complexity_reward/mean": 0.9669921398162842, "rewards/code_complexity_reward/std": 0.13441303372383118, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1538, "step_time": 45.10238826088607 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 124.314453125, "completions/mean_terminated_length": 124.314453125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2395035291556269, "epoch": 0.8769230769230769, "frac_reward_zero_std": 0.3125, "grad_norm": 0.10724388062953949, "kl": 0.23696639016270638, "learning_rate": 2.2940784972654534e-07, "loss": 0.0011853629257529974, "num_tokens": 226839032.0, "reward": 2.3575193881988525, "reward_std": 0.5076875686645508, "rewards/code_complexity_reward/mean": 0.9686523675918579, "rewards/code_complexity_reward/std": 0.11992635577917099, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 1539, "step_time": 43.054856549948454 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 253.0, "completions/max_terminated_length": 253.0, "completions/mean_length": 114.51171875, "completions/mean_terminated_length": 114.51171875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24303846876136959, "epoch": 0.8774928774928775, "frac_reward_zero_std": 0.375, "grad_norm": 0.09962659329175949, "kl": 0.2389188597444445, "learning_rate": 2.2733093336001323e-07, "loss": 0.0011949667241424322, "num_tokens": 226965718.0, "reward": 2.4205079078674316, "reward_std": 0.5110130906105042, "rewards/code_complexity_reward/mean": 0.977734386920929, "rewards/code_complexity_reward/std": 0.10214118659496307, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1540, "step_time": 33.41113889217377 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 245.0, "completions/max_terminated_length": 245.0, "completions/mean_length": 109.685546875, "completions/mean_terminated_length": 109.685546875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2405647011473775, "epoch": 0.878062678062678, "frac_reward_zero_std": 0.390625, "grad_norm": 0.08934968709945679, "kl": 0.29116048058494925, "learning_rate": 2.2526301345119295e-07, "loss": 0.0014552271459251642, "num_tokens": 227087605.0, "reward": 2.3905272483825684, "reward_std": 0.48860105872154236, "rewards/code_complexity_reward/mean": 0.98388671875, "rewards/code_complexity_reward/std": 0.09038475155830383, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1541, "step_time": 38.657145702280104 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 396.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 114.822265625, "completions/mean_terminated_length": 114.822265625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.25573731306940317, "epoch": 0.8786324786324786, "frac_reward_zero_std": 0.296875, "grad_norm": 0.08574899286031723, "kl": 0.25758677651174366, "learning_rate": 2.2320409818603478e-07, "loss": 0.001288228784687817, "num_tokens": 227214714.0, "reward": 2.330078125, "reward_std": 0.5594531297683716, "rewards/code_complexity_reward/mean": 0.9546874761581421, "rewards/code_complexity_reward/std": 0.18446913361549377, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1542, "step_time": 51.34434133581817 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 110.0546875, "completions/mean_terminated_length": 109.26810455322266, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.25074276025407016, "epoch": 0.8792022792022792, "frac_reward_zero_std": 0.375, "grad_norm": 0.10440237820148468, "kl": 0.22391554387286305, "learning_rate": 2.2115419571484193e-07, "loss": 0.0011201680172234774, "num_tokens": 227340422.0, "reward": 2.4019529819488525, "reward_std": 0.5133573412895203, "rewards/code_complexity_reward/mean": 0.9803711175918579, "rewards/code_complexity_reward/std": 0.10934987664222717, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1543, "step_time": 59.2794208759442 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 268.0, "completions/max_terminated_length": 268.0, "completions/mean_length": 113.52734375, "completions/mean_terminated_length": 113.52734375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.25092597655020654, "epoch": 0.8797720797720797, "frac_reward_zero_std": 0.359375, "grad_norm": 0.10289472341537476, "kl": 0.23075835360214114, "learning_rate": 2.1911331415224196e-07, "loss": 0.0011541617568582296, "num_tokens": 227467284.0, "reward": 2.315185546875, "reward_std": 0.4564659595489502, "rewards/code_complexity_reward/mean": 0.9820312261581421, "rewards/code_complexity_reward/std": 0.1003875806927681, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1544, "step_time": 42.954444508999586 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 378.0, "completions/max_terminated_length": 378.0, "completions/mean_length": 123.21484375, "completions/mean_terminated_length": 123.21484375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2603917964734137, "epoch": 0.8803418803418803, "frac_reward_zero_std": 0.296875, "grad_norm": 0.13965453207492828, "kl": 0.23012289335019886, "learning_rate": 2.1708146157715132e-07, "loss": 0.0011512679047882557, "num_tokens": 227600786.0, "reward": 2.295849323272705, "reward_std": 0.44924768805503845, "rewards/code_complexity_reward/mean": 0.9765625, "rewards/code_complexity_reward/std": 0.10269127041101456, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1545, "step_time": 72.94929880090058 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 115.4453125, "completions/mean_terminated_length": 114.66927337646484, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.2412831608671695, "epoch": 0.8809116809116809, "frac_reward_zero_std": 0.203125, "grad_norm": 0.14464876055717468, "kl": 0.24191062641330063, "learning_rate": 2.15058646032745e-07, "loss": 0.0012102739419788122, "num_tokens": 227728878.0, "reward": 2.314990282058716, "reward_std": 0.5484377145767212, "rewards/code_complexity_reward/mean": 0.9564452767372131, "rewards/code_complexity_reward/std": 0.17944103479385376, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1546, "step_time": 55.89473673887551 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 115.23828125, "completions/mean_terminated_length": 114.46183776855469, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24292326136492193, "epoch": 0.8814814814814815, "frac_reward_zero_std": 0.3125, "grad_norm": 0.09935322403907776, "kl": 0.2296211402863264, "learning_rate": 2.1304487552642612e-07, "loss": 0.0011489446042105556, "num_tokens": 227855944.0, "reward": 2.339599609375, "reward_std": 0.4686064124107361, "rewards/code_complexity_reward/mean": 0.98095703125, "rewards/code_complexity_reward/std": 0.09254851192235947, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1547, "step_time": 58.65830635651946 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 315.0, "completions/max_terminated_length": 315.0, "completions/mean_length": 120.689453125, "completions/mean_terminated_length": 120.689453125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24491168442182243, "epoch": 0.882051282051282, "frac_reward_zero_std": 0.25, "grad_norm": 0.10088816285133362, "kl": 0.22247042530216277, "learning_rate": 2.110401580297902e-07, "loss": 0.0011128181358799338, "num_tokens": 227987241.0, "reward": 2.2610838413238525, "reward_std": 0.4662802219390869, "rewards/code_complexity_reward/mean": 0.9630858898162842, "rewards/code_complexity_reward/std": 0.13793504238128662, "rewards/code_execution_reward/mean": 0.20703125, "rewards/code_execution_reward/std": 0.40557438135147095, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1548, "step_time": 35.793976549990475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 331.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 109.84765625, "completions/mean_terminated_length": 109.84765625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23426757799461484, "epoch": 0.8826210826210826, "frac_reward_zero_std": 0.328125, "grad_norm": 0.10817507654428482, "kl": 0.24382159463129938, "learning_rate": 2.0904450147859772e-07, "loss": 0.0012198139447718859, "num_tokens": 228110355.0, "reward": 2.4050779342651367, "reward_std": 0.4976300001144409, "rewards/code_complexity_reward/mean": 0.9828125238418579, "rewards/code_complexity_reward/std": 0.09219809621572495, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1549, "step_time": 42.76493265107274 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 106.455078125, "completions/mean_terminated_length": 105.66144561767578, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23645786731503904, "epoch": 0.8831908831908832, "frac_reward_zero_std": 0.34375, "grad_norm": 0.10019472241401672, "kl": 0.2347307086456567, "learning_rate": 2.0705791377273993e-07, "loss": 0.0011736599262803793, "num_tokens": 228232004.0, "reward": 2.438915967941284, "reward_std": 0.5355059504508972, "rewards/code_complexity_reward/mean": 0.9719727039337158, "rewards/code_complexity_reward/std": 0.12633094191551208, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1550, "step_time": 55.693617053329945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 324.0, "completions/max_terminated_length": 324.0, "completions/mean_length": 107.962890625, "completions/mean_terminated_length": 107.962890625, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.2394268186762929, "epoch": 0.8837606837606837, "frac_reward_zero_std": 0.390625, "grad_norm": 0.0883922427892685, "kl": 0.2528938625473529, "learning_rate": 2.0508040277620989e-07, "loss": 0.0012646280229091644, "num_tokens": 228354857.0, "reward": 2.4652342796325684, "reward_std": 0.5084918141365051, "rewards/code_complexity_reward/mean": 0.984375, "rewards/code_complexity_reward/std": 0.08036121726036072, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 1551, "step_time": 38.00013398099691 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 118.587890625, "completions/mean_terminated_length": 117.81800079345703, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24375138408504426, "epoch": 0.8843304843304843, "frac_reward_zero_std": 0.328125, "grad_norm": 0.09180878102779388, "kl": 0.2304659546352923, "learning_rate": 2.0311197631706886e-07, "loss": 0.0011526872403919697, "num_tokens": 228483758.0, "reward": 2.4641599655151367, "reward_std": 0.5318589210510254, "rewards/code_complexity_reward/mean": 0.9751952886581421, "rewards/code_complexity_reward/std": 0.10505781322717667, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1552, "step_time": 47.66419192124158 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 116.984375, "completions/mean_terminated_length": 116.21134948730469, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24284847965463996, "epoch": 0.8849002849002849, "frac_reward_zero_std": 0.28125, "grad_norm": 0.11236079037189484, "kl": 0.24437801516614854, "learning_rate": 2.0115264218741798e-07, "loss": 0.001222499180585146, "num_tokens": 228612742.0, "reward": 2.3551268577575684, "reward_std": 0.5149376392364502, "rewards/code_complexity_reward/mean": 0.97119140625, "rewards/code_complexity_reward/std": 0.13386891782283783, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1553, "step_time": 48.62475370708853 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 117.763671875, "completions/mean_terminated_length": 117.763671875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.25172798288986087, "epoch": 0.8854700854700854, "frac_reward_zero_std": 0.359375, "grad_norm": 0.08198084682226181, "kl": 0.24294987879693508, "learning_rate": 1.9920240814336412e-07, "loss": 0.0012149029644206166, "num_tokens": 228740341.0, "reward": 2.3172850608825684, "reward_std": 0.47032177448272705, "rewards/code_complexity_reward/mean": 0.97509765625, "rewards/code_complexity_reward/std": 0.11094678193330765, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1554, "step_time": 41.17122540343553 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 439.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 112.884765625, "completions/mean_terminated_length": 112.884765625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2428789830300957, "epoch": 0.886039886039886, "frac_reward_zero_std": 0.34375, "grad_norm": 0.13641579449176788, "kl": 0.2870614044368267, "learning_rate": 1.972612819049932e-07, "loss": 0.001436011167243123, "num_tokens": 228866850.0, "reward": 2.3294920921325684, "reward_std": 0.5103041529655457, "rewards/code_complexity_reward/mean": 0.9677734375, "rewards/code_complexity_reward/std": 0.14130550622940063, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1555, "step_time": 56.62482966855168 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 352.0, "completions/max_terminated_length": 352.0, "completions/mean_length": 112.666015625, "completions/mean_terminated_length": 112.666015625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24828543653711677, "epoch": 0.8866096866096866, "frac_reward_zero_std": 0.40625, "grad_norm": 0.1029304638504982, "kl": 0.2526621257420629, "learning_rate": 1.953292711563362e-07, "loss": 0.0012638876214623451, "num_tokens": 228992839.0, "reward": 2.3839354515075684, "reward_std": 0.48800379037857056, "rewards/code_complexity_reward/mean": 0.981640636920929, "rewards/code_complexity_reward/std": 0.09229008108377457, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1556, "step_time": 46.418326993472874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 254.0, "completions/max_terminated_length": 254.0, "completions/mean_length": 109.01953125, "completions/mean_terminated_length": 109.01953125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24222791125066578, "epoch": 0.8871794871794871, "frac_reward_zero_std": 0.265625, "grad_norm": 0.10200370848178864, "kl": 0.2637674678117037, "learning_rate": 1.9340638354533981e-07, "loss": 0.0013189398450776935, "num_tokens": 229115689.0, "reward": 2.434033155441284, "reward_std": 0.5373707413673401, "rewards/code_complexity_reward/mean": 0.97314453125, "rewards/code_complexity_reward/std": 0.12565474212169647, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1557, "step_time": 56.10587207879871 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 317.0, "completions/max_terminated_length": 317.0, "completions/mean_length": 110.64453125, "completions/mean_terminated_length": 110.64453125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2515884900931269, "epoch": 0.8877492877492877, "frac_reward_zero_std": 0.296875, "grad_norm": 0.10982188582420349, "kl": 0.23219402926042676, "learning_rate": 1.9149262668383768e-07, "loss": 0.0011614016257226467, "num_tokens": 229240459.0, "reward": 2.3975586891174316, "reward_std": 0.4937068521976471, "rewards/code_complexity_reward/mean": 0.979199230670929, "rewards/code_complexity_reward/std": 0.09136893600225449, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1558, "step_time": 36.82224499434233 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 394.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 118.744140625, "completions/mean_terminated_length": 118.744140625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.25294399727135897, "epoch": 0.8883190883190883, "frac_reward_zero_std": 0.3125, "grad_norm": 0.1199958324432373, "kl": 0.23353812866844237, "learning_rate": 1.8958800814751737e-07, "loss": 0.0011684902710840106, "num_tokens": 229370960.0, "reward": 2.3467283248901367, "reward_std": 0.5209366679191589, "rewards/code_complexity_reward/mean": 0.9678710699081421, "rewards/code_complexity_reward/std": 0.1408596783876419, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1559, "step_time": 42.33786161709577 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 309.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 111.111328125, "completions/mean_terminated_length": 111.111328125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23596758511848748, "epoch": 0.8888888888888888, "frac_reward_zero_std": 0.421875, "grad_norm": 0.11971012502908707, "kl": 0.2359157339669764, "learning_rate": 1.876925354758935e-07, "loss": 0.001179469982162118, "num_tokens": 229496801.0, "reward": 2.3189454078674316, "reward_std": 0.45081189274787903, "rewards/code_complexity_reward/mean": 0.981640636920929, "rewards/code_complexity_reward/std": 0.08123824000358582, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1560, "step_time": 35.65288659185171 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 266.0, "completions/max_terminated_length": 266.0, "completions/mean_length": 110.396484375, "completions/mean_terminated_length": 110.396484375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.25233052275143564, "epoch": 0.8894586894586894, "frac_reward_zero_std": 0.328125, "grad_norm": 0.11827950924634933, "kl": 0.23685055156238377, "learning_rate": 1.8580621617227568e-07, "loss": 0.0011845112312585115, "num_tokens": 229621732.0, "reward": 2.331737995147705, "reward_std": 0.49158626794815063, "rewards/code_complexity_reward/mean": 0.973925769329071, "rewards/code_complexity_reward/std": 0.12616896629333496, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1561, "step_time": 52.09682976920158 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 278.0, "completions/max_terminated_length": 278.0, "completions/mean_length": 116.505859375, "completions/mean_terminated_length": 116.505859375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23652101773768663, "epoch": 0.89002849002849, "frac_reward_zero_std": 0.34375, "grad_norm": 0.1024155244231224, "kl": 0.23197998735122383, "learning_rate": 1.839290577037392e-07, "loss": 0.0011605564504861832, "num_tokens": 229748767.0, "reward": 2.317138671875, "reward_std": 0.5125293731689453, "rewards/code_complexity_reward/mean": 0.969042956829071, "rewards/code_complexity_reward/std": 0.14592865109443665, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 1562, "step_time": 41.053696412593126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 111.224609375, "completions/mean_terminated_length": 111.224609375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24913461762480438, "epoch": 0.8905982905982905, "frac_reward_zero_std": 0.3125, "grad_norm": 0.09986277669668198, "kl": 0.22749623330309987, "learning_rate": 1.8206106750109586e-07, "loss": 0.0011379118077456951, "num_tokens": 229873434.0, "reward": 2.307812452316284, "reward_std": 0.463731050491333, "rewards/code_complexity_reward/mean": 0.9753906726837158, "rewards/code_complexity_reward/std": 0.11094614118337631, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1563, "step_time": 40.35340981092304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 252.0, "completions/max_terminated_length": 252.0, "completions/mean_length": 112.076171875, "completions/mean_terminated_length": 112.076171875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24678747239522636, "epoch": 0.8911680911680911, "frac_reward_zero_std": 0.28125, "grad_norm": 0.11312607675790787, "kl": 0.25955664878711104, "learning_rate": 1.8020225295886568e-07, "loss": 0.0012980920728296041, "num_tokens": 230000297.0, "reward": 2.3142576217651367, "reward_std": 0.4944722056388855, "rewards/code_complexity_reward/mean": 0.9681640863418579, "rewards/code_complexity_reward/std": 0.14077000319957733, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1564, "step_time": 33.70184741728008 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 117.423828125, "completions/mean_terminated_length": 117.423828125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.23859159369021654, "epoch": 0.8917378917378918, "frac_reward_zero_std": 0.265625, "grad_norm": 0.13448306918144226, "kl": 0.23865308659151196, "learning_rate": 1.7835262143524463e-07, "loss": 0.0011936626397073269, "num_tokens": 230129434.0, "reward": 2.286328077316284, "reward_std": 0.5202443599700928, "rewards/code_complexity_reward/mean": 0.958789050579071, "rewards/code_complexity_reward/std": 0.16938041150569916, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1565, "step_time": 45.3935819119215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 411.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 111.357421875, "completions/mean_terminated_length": 111.357421875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24151936429552734, "epoch": 0.8923076923076924, "frac_reward_zero_std": 0.28125, "grad_norm": 0.10132500529289246, "kl": 0.23467145999893546, "learning_rate": 1.7651218025207833e-07, "loss": 0.0011739266337826848, "num_tokens": 230254377.0, "reward": 2.454785108566284, "reward_std": 0.5335986614227295, "rewards/code_complexity_reward/mean": 0.976855456829071, "rewards/code_complexity_reward/std": 0.11802548170089722, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1566, "step_time": 41.99395354092121 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 500.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 116.4140625, "completions/mean_terminated_length": 116.4140625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23898522043600678, "epoch": 0.8928774928774929, "frac_reward_zero_std": 0.28125, "grad_norm": 0.09789708256721497, "kl": 0.23289611632935703, "learning_rate": 1.7468093669483294e-07, "loss": 0.0011647811625152826, "num_tokens": 230384477.0, "reward": 2.3561034202575684, "reward_std": 0.5268110632896423, "rewards/code_complexity_reward/mean": 0.9640624523162842, "rewards/code_complexity_reward/std": 0.14330287277698517, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1567, "step_time": 49.76546695642173 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 273.0, "completions/max_terminated_length": 273.0, "completions/mean_length": 110.123046875, "completions/mean_terminated_length": 110.123046875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2402895197737962, "epoch": 0.8934472934472935, "frac_reward_zero_std": 0.390625, "grad_norm": 0.10411947965621948, "kl": 0.2501856591552496, "learning_rate": 1.728588980125634e-07, "loss": 0.0012515339767560363, "num_tokens": 230508284.0, "reward": 2.3267576694488525, "reward_std": 0.4594190716743469, "rewards/code_complexity_reward/mean": 0.980175793170929, "rewards/code_complexity_reward/std": 0.08550854027271271, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1568, "step_time": 33.08570025116205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 116.681640625, "completions/mean_terminated_length": 115.90802001953125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2385903934482485, "epoch": 0.8940170940170941, "frac_reward_zero_std": 0.203125, "grad_norm": 0.10428750514984131, "kl": 0.24130183621309698, "learning_rate": 1.7104607141788826e-07, "loss": 0.001207361463457346, "num_tokens": 230636753.0, "reward": 2.308300733566284, "reward_std": 0.5213348865509033, "rewards/code_complexity_reward/mean": 0.965039074420929, "rewards/code_complexity_reward/std": 0.1582944691181183, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 1569, "step_time": 49.00775744859129 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 111.560546875, "completions/mean_terminated_length": 111.560546875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23950362135656178, "epoch": 0.8945868945868946, "frac_reward_zero_std": 0.390625, "grad_norm": 0.09080207347869873, "kl": 0.255544945364818, "learning_rate": 1.692424640869586e-07, "loss": 0.0012784902937710285, "num_tokens": 230763480.0, "reward": 2.416210889816284, "reward_std": 0.5109858512878418, "rewards/code_complexity_reward/mean": 0.981249988079071, "rewards/code_complexity_reward/std": 0.10270319133996964, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1570, "step_time": 42.192081056535244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 275.0, "completions/max_terminated_length": 275.0, "completions/mean_length": 107.857421875, "completions/mean_terminated_length": 107.857421875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24614000297151506, "epoch": 0.8951566951566952, "frac_reward_zero_std": 0.359375, "grad_norm": 0.11282183229923248, "kl": 0.2409557537175715, "learning_rate": 1.6744808315943218e-07, "loss": 0.0012054890394210815, "num_tokens": 230887159.0, "reward": 2.3958983421325684, "reward_std": 0.5139447450637817, "rewards/code_complexity_reward/mean": 0.9813476800918579, "rewards/code_complexity_reward/std": 0.11724471300840378, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1571, "step_time": 40.18865989986807 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 112.98828125, "completions/mean_terminated_length": 112.98828125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24112621997483075, "epoch": 0.8957264957264958, "frac_reward_zero_std": 0.359375, "grad_norm": 0.0981912612915039, "kl": 0.23862676229327917, "learning_rate": 1.6566293573844123e-07, "loss": 0.0011935275979340076, "num_tokens": 231012089.0, "reward": 2.432421922683716, "reward_std": 0.5165780186653137, "rewards/code_complexity_reward/mean": 0.9798828363418579, "rewards/code_complexity_reward/std": 0.1026345044374466, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1572, "step_time": 50.170193302445114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 238.0, "completions/max_terminated_length": 238.0, "completions/mean_length": 108.51171875, "completions/mean_terminated_length": 108.51171875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2361773932352662, "epoch": 0.8962962962962963, "frac_reward_zero_std": 0.328125, "grad_norm": 0.09585686028003693, "kl": 0.24692250951193273, "learning_rate": 1.6388702889056974e-07, "loss": 0.0012348826276138425, "num_tokens": 231135383.0, "reward": 2.362988233566284, "reward_std": 0.48475009202957153, "rewards/code_complexity_reward/mean": 0.982714831829071, "rewards/code_complexity_reward/std": 0.09975044429302216, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1573, "step_time": 31.166042409837246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 302.0, "completions/max_terminated_length": 302.0, "completions/mean_length": 115.921875, "completions/mean_terminated_length": 115.921875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24174082232639194, "epoch": 0.8968660968660969, "frac_reward_zero_std": 0.359375, "grad_norm": 0.08730493485927582, "kl": 0.246593717020005, "learning_rate": 1.6212036964581983e-07, "loss": 0.0012329884339123964, "num_tokens": 231261751.0, "reward": 2.3255860805511475, "reward_std": 0.5087296962738037, "rewards/code_complexity_reward/mean": 0.9677734375, "rewards/code_complexity_reward/std": 0.1404024213552475, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1574, "step_time": 35.321561771444976 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 418.0, "completions/max_terminated_length": 418.0, "completions/mean_length": 119.123046875, "completions/mean_terminated_length": 119.123046875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23797227628529072, "epoch": 0.8974358974358975, "frac_reward_zero_std": 0.296875, "grad_norm": 0.10918323695659637, "kl": 0.2245015089865774, "learning_rate": 1.603629649975877e-07, "loss": 0.0011230830568820238, "num_tokens": 231390422.0, "reward": 2.313281297683716, "reward_std": 0.5003710389137268, "rewards/code_complexity_reward/mean": 0.9671875238418579, "rewards/code_complexity_reward/std": 0.13974221050739288, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1575, "step_time": 43.317649744451046 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 294.0, "completions/max_terminated_length": 294.0, "completions/mean_length": 117.13671875, "completions/mean_terminated_length": 117.13671875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2372456407174468, "epoch": 0.898005698005698, "frac_reward_zero_std": 0.3125, "grad_norm": 0.11124950647354126, "kl": 0.22725655999965966, "learning_rate": 1.5861482190263566e-07, "loss": 0.0011369442800059915, "num_tokens": 231518860.0, "reward": 2.3719725608825684, "reward_std": 0.5192484855651855, "rewards/code_complexity_reward/mean": 0.96923828125, "rewards/code_complexity_reward/std": 0.12792949378490448, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1576, "step_time": 36.43202144373208 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 120.775390625, "completions/mean_terminated_length": 120.775390625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2529130601324141, "epoch": 0.8985754985754986, "frac_reward_zero_std": 0.25, "grad_norm": 0.1330396682024002, "kl": 0.2231943851802498, "learning_rate": 1.5687594728106186e-07, "loss": 0.0011165464529767632, "num_tokens": 231650257.0, "reward": 2.2603514194488525, "reward_std": 0.46820664405822754, "rewards/code_complexity_reward/mean": 0.9660156965255737, "rewards/code_complexity_reward/std": 0.1463427096605301, "rewards/code_execution_reward/mean": 0.205078125, "rewards/code_execution_reward/std": 0.4041535556316376, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1577, "step_time": 46.902925945818424 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 267.0, "completions/max_terminated_length": 267.0, "completions/mean_length": 117.62890625, "completions/mean_terminated_length": 117.62890625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24247565818950534, "epoch": 0.8991452991452992, "frac_reward_zero_std": 0.328125, "grad_norm": 0.09185078740119934, "kl": 0.22811023169197142, "learning_rate": 1.5514634801627708e-07, "loss": 0.0011407590936869383, "num_tokens": 231780107.0, "reward": 2.3770995140075684, "reward_std": 0.4757569432258606, "rewards/code_complexity_reward/mean": 0.985546886920929, "rewards/code_complexity_reward/std": 0.0793570950627327, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1578, "step_time": 51.76154459454119 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 302.0, "completions/max_terminated_length": 302.0, "completions/mean_length": 110.775390625, "completions/mean_terminated_length": 110.775390625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24681939487345517, "epoch": 0.8997150997150997, "frac_reward_zero_std": 0.328125, "grad_norm": 0.1464090645313263, "kl": 0.24999096943065524, "learning_rate": 1.534260309549726e-07, "loss": 0.0012505180202424526, "num_tokens": 231902984.0, "reward": 2.4861326217651367, "reward_std": 0.5122406482696533, "rewards/code_complexity_reward/mean": 0.9847656488418579, "rewards/code_complexity_reward/std": 0.0797644853591919, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1579, "step_time": 35.817458231933415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 350.0, "completions/max_terminated_length": 350.0, "completions/mean_length": 115.52734375, "completions/mean_terminated_length": 115.52734375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24721977254375815, "epoch": 0.9002849002849003, "frac_reward_zero_std": 0.265625, "grad_norm": 0.1411111056804657, "kl": 0.2278847135603428, "learning_rate": 1.5171500290709824e-07, "loss": 0.0011400396469980478, "num_tokens": 232033782.0, "reward": 2.3067383766174316, "reward_std": 0.5114309787750244, "rewards/code_complexity_reward/mean": 0.9652343988418579, "rewards/code_complexity_reward/std": 0.15328224003314972, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1580, "step_time": 39.75041578616947 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 326.0, "completions/max_terminated_length": 326.0, "completions/mean_length": 115.255859375, "completions/mean_terminated_length": 115.255859375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.25632293545641005, "epoch": 0.9008547008547009, "frac_reward_zero_std": 0.359375, "grad_norm": 0.11556249856948853, "kl": 0.2610099809244275, "learning_rate": 1.50013270645831e-07, "loss": 0.0013050608104094863, "num_tokens": 232159929.0, "reward": 2.3391599655151367, "reward_std": 0.501255452632904, "rewards/code_complexity_reward/mean": 0.9715820550918579, "rewards/code_complexity_reward/std": 0.12639839947223663, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1581, "step_time": 37.34737015981227 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 118.8671875, "completions/mean_terminated_length": 118.8671875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.25691122631542385, "epoch": 0.9014245014245015, "frac_reward_zero_std": 0.296875, "grad_norm": 0.17384101450443268, "kl": 0.24550824286416173, "learning_rate": 1.483208409075515e-07, "loss": 0.0012280609225854278, "num_tokens": 232291509.0, "reward": 2.3353028297424316, "reward_std": 0.48202019929885864, "rewards/code_complexity_reward/mean": 0.9730468988418579, "rewards/code_complexity_reward/std": 0.10377462208271027, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1582, "step_time": 56.72869786247611 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 110.951171875, "completions/mean_terminated_length": 110.951171875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24129432998597622, "epoch": 0.901994301994302, "frac_reward_zero_std": 0.4375, "grad_norm": 0.08385378122329712, "kl": 0.2438399156089872, "learning_rate": 1.4663772039181455e-07, "loss": 0.0012196637690067291, "num_tokens": 232418772.0, "reward": 2.4754881858825684, "reward_std": 0.5375273823738098, "rewards/code_complexity_reward/mean": 0.9808593988418579, "rewards/code_complexity_reward/std": 0.11768662184476852, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1583, "step_time": 42.884357549250126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 320.0, "completions/max_terminated_length": 320.0, "completions/mean_length": 115.337890625, "completions/mean_terminated_length": 115.337890625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24324478884227574, "epoch": 0.9025641025641026, "frac_reward_zero_std": 0.28125, "grad_norm": 0.12038461863994598, "kl": 0.24661506176926196, "learning_rate": 1.449639157613253e-07, "loss": 0.0012336699292063713, "num_tokens": 232547897.0, "reward": 2.3700194358825684, "reward_std": 0.47725629806518555, "rewards/code_complexity_reward/mean": 0.98193359375, "rewards/code_complexity_reward/std": 0.08163430541753769, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1584, "step_time": 36.99385975394398 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 113.833984375, "completions/mean_terminated_length": 113.05479431152344, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2498420716729015, "epoch": 0.9031339031339032, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09634453803300858, "kl": 0.22899757255800068, "learning_rate": 1.4329943364191023e-07, "loss": 0.0011453612241894007, "num_tokens": 232675820.0, "reward": 2.3301756381988525, "reward_std": 0.5106403231620789, "rewards/code_complexity_reward/mean": 0.9710937142372131, "rewards/code_complexity_reward/std": 0.13920508325099945, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1585, "step_time": 50.16746863909066 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 407.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 115.47265625, "completions/mean_terminated_length": 115.47265625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2584596602246165, "epoch": 0.9037037037037037, "frac_reward_zero_std": 0.203125, "grad_norm": 0.13611502945423126, "kl": 0.2426253838930279, "learning_rate": 1.4164428062249407e-07, "loss": 0.0012133866548538208, "num_tokens": 232804462.0, "reward": 2.36083984375, "reward_std": 0.5033568143844604, "rewards/code_complexity_reward/mean": 0.9736328125, "rewards/code_complexity_reward/std": 0.11869350075721741, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1586, "step_time": 43.408743060193956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 254.0, "completions/max_terminated_length": 254.0, "completions/mean_length": 113.25390625, "completions/mean_terminated_length": 113.25390625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24476235103793442, "epoch": 0.9042735042735043, "frac_reward_zero_std": 0.390625, "grad_norm": 0.08706629276275635, "kl": 0.23316343780606985, "learning_rate": 1.3999846325506994e-07, "loss": 0.0011661698808893561, "num_tokens": 232929280.0, "reward": 2.3328123092651367, "reward_std": 0.4527449309825897, "rewards/code_complexity_reward/mean": 0.9828124642372131, "rewards/code_complexity_reward/std": 0.07192829251289368, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1587, "step_time": 32.936519738286734 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 393.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 117.42578125, "completions/mean_terminated_length": 117.42578125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24802067829295993, "epoch": 0.9048433048433049, "frac_reward_zero_std": 0.328125, "grad_norm": 0.1036006435751915, "kl": 0.22790461243130267, "learning_rate": 1.383619880546766e-07, "loss": 0.0011398009955883026, "num_tokens": 233059050.0, "reward": 2.2396974563598633, "reward_std": 0.44193971157073975, "rewards/code_complexity_reward/mean": 0.97119140625, "rewards/code_complexity_reward/std": 0.133062481880188, "rewards/code_execution_reward/mean": 0.177734375, "rewards/code_execution_reward/std": 0.3826628625392914, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1588, "step_time": 41.17616145685315 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 267.0, "completions/max_terminated_length": 267.0, "completions/mean_length": 117.939453125, "completions/mean_terminated_length": 117.939453125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2354457383044064, "epoch": 0.9054131054131054, "frac_reward_zero_std": 0.390625, "grad_norm": 0.08348619937896729, "kl": 0.230756517034024, "learning_rate": 1.3673486149937133e-07, "loss": 0.0011542246211320162, "num_tokens": 233191187.0, "reward": 2.3916990756988525, "reward_std": 0.5149567127227783, "rewards/code_complexity_reward/mean": 0.9762694835662842, "rewards/code_complexity_reward/std": 0.11898276209831238, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1589, "step_time": 51.308826461434364 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 116.564453125, "completions/mean_terminated_length": 115.79060363769531, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24417507834732533, "epoch": 0.905982905982906, "frac_reward_zero_std": 0.3125, "grad_norm": 0.09343193471431732, "kl": 0.243659877916798, "learning_rate": 1.3511709003020346e-07, "loss": 0.001218733610585332, "num_tokens": 233319692.0, "reward": 2.3709471225738525, "reward_std": 0.5587166547775269, "rewards/code_complexity_reward/mean": 0.959667980670929, "rewards/code_complexity_reward/std": 0.16920237243175507, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1590, "step_time": 48.288739495910704 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 314.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 111.556640625, "completions/mean_terminated_length": 111.556640625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24382577487267554, "epoch": 0.9065527065527066, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09519525617361069, "kl": 0.2543085440993309, "learning_rate": 1.3350868005119172e-07, "loss": 0.0012719258666038513, "num_tokens": 233445729.0, "reward": 2.3987302780151367, "reward_std": 0.5691114068031311, "rewards/code_complexity_reward/mean": 0.9637694954872131, "rewards/code_complexity_reward/std": 0.16888852417469025, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1591, "step_time": 44.94133491907269 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 273.0, "completions/max_terminated_length": 273.0, "completions/mean_length": 110.0546875, "completions/mean_terminated_length": 110.0546875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24795782403089106, "epoch": 0.9071225071225071, "frac_reward_zero_std": 0.375, "grad_norm": 0.08679347485303879, "kl": 0.24270188040100038, "learning_rate": 1.3190963792929472e-07, "loss": 0.0012142565101385117, "num_tokens": 233568717.0, "reward": 2.3499512672424316, "reward_std": 0.5639844536781311, "rewards/code_complexity_reward/mean": 0.9603515863418579, "rewards/code_complexity_reward/std": 0.17925775051116943, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1592, "step_time": 35.370028814300895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 115.537109375, "completions/mean_terminated_length": 115.537109375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24878399423323572, "epoch": 0.9076923076923077, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09707827121019363, "kl": 0.23952867300249636, "learning_rate": 1.3031996999438967e-07, "loss": 0.0011980114504694939, "num_tokens": 233694408.0, "reward": 2.3128905296325684, "reward_std": 0.49055641889572144, "rewards/code_complexity_reward/mean": 0.9716796875, "rewards/code_complexity_reward/std": 0.13281798362731934, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1593, "step_time": 36.13285419996828 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 485.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 113.400390625, "completions/mean_terminated_length": 113.400390625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24001526506617665, "epoch": 0.9082621082621083, "frac_reward_zero_std": 0.34375, "grad_norm": 0.10370863974094391, "kl": 0.2351057690102607, "learning_rate": 1.2873968253924506e-07, "loss": 0.0011761384084820747, "num_tokens": 233821021.0, "reward": 2.4595704078674316, "reward_std": 0.5365836024284363, "rewards/code_complexity_reward/mean": 0.971875011920929, "rewards/code_complexity_reward/std": 0.12090655416250229, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1594, "step_time": 47.52486567012966 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 254.0, "completions/max_terminated_length": 254.0, "completions/mean_length": 114.212890625, "completions/mean_terminated_length": 114.212890625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.23290016152895987, "epoch": 0.9088319088319088, "frac_reward_zero_std": 0.28125, "grad_norm": 0.10971704870462418, "kl": 0.25454056123271585, "learning_rate": 1.2716878181949637e-07, "loss": 0.0012730127200484276, "num_tokens": 233946186.0, "reward": 2.344482421875, "reward_std": 0.4918564260005951, "rewards/code_complexity_reward/mean": 0.9800781011581421, "rewards/code_complexity_reward/std": 0.1173066720366478, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1595, "step_time": 41.430918267928064 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 291.0, "completions/max_terminated_length": 291.0, "completions/mean_length": 105.9921875, "completions/mean_terminated_length": 105.9921875, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.2480096649378538, "epoch": 0.9094017094017094, "frac_reward_zero_std": 0.390625, "grad_norm": 0.07245469838380814, "kl": 0.2472965985070914, "learning_rate": 1.25607274053621e-07, "loss": 0.0012364808935672045, "num_tokens": 234067470.0, "reward": 2.4290037155151367, "reward_std": 0.5173445343971252, "rewards/code_complexity_reward/mean": 0.9833008050918579, "rewards/code_complexity_reward/std": 0.10876213759183884, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1596, "step_time": 35.20318827126175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 114.033203125, "completions/mean_terminated_length": 114.033203125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2391778260935098, "epoch": 0.90997150997151, "frac_reward_zero_std": 0.375, "grad_norm": 0.10659979283809662, "kl": 0.24033389729447663, "learning_rate": 1.2405516542291413e-07, "loss": 0.0012022230075672269, "num_tokens": 234195015.0, "reward": 2.425488233566284, "reward_std": 0.5156263709068298, "rewards/code_complexity_reward/mean": 0.978808581829071, "rewards/code_complexity_reward/std": 0.10395856946706772, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1597, "step_time": 42.69619536679238 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 118.841796875, "completions/mean_terminated_length": 118.07240295410156, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2370168431662023, "epoch": 0.9105413105413105, "frac_reward_zero_std": 0.28125, "grad_norm": 0.09297078102827072, "kl": 0.22410888667218387, "learning_rate": 1.2251246207146516e-07, "loss": 0.0011212709359824657, "num_tokens": 234324526.0, "reward": 2.3911619186401367, "reward_std": 0.5369618535041809, "rewards/code_complexity_reward/mean": 0.9691406488418579, "rewards/code_complexity_reward/std": 0.14102241396903992, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1598, "step_time": 48.59318763110787 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 328.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 114.21875, "completions/mean_terminated_length": 114.21875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.255998627981171, "epoch": 0.9111111111111111, "frac_reward_zero_std": 0.34375, "grad_norm": 0.1006079837679863, "kl": 0.27930801431648433, "learning_rate": 1.2097917010613052e-07, "loss": 0.0013978071510791779, "num_tokens": 234451430.0, "reward": 2.370800733566284, "reward_std": 0.4811840355396271, "rewards/code_complexity_reward/mean": 0.981738269329071, "rewards/code_complexity_reward/std": 0.09164460003376007, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1599, "step_time": 36.69420419353992 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 248.0, "completions/max_terminated_length": 248.0, "completions/mean_length": 109.453125, "completions/mean_terminated_length": 109.453125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.25037648109719157, "epoch": 0.9116809116809117, "frac_reward_zero_std": 0.34375, "grad_norm": 0.09928230941295624, "kl": 0.24994084727950394, "learning_rate": 1.1945529559651225e-07, "loss": 0.0012502026511356235, "num_tokens": 234576006.0, "reward": 2.311279296875, "reward_std": 0.5072382092475891, "rewards/code_complexity_reward/mean": 0.969531238079071, "rewards/code_complexity_reward/std": 0.1519906222820282, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1600, "step_time": 40.96390645299107 }, { "epoch": 0.9116809116809117, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 156.85, "eval_completions/max_terminated_length": 156.85, "eval_completions/mean_length": 114.66625, "eval_completions/mean_terminated_length": 114.66625, "eval_completions/min_length": 87.73, "eval_completions/min_terminated_length": 87.73, "eval_entropy": 0.23565456144511698, "eval_frac_reward_zero_std": 0.33, "eval_kl": 0.2395239368453622, "eval_loss": 0.0029065755661576986, "eval_num_tokens": 234576006.0, "eval_reward": 2.345624942779541, "eval_reward_std": 0.23355471447110177, "eval_rewards/code_complexity_reward/mean": 0.9759375005960464, "eval_rewards/code_complexity_reward/std": 0.047054351456463334, "eval_rewards/code_execution_reward/mean": 0.2775, "eval_rewards/code_execution_reward/std": 0.17860785096883774, "eval_rewards/code_syntax_reward/mean": 0.4925, "eval_rewards/code_syntax_reward/std": 0.018771235942840577, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.4996875, "eval_rewards/xmlcount_reward_func/std": 0.000883883461356163, "eval_runtime": 732.6724, "eval_samples_per_second": 0.136, "eval_steps_per_second": 0.018, "step": 1600 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 426.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 115.775390625, "completions/mean_terminated_length": 115.775390625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24624437931925058, "epoch": 0.9122507122507123, "frac_reward_zero_std": 0.34375, "grad_norm": 0.08689479529857635, "kl": 0.23934047715738416, "learning_rate": 1.1794084457493222e-07, "loss": 0.0011969064362347126, "num_tokens": 234705363.0, "reward": 2.4093260765075684, "reward_std": 0.5029241442680359, "rewards/code_complexity_reward/mean": 0.976757824420929, "rewards/code_complexity_reward/std": 0.10268811881542206, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1601, "step_time": 51.842479187995195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 110.08203125, "completions/mean_terminated_length": 109.29550170898438, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24434620095416903, "epoch": 0.9128205128205128, "frac_reward_zero_std": 0.34375, "grad_norm": 0.10507624596357346, "kl": 0.2824416181538254, "learning_rate": 1.1643582303641015e-07, "loss": 0.0014119320549070835, "num_tokens": 234830389.0, "reward": 2.328418016433716, "reward_std": 0.5455885529518127, "rewards/code_complexity_reward/mean": 0.9617187976837158, "rewards/code_complexity_reward/std": 0.16914477944374084, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 1602, "step_time": 68.0745660988614 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 335.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 112.259765625, "completions/mean_terminated_length": 112.259765625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.24880889384076, "epoch": 0.9133903133903134, "frac_reward_zero_std": 0.375, "grad_norm": 0.11769481748342514, "kl": 0.25190600100904703, "learning_rate": 1.1494023693863792e-07, "loss": 0.0012598595349118114, "num_tokens": 234957322.0, "reward": 2.3114256858825684, "reward_std": 0.4264853298664093, "rewards/code_complexity_reward/mean": 0.98974609375, "rewards/code_complexity_reward/std": 0.04903505742549896, "rewards/code_execution_reward/mean": 0.22265625, "rewards/code_execution_reward/std": 0.41643625497817993, "rewards/code_syntax_reward/mean": 0.4990234375, "rewards/code_syntax_reward/std": 0.022097086533904076, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1603, "step_time": 43.40677141677588 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 265.0, "completions/max_terminated_length": 265.0, "completions/mean_length": 112.865234375, "completions/mean_terminated_length": 112.865234375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2414850725326687, "epoch": 0.913960113960114, "frac_reward_zero_std": 0.46875, "grad_norm": 0.07967466861009598, "kl": 0.2534245338756591, "learning_rate": 1.1345409220195752e-07, "loss": 0.001267389627173543, "num_tokens": 235082493.0, "reward": 2.3416991233825684, "reward_std": 0.46863922476768494, "rewards/code_complexity_reward/mean": 0.98388671875, "rewards/code_complexity_reward/std": 0.09070894122123718, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1604, "step_time": 40.77702565398067 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 273.0, "completions/max_terminated_length": 273.0, "completions/mean_length": 109.15625, "completions/mean_terminated_length": 109.15625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23937580874189734, "epoch": 0.9145299145299145, "frac_reward_zero_std": 0.328125, "grad_norm": 0.12088024616241455, "kl": 0.2503120480105281, "learning_rate": 1.1197739470933583e-07, "loss": 0.0012520232703536749, "num_tokens": 235204117.0, "reward": 2.4158689975738525, "reward_std": 0.513990581035614, "rewards/code_complexity_reward/mean": 0.9803711175918579, "rewards/code_complexity_reward/std": 0.1104627251625061, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1605, "step_time": 39.696815703995526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 111.798828125, "completions/mean_terminated_length": 111.798828125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22617860604077578, "epoch": 0.9150997150997151, "frac_reward_zero_std": 0.296875, "grad_norm": 0.10982669144868851, "kl": 0.2346725792158395, "learning_rate": 1.1051015030634327e-07, "loss": 0.0011740389745682478, "num_tokens": 235328566.0, "reward": 2.3879880905151367, "reward_std": 0.5185853838920593, "rewards/code_complexity_reward/mean": 0.9725585579872131, "rewards/code_complexity_reward/std": 0.12076038122177124, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1606, "step_time": 46.635095407254994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 278.0, "completions/max_terminated_length": 278.0, "completions/mean_length": 113.1796875, "completions/mean_terminated_length": 113.1796875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24468037020415068, "epoch": 0.9156695156695157, "frac_reward_zero_std": 0.375, "grad_norm": 0.1348838210105896, "kl": 0.2303679594770074, "learning_rate": 1.0905236480113018e-07, "loss": 0.001152078970335424, "num_tokens": 235455362.0, "reward": 2.3821287155151367, "reward_std": 0.4792851507663727, "rewards/code_complexity_reward/mean": 0.9842773675918579, "rewards/code_complexity_reward/std": 0.0798841193318367, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1607, "step_time": 41.773376835510135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 279.0, "completions/max_terminated_length": 279.0, "completions/mean_length": 112.521484375, "completions/mean_terminated_length": 112.521484375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2524876475799829, "epoch": 0.9162393162393162, "frac_reward_zero_std": 0.3125, "grad_norm": 0.10105152428150177, "kl": 0.23639614414423704, "learning_rate": 1.0760404396440188e-07, "loss": 0.0011827194830402732, "num_tokens": 235581461.0, "reward": 2.385546922683716, "reward_std": 0.5055427551269531, "rewards/code_complexity_reward/mean": 0.9769531488418579, "rewards/code_complexity_reward/std": 0.1107972040772438, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1608, "step_time": 32.90836046077311 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 316.0, "completions/max_terminated_length": 316.0, "completions/mean_length": 111.6796875, "completions/mean_terminated_length": 111.6796875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23919075494632125, "epoch": 0.9168091168091168, "frac_reward_zero_std": 0.4375, "grad_norm": 0.09250428527593613, "kl": 0.23118392121978104, "learning_rate": 1.0616519352939891e-07, "loss": 0.0011565880849957466, "num_tokens": 235707825.0, "reward": 2.3768553733825684, "reward_std": 0.4860846996307373, "rewards/code_complexity_reward/mean": 0.98193359375, "rewards/code_complexity_reward/std": 0.09295523911714554, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1609, "step_time": 37.27500251773745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 381.0, "completions/max_terminated_length": 381.0, "completions/mean_length": 112.1875, "completions/mean_terminated_length": 112.1875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24151571048423648, "epoch": 0.9173789173789174, "frac_reward_zero_std": 0.359375, "grad_norm": 0.0956762507557869, "kl": 0.24171503516845405, "learning_rate": 1.0473581919187208e-07, "loss": 0.0012088268995285034, "num_tokens": 235832897.0, "reward": 2.394775390625, "reward_std": 0.5316920876502991, "rewards/code_complexity_reward/mean": 0.9724608659744263, "rewards/code_complexity_reward/std": 0.13437321782112122, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1610, "step_time": 39.54988188110292 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 255.0, "completions/max_terminated_length": 255.0, "completions/mean_length": 114.0625, "completions/mean_terminated_length": 114.0625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24840725772082806, "epoch": 0.9179487179487179, "frac_reward_zero_std": 0.34375, "grad_norm": 0.08804909884929657, "kl": 0.23036280111409724, "learning_rate": 1.0331592661006113e-07, "loss": 0.0011521752458065748, "num_tokens": 235961745.0, "reward": 2.3416991233825684, "reward_std": 0.4978151023387909, "rewards/code_complexity_reward/mean": 0.9772460460662842, "rewards/code_complexity_reward/std": 0.12486717849969864, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.017459021881222725, "step": 1611, "step_time": 50.219332590699196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 115.765625, "completions/mean_terminated_length": 115.765625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23482640483416617, "epoch": 0.9185185185185185, "frac_reward_zero_std": 0.328125, "grad_norm": 0.09974487125873566, "kl": 0.2292793917004019, "learning_rate": 1.0190552140467131e-07, "loss": 0.001146901398897171, "num_tokens": 236089377.0, "reward": 2.4059081077575684, "reward_std": 0.5314663052558899, "rewards/code_complexity_reward/mean": 0.97314453125, "rewards/code_complexity_reward/std": 0.13317762315273285, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1612, "step_time": 46.272325151599944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 113.57421875, "completions/mean_terminated_length": 113.57421875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24422921752557158, "epoch": 0.9190883190883191, "frac_reward_zero_std": 0.375, "grad_norm": 0.0865299180150032, "kl": 0.280281123239547, "learning_rate": 1.0050460915885213e-07, "loss": 0.0014021715614944696, "num_tokens": 236216063.0, "reward": 2.4173827171325684, "reward_std": 0.5147725343704224, "rewards/code_complexity_reward/mean": 0.9814453125, "rewards/code_complexity_reward/std": 0.10933646559715271, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1613, "step_time": 39.08495301939547 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 328.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 112.751953125, "completions/mean_terminated_length": 112.751953125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24817859008908272, "epoch": 0.9196581196581196, "frac_reward_zero_std": 0.265625, "grad_norm": 0.09694387763738632, "kl": 0.23616237589158118, "learning_rate": 9.911319541817399e-08, "loss": 0.001181001658551395, "num_tokens": 236341832.0, "reward": 2.2884767055511475, "reward_std": 0.4906802773475647, "rewards/code_complexity_reward/mean": 0.96875, "rewards/code_complexity_reward/std": 0.1458492875099182, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1614, "step_time": 36.91206087265164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 113.890625, "completions/mean_terminated_length": 113.890625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24225862883031368, "epoch": 0.9202279202279202, "frac_reward_zero_std": 0.359375, "grad_norm": 0.11077986657619476, "kl": 0.22373281745240092, "learning_rate": 9.773128569060874e-08, "loss": 0.0011194461258128285, "num_tokens": 236470120.0, "reward": 2.335155963897705, "reward_std": 0.5114389657974243, "rewards/code_complexity_reward/mean": 0.971484363079071, "rewards/code_complexity_reward/std": 0.13889887928962708, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1615, "step_time": 42.80461244098842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 287.0, "completions/max_terminated_length": 287.0, "completions/mean_length": 111.82421875, "completions/mean_terminated_length": 111.82421875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.24163542035967112, "epoch": 0.9207977207977208, "frac_reward_zero_std": 0.296875, "grad_norm": 0.10139567404985428, "kl": 0.23555571143515408, "learning_rate": 9.635888544650418e-08, "loss": 0.0011783638037741184, "num_tokens": 236595062.0, "reward": 2.3971681594848633, "reward_std": 0.5083324909210205, "rewards/code_complexity_reward/mean": 0.9768555164337158, "rewards/code_complexity_reward/std": 0.11000122129917145, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1616, "step_time": 34.01753431465477 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 285.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 114.2578125, "completions/mean_terminated_length": 114.2578125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2465424642432481, "epoch": 0.9213675213675213, "frac_reward_zero_std": 0.3125, "grad_norm": 0.10278619080781937, "kl": 0.23767873016186059, "learning_rate": 9.499600011856569e-08, "loss": 0.001188603462651372, "num_tokens": 236724002.0, "reward": 2.4668943881988525, "reward_std": 0.5384462475776672, "rewards/code_complexity_reward/mean": 0.973339855670929, "rewards/code_complexity_reward/std": 0.12081417441368103, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1617, "step_time": 42.24698409810662 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 305.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 115.388671875, "completions/mean_terminated_length": 115.388671875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2403623266145587, "epoch": 0.9219373219373219, "frac_reward_zero_std": 0.296875, "grad_norm": 0.12314746528863907, "kl": 0.27910409471951425, "learning_rate": 9.36426351018338e-08, "loss": 0.00139580387622118, "num_tokens": 236849929.0, "reward": 2.3956053256988525, "reward_std": 0.5184230208396912, "rewards/code_complexity_reward/mean": 0.976269543170929, "rewards/code_complexity_reward/std": 0.1190238744020462, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1618, "step_time": 63.09332710132003 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 113.728515625, "completions/mean_terminated_length": 113.728515625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.25241202116012573, "epoch": 0.9225071225071225, "frac_reward_zero_std": 0.328125, "grad_norm": 0.11024406552314758, "kl": 0.25729292957112193, "learning_rate": 9.229879575366085e-08, "loss": 0.0012871737126260996, "num_tokens": 236977534.0, "reward": 2.37646484375, "reward_std": 0.4760175943374634, "rewards/code_complexity_reward/mean": 0.9844726324081421, "rewards/code_complexity_reward/std": 0.0809563398361206, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1619, "step_time": 49.71232144534588 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 114.83984375, "completions/mean_terminated_length": 114.83984375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24612490623258054, "epoch": 0.9230769230769231, "frac_reward_zero_std": 0.390625, "grad_norm": 0.09094381332397461, "kl": 0.24833354074507952, "learning_rate": 9.096448739369324e-08, "loss": 0.0012419144622981548, "num_tokens": 237104420.0, "reward": 2.290722608566284, "reward_std": 0.4345599114894867, "rewards/code_complexity_reward/mean": 0.980761706829071, "rewards/code_complexity_reward/std": 0.08489665389060974, "rewards/code_execution_reward/mean": 0.212890625, "rewards/code_execution_reward/std": 0.409751296043396, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1620, "step_time": 37.92209427896887 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 283.0, "completions/mean_length": 116.25390625, "completions/mean_terminated_length": 114.70196533203125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.23737466568127275, "epoch": 0.9236467236467236, "frac_reward_zero_std": 0.359375, "grad_norm": 0.08233334124088287, "kl": 0.24152550753206015, "learning_rate": 8.963971530384669e-08, "loss": 0.0012081133900210261, "num_tokens": 237232422.0, "reward": 2.279003858566284, "reward_std": 0.47548621892929077, "rewards/code_complexity_reward/mean": 0.974414050579071, "rewards/code_complexity_reward/std": 0.1393290013074875, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1621, "step_time": 47.985418980941176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 116.767578125, "completions/mean_terminated_length": 116.767578125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.24973936891183257, "epoch": 0.9242165242165242, "frac_reward_zero_std": 0.375, "grad_norm": 0.09094134718179703, "kl": 0.2434071754105389, "learning_rate": 8.832448472828936e-08, "loss": 0.0012174691073596478, "num_tokens": 237360759.0, "reward": 2.326367139816284, "reward_std": 0.4892664849758148, "rewards/code_complexity_reward/mean": 0.9744141101837158, "rewards/code_complexity_reward/std": 0.12686823308467865, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1622, "step_time": 39.17262569908053 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 112.17578125, "completions/mean_terminated_length": 112.17578125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.26218557078391314, "epoch": 0.9247863247863248, "frac_reward_zero_std": 0.46875, "grad_norm": 0.07608350366353989, "kl": 0.24414005922153592, "learning_rate": 8.701880087341658e-08, "loss": 0.0012210796121507883, "num_tokens": 237491281.0, "reward": 2.340576171875, "reward_std": 0.4885531961917877, "rewards/code_complexity_reward/mean": 0.9802733659744263, "rewards/code_complexity_reward/std": 0.117047518491745, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1623, "step_time": 50.681752876378596 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 285.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 111.51171875, "completions/mean_terminated_length": 111.51171875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24969371245242655, "epoch": 0.9253561253561253, "frac_reward_zero_std": 0.34375, "grad_norm": 0.10669136792421341, "kl": 0.23629186442121863, "learning_rate": 8.57226689078347e-08, "loss": 0.0011820519575849175, "num_tokens": 237617791.0, "reward": 2.3408203125, "reward_std": 0.4995166063308716, "rewards/code_complexity_reward/mean": 0.97607421875, "rewards/code_complexity_reward/std": 0.12535202503204346, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1624, "step_time": 34.84417388681322 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 302.0, "completions/max_terminated_length": 302.0, "completions/mean_length": 111.78515625, "completions/mean_terminated_length": 111.78515625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23988860473036766, "epoch": 0.9259259259259259, "frac_reward_zero_std": 0.28125, "grad_norm": 0.08498362451791763, "kl": 0.24798645731061697, "learning_rate": 8.443609396233731e-08, "loss": 0.0012399036204442382, "num_tokens": 237742129.0, "reward": 2.4322266578674316, "reward_std": 0.5491378307342529, "rewards/code_complexity_reward/mean": 0.972851574420929, "rewards/code_complexity_reward/std": 0.14008408784866333, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1625, "step_time": 36.089989754371345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 290.0, "completions/max_terminated_length": 290.0, "completions/mean_length": 117.99609375, "completions/mean_terminated_length": 117.99609375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24387042014859617, "epoch": 0.9264957264957265, "frac_reward_zero_std": 0.3125, "grad_norm": 0.10868071019649506, "kl": 0.23848014208488166, "learning_rate": 8.315908112988574e-08, "loss": 0.0011931464541703463, "num_tokens": 237877335.0, "reward": 2.3397459983825684, "reward_std": 0.5337414741516113, "rewards/code_complexity_reward/mean": 0.96337890625, "rewards/code_complexity_reward/std": 0.1580572873353958, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1626, "step_time": 36.72012913785875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 110.34765625, "completions/mean_terminated_length": 110.34765625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2392652288544923, "epoch": 0.927065527065527, "frac_reward_zero_std": 0.34375, "grad_norm": 0.09372647106647491, "kl": 0.24862071126699448, "learning_rate": 8.189163546559076e-08, "loss": 0.00124332495033741, "num_tokens": 238000985.0, "reward": 2.3028321266174316, "reward_std": 0.5122346878051758, "rewards/code_complexity_reward/mean": 0.967480480670929, "rewards/code_complexity_reward/std": 0.15796560049057007, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1627, "step_time": 33.60138705559075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 453.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 114.6171875, "completions/mean_terminated_length": 114.6171875, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.22558934800326824, "epoch": 0.9276353276353276, "frac_reward_zero_std": 0.390625, "grad_norm": 0.08482906967401505, "kl": 0.2399180990178138, "learning_rate": 8.063376198668954e-08, "loss": 0.00119990692473948, "num_tokens": 238127709.0, "reward": 2.44140625, "reward_std": 0.5457220077514648, "rewards/code_complexity_reward/mean": 0.9712890386581421, "rewards/code_complexity_reward/std": 0.13335904479026794, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1628, "step_time": 44.598544920794666 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 225.0, "completions/max_terminated_length": 225.0, "completions/mean_length": 110.447265625, "completions/mean_terminated_length": 110.447265625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23845631326548755, "epoch": 0.9282051282051282, "frac_reward_zero_std": 0.328125, "grad_norm": 0.11168055981397629, "kl": 0.24200921086594462, "learning_rate": 7.938546567252848e-08, "loss": 0.0012103902408853173, "num_tokens": 238249858.0, "reward": 2.4112305641174316, "reward_std": 0.4921853542327881, "rewards/code_complexity_reward/mean": 0.9860351085662842, "rewards/code_complexity_reward/std": 0.07953696697950363, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1629, "step_time": 39.38010995090008 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 333.0, "completions/max_terminated_length": 333.0, "completions/mean_length": 119.310546875, "completions/mean_terminated_length": 119.310546875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.25040004355832934, "epoch": 0.9287749287749287, "frac_reward_zero_std": 0.296875, "grad_norm": 0.10341273993253708, "kl": 0.26315431809052825, "learning_rate": 7.814675146454176e-08, "loss": 0.0013160407543182373, "num_tokens": 238377449.0, "reward": 2.3487303256988525, "reward_std": 0.4935031533241272, "rewards/code_complexity_reward/mean": 0.975292980670929, "rewards/code_complexity_reward/std": 0.11204341799020767, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1630, "step_time": 37.24792941566557 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 250.0, "completions/max_terminated_length": 250.0, "completions/mean_length": 109.669921875, "completions/mean_terminated_length": 109.669921875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.24793562293052673, "epoch": 0.9293447293447293, "frac_reward_zero_std": 0.296875, "grad_norm": 0.1067635715007782, "kl": 0.2547203141730279, "learning_rate": 7.691762426623283e-08, "loss": 0.0012737889774143696, "num_tokens": 238502128.0, "reward": 2.3293943405151367, "reward_std": 0.5143633484840393, "rewards/code_complexity_reward/mean": 0.9706054925918579, "rewards/code_complexity_reward/std": 0.145682230591774, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1631, "step_time": 48.184503740631044 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 113.431640625, "completions/mean_terminated_length": 112.65166473388672, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2407876937650144, "epoch": 0.9299145299145299, "frac_reward_zero_std": 0.28125, "grad_norm": 0.12412822246551514, "kl": 0.24497585510835052, "learning_rate": 7.569808894315383e-08, "loss": 0.0012251478619873524, "num_tokens": 238630381.0, "reward": 2.4356932640075684, "reward_std": 0.50532066822052, "rewards/code_complexity_reward/mean": 0.9833984375, "rewards/code_complexity_reward/std": 0.08155637234449387, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1632, "step_time": 94.77483958564699 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 109.4765625, "completions/mean_terminated_length": 109.4765625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.23432232555933297, "epoch": 0.9304843304843304, "frac_reward_zero_std": 0.375, "grad_norm": 0.08955885469913483, "kl": 0.25963957188650966, "learning_rate": 7.448815032288781e-08, "loss": 0.0012986718211323023, "num_tokens": 238753409.0, "reward": 2.3993163108825684, "reward_std": 0.5067204833030701, "rewards/code_complexity_reward/mean": 0.97802734375, "rewards/code_complexity_reward/std": 0.10450056195259094, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1633, "step_time": 59.61460848804563 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 114.001953125, "completions/mean_terminated_length": 114.001953125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.243970645358786, "epoch": 0.931054131054131, "frac_reward_zero_std": 0.4375, "grad_norm": 0.08114432543516159, "kl": 0.24820203590206802, "learning_rate": 7.328781319502847e-08, "loss": 0.0012410006020218134, "num_tokens": 238879138.0, "reward": 2.3374998569488525, "reward_std": 0.5104101896286011, "rewards/code_complexity_reward/mean": 0.9699218273162842, "rewards/code_complexity_reward/std": 0.1401829719543457, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1634, "step_time": 38.368705436587334 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 114.732421875, "completions/mean_terminated_length": 114.732421875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.23904175567440689, "epoch": 0.9316239316239316, "frac_reward_zero_std": 0.375, "grad_norm": 0.09678785502910614, "kl": 0.26505147805437446, "learning_rate": 7.209708231116191e-08, "loss": 0.0013255890225991607, "num_tokens": 239006945.0, "reward": 2.3761229515075684, "reward_std": 0.4771208167076111, "rewards/code_complexity_reward/mean": 0.984375, "rewards/code_complexity_reward/std": 0.06867490708827972, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.0385771282017231, "step": 1635, "step_time": 52.68957661930472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 127.16796875, "completions/mean_terminated_length": 127.16796875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2467608954757452, "epoch": 0.9321937321937321, "frac_reward_zero_std": 0.265625, "grad_norm": 0.09718461334705353, "kl": 0.22724598133936524, "learning_rate": 7.091596238484655e-08, "loss": 0.001136542996391654, "num_tokens": 239142359.0, "reward": 2.2757325172424316, "reward_std": 0.49580252170562744, "rewards/code_complexity_reward/mean": 0.9630860090255737, "rewards/code_complexity_reward/std": 0.15268197655677795, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1636, "step_time": 49.050274822860956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 300.0, "completions/max_terminated_length": 300.0, "completions/mean_length": 108.4375, "completions/mean_terminated_length": 108.4375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2433680349495262, "epoch": 0.9327635327635327, "frac_reward_zero_std": 0.296875, "grad_norm": 0.0856422707438469, "kl": 0.2486098853405565, "learning_rate": 6.974445809159708e-08, "loss": 0.001243347767740488, "num_tokens": 239267367.0, "reward": 2.427197217941284, "reward_std": 0.5640040636062622, "rewards/code_complexity_reward/mean": 0.96533203125, "rewards/code_complexity_reward/std": 0.15855970978736877, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1637, "step_time": 34.77514187991619 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 112.17578125, "completions/mean_terminated_length": 111.39334869384766, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24136825907044113, "epoch": 0.9333333333333333, "frac_reward_zero_std": 0.3125, "grad_norm": 0.10281635820865631, "kl": 0.23646768555045128, "learning_rate": 6.858257406886282e-08, "loss": 0.0011830208823084831, "num_tokens": 239397417.0, "reward": 2.343945264816284, "reward_std": 0.5155696868896484, "rewards/code_complexity_reward/mean": 0.97314453125, "rewards/code_complexity_reward/std": 0.13910700380802155, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1638, "step_time": 65.67758090887219 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 371.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 109.46875, "completions/mean_terminated_length": 109.46875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.24079245352186263, "epoch": 0.933903133903134, "frac_reward_zero_std": 0.34375, "grad_norm": 0.09238994121551514, "kl": 0.24011625698767602, "learning_rate": 6.743031491601132e-08, "loss": 0.0012008151970803738, "num_tokens": 239520521.0, "reward": 2.3956055641174316, "reward_std": 0.4988393783569336, "rewards/code_complexity_reward/mean": 0.9840819835662842, "rewards/code_complexity_reward/std": 0.10007598251104355, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1639, "step_time": 39.627950890921056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 265.0, "completions/max_terminated_length": 265.0, "completions/mean_length": 112.771484375, "completions/mean_terminated_length": 112.771484375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23794647702015936, "epoch": 0.9344729344729344, "frac_reward_zero_std": 0.453125, "grad_norm": 0.07850667834281921, "kl": 0.24498363607563078, "learning_rate": 6.628768519431006e-08, "loss": 0.0012250184081494808, "num_tokens": 239645124.0, "reward": 2.354980230331421, "reward_std": 0.4916321039199829, "rewards/code_complexity_reward/mean": 0.9776366949081421, "rewards/code_complexity_reward/std": 0.11051773279905319, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1640, "step_time": 33.08400842640549 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 114.1015625, "completions/mean_terminated_length": 114.1015625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2385843249503523, "epoch": 0.935042735042735, "frac_reward_zero_std": 0.375, "grad_norm": 0.0761294960975647, "kl": 0.22985327197238803, "learning_rate": 6.515468942690728e-08, "loss": 0.0011494627688080072, "num_tokens": 239774352.0, "reward": 2.366894245147705, "reward_std": 0.5454851388931274, "rewards/code_complexity_reward/mean": 0.9640624523162842, "rewards/code_complexity_reward/std": 0.1595538705587387, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1641, "step_time": 39.841756218113005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 245.0, "completions/mean_length": 109.833984375, "completions/mean_terminated_length": 109.04696655273438, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23427389678545296, "epoch": 0.9356125356125357, "frac_reward_zero_std": 0.4375, "grad_norm": 0.0791163370013237, "kl": 0.2337801295798272, "learning_rate": 6.403133209881562e-08, "loss": 0.0011692277621477842, "num_tokens": 239900795.0, "reward": 2.3787596225738525, "reward_std": 0.5071738958358765, "rewards/code_complexity_reward/mean": 0.9811522960662842, "rewards/code_complexity_reward/std": 0.11708812415599823, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1642, "step_time": 58.10950386617333 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 434.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 114.30078125, "completions/mean_terminated_length": 114.30078125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23688168195076287, "epoch": 0.9361823361823362, "frac_reward_zero_std": 0.3125, "grad_norm": 0.09029491245746613, "kl": 0.22410123678855598, "learning_rate": 6.29176176568927e-08, "loss": 0.0011209596414119005, "num_tokens": 240028349.0, "reward": 2.4290037155151367, "reward_std": 0.5038827657699585, "rewards/code_complexity_reward/mean": 0.9833007454872131, "rewards/code_complexity_reward/std": 0.09087228775024414, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1643, "step_time": 43.31000852584839 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 114.94140625, "completions/mean_terminated_length": 114.94140625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.24327644566074014, "epoch": 0.9367521367521368, "frac_reward_zero_std": 0.3125, "grad_norm": 0.08896277844905853, "kl": 0.24173195660114288, "learning_rate": 6.181355050982496e-08, "loss": 0.0012092425022274256, "num_tokens": 240156975.0, "reward": 2.3850584030151367, "reward_std": 0.5040019750595093, "rewards/code_complexity_reward/mean": 0.9803711175918579, "rewards/code_complexity_reward/std": 0.10934987664222717, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1644, "step_time": 41.440633077174425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 122.267578125, "completions/mean_terminated_length": 122.267578125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24342207377776504, "epoch": 0.9373219373219374, "frac_reward_zero_std": 0.3125, "grad_norm": 0.10372038930654526, "kl": 0.2285087730269879, "learning_rate": 6.071913502810944e-08, "loss": 0.001142807537689805, "num_tokens": 240289560.0, "reward": 2.3681640625, "reward_std": 0.4718308746814728, "rewards/code_complexity_reward/mean": 0.9830078482627869, "rewards/code_complexity_reward/std": 0.07039402425289154, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1645, "step_time": 38.90504914987832 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 294.0, "completions/mean_length": 114.279296875, "completions/mean_terminated_length": 113.5009765625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24723520455881953, "epoch": 0.9378917378917379, "frac_reward_zero_std": 0.234375, "grad_norm": 0.10770728439092636, "kl": 0.23994127148762345, "learning_rate": 5.96343755440365e-08, "loss": 0.0012001104187220335, "num_tokens": 240416023.0, "reward": 2.3642578125, "reward_std": 0.5026464462280273, "rewards/code_complexity_reward/mean": 0.976367175579071, "rewards/code_complexity_reward/std": 0.1118169054389, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 1646, "step_time": 47.38648602832109 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 110.783203125, "completions/mean_terminated_length": 110.783203125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24887246661819518, "epoch": 0.9384615384615385, "frac_reward_zero_std": 0.34375, "grad_norm": 0.11031878739595413, "kl": 0.23959998018108308, "learning_rate": 5.855927635167319e-08, "loss": 0.001198378624394536, "num_tokens": 240542696.0, "reward": 2.3924806118011475, "reward_std": 0.518037736415863, "rewards/code_complexity_reward/mean": 0.97509765625, "rewards/code_complexity_reward/std": 0.12086604535579681, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1647, "step_time": 48.70031900238246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 345.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 116.96484375, "completions/mean_terminated_length": 116.96484375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24693521251901984, "epoch": 0.9390313390313391, "frac_reward_zero_std": 0.34375, "grad_norm": 0.08989310264587402, "kl": 0.22232410381548107, "learning_rate": 5.749384170684574e-08, "loss": 0.0011123442091047764, "num_tokens": 240672798.0, "reward": 2.3429198265075684, "reward_std": 0.49179163575172424, "rewards/code_complexity_reward/mean": 0.976757824420929, "rewards/code_complexity_reward/std": 0.11790256202220917, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1648, "step_time": 55.260687625035644 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 111.041015625, "completions/mean_terminated_length": 111.041015625, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.2328549767844379, "epoch": 0.9396011396011396, "frac_reward_zero_std": 0.375, "grad_norm": 0.08372664451599121, "kl": 0.24597032321617007, "learning_rate": 5.643807582712213e-08, "loss": 0.0012301987735554576, "num_tokens": 240798043.0, "reward": 2.3411130905151367, "reward_std": 0.4957491457462311, "rewards/code_complexity_reward/mean": 0.9793945550918579, "rewards/code_complexity_reward/std": 0.12461378425359726, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1649, "step_time": 40.66769239958376 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 335.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 114.69921875, "completions/mean_terminated_length": 114.69921875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.24262817576527596, "epoch": 0.9401709401709402, "frac_reward_zero_std": 0.34375, "grad_norm": 0.11716098338365555, "kl": 0.23610139288939536, "learning_rate": 5.539198289179759e-08, "loss": 0.0011810266878455877, "num_tokens": 240927281.0, "reward": 2.295849323272705, "reward_std": 0.4794439971446991, "rewards/code_complexity_reward/mean": 0.9716796875, "rewards/code_complexity_reward/std": 0.1277484893798828, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1650, "step_time": 46.98453640099615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 117.365234375, "completions/mean_terminated_length": 116.59295654296875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.25427398341707885, "epoch": 0.9407407407407408, "frac_reward_zero_std": 0.359375, "grad_norm": 0.08773362636566162, "kl": 0.25049095251597464, "learning_rate": 5.435556704187522e-08, "loss": 0.001252694521099329, "num_tokens": 241056940.0, "reward": 2.3936033248901367, "reward_std": 0.5257686972618103, "rewards/code_complexity_reward/mean": 0.9735351800918579, "rewards/code_complexity_reward/std": 0.12701527774333954, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1651, "step_time": 49.230219474993646 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 113.7109375, "completions/mean_terminated_length": 113.7109375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24274990428239107, "epoch": 0.9413105413105413, "frac_reward_zero_std": 0.421875, "grad_norm": 0.07767631113529205, "kl": 0.24050836800597608, "learning_rate": 5.332883238005154e-08, "loss": 0.0012026445474475622, "num_tokens": 241185824.0, "reward": 2.3077149391174316, "reward_std": 0.4561002850532532, "rewards/code_complexity_reward/mean": 0.980175793170929, "rewards/code_complexity_reward/std": 0.1022380143404007, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1652, "step_time": 50.57556617446244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 283.0, "completions/max_terminated_length": 283.0, "completions/mean_length": 112.72265625, "completions/mean_terminated_length": 112.72265625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24452250800095499, "epoch": 0.9418803418803419, "frac_reward_zero_std": 0.265625, "grad_norm": 0.10508813709020615, "kl": 0.2782375586684793, "learning_rate": 5.231178297069983e-08, "loss": 0.0013919929042458534, "num_tokens": 241310818.0, "reward": 2.3697752952575684, "reward_std": 0.4878772795200348, "rewards/code_complexity_reward/mean": 0.97998046875, "rewards/code_complexity_reward/std": 0.10205616056919098, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1653, "step_time": 44.0243728980422 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 295.0, "completions/max_terminated_length": 295.0, "completions/mean_length": 113.904296875, "completions/mean_terminated_length": 113.904296875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24655901407822967, "epoch": 0.9424501424501425, "frac_reward_zero_std": 0.328125, "grad_norm": 0.09160270541906357, "kl": 0.27051283954642713, "learning_rate": 5.130442283985321e-08, "loss": 0.0013529825955629349, "num_tokens": 241436553.0, "reward": 2.365917921066284, "reward_std": 0.5179323554039001, "rewards/code_complexity_reward/mean": 0.966113269329071, "rewards/code_complexity_reward/std": 0.13450326025485992, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1654, "step_time": 43.69611132051796 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 112.798828125, "completions/mean_terminated_length": 112.01760864257812, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22800694056786597, "epoch": 0.9430199430199431, "frac_reward_zero_std": 0.453125, "grad_norm": 0.07796894013881683, "kl": 0.24910639133304358, "learning_rate": 5.0306755975190746e-08, "loss": 0.0012457503471523523, "num_tokens": 241561650.0, "reward": 2.493896484375, "reward_std": 0.558402419090271, "rewards/code_complexity_reward/mean": 0.9751952886581421, "rewards/code_complexity_reward/std": 0.1335938572883606, "rewards/code_execution_reward/mean": 0.427734375, "rewards/code_execution_reward/std": 0.4952339828014374, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1655, "step_time": 55.575582639314234 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 117.208984375, "completions/mean_terminated_length": 116.4364013671875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.24179829470813274, "epoch": 0.9435897435897436, "frac_reward_zero_std": 0.4375, "grad_norm": 0.08311553299427032, "kl": 0.23090132512152195, "learning_rate": 4.9318786326018334e-08, "loss": 0.0011546225287020206, "num_tokens": 241689877.0, "reward": 2.3191893100738525, "reward_std": 0.4867914617061615, "rewards/code_complexity_reward/mean": 0.975292980670929, "rewards/code_complexity_reward/std": 0.12656036019325256, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1656, "step_time": 68.56891671940684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 295.0, "completions/max_terminated_length": 295.0, "completions/mean_length": 112.7109375, "completions/mean_terminated_length": 112.7109375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24072721134871244, "epoch": 0.9441595441595442, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09109204262495041, "kl": 0.2610052137169987, "learning_rate": 4.8340517803256714e-08, "loss": 0.0013052786234766245, "num_tokens": 241816921.0, "reward": 2.4540038108825684, "reward_std": 0.49467673897743225, "rewards/code_complexity_reward/mean": 0.99072265625, "rewards/code_complexity_reward/std": 0.06506706774234772, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1657, "step_time": 41.470469580963254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 254.0, "completions/max_terminated_length": 254.0, "completions/mean_length": 111.88671875, "completions/mean_terminated_length": 111.88671875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2351250376086682, "epoch": 0.9447293447293448, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09592461585998535, "kl": 0.24581353552639484, "learning_rate": 4.737195427942376e-08, "loss": 0.0012292151805013418, "num_tokens": 241943479.0, "reward": 2.373779296875, "reward_std": 0.4989639222621918, "rewards/code_complexity_reward/mean": 0.9815429449081421, "rewards/code_complexity_reward/std": 0.10977722704410553, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.01657281443476677, "step": 1658, "step_time": 44.09441900253296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 424.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 113.978515625, "completions/mean_terminated_length": 113.978515625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24675831175409257, "epoch": 0.9452991452991453, "frac_reward_zero_std": 0.234375, "grad_norm": 0.1008991077542305, "kl": 0.24531378992833197, "learning_rate": 4.641309958861917e-08, "loss": 0.0012271841987967491, "num_tokens": 242069516.0, "reward": 2.4756836891174316, "reward_std": 0.5418128967285156, "rewards/code_complexity_reward/mean": 0.974316418170929, "rewards/code_complexity_reward/std": 0.12021474540233612, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1659, "step_time": 51.128915750421584 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 271.0, "completions/mean_length": 118.17578125, "completions/mean_terminated_length": 112.71683502197266, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23605242162011564, "epoch": 0.9458689458689459, "frac_reward_zero_std": 0.359375, "grad_norm": 0.10139289498329163, "kl": 0.2817030732985586, "learning_rate": 4.5463957526510894e-08, "loss": 0.0014084761496633291, "num_tokens": 242200278.0, "reward": 2.2818846702575684, "reward_std": 0.545002818107605, "rewards/code_complexity_reward/mean": 0.9562499523162842, "rewards/code_complexity_reward/std": 0.18423694372177124, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.026382790878415108, "step": 1660, "step_time": 80.50759978033602 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 289.0, "completions/max_terminated_length": 289.0, "completions/mean_length": 112.87109375, "completions/mean_terminated_length": 112.87109375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22808792255818844, "epoch": 0.9464387464387465, "frac_reward_zero_std": 0.3125, "grad_norm": 0.11004520207643509, "kl": 0.24004352698102593, "learning_rate": 4.452453185031763e-08, "loss": 0.0012009821366518736, "num_tokens": 242325444.0, "reward": 2.4089841842651367, "reward_std": 0.5121110081672668, "rewards/code_complexity_reward/mean": 0.9808593392372131, "rewards/code_complexity_reward/std": 0.10932476073503494, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1661, "step_time": 43.62429711967707 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 113.072265625, "completions/mean_terminated_length": 112.29158782958984, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24133162340149283, "epoch": 0.947008547008547, "frac_reward_zero_std": 0.25, "grad_norm": 0.10418011993169785, "kl": 0.24368975637480617, "learning_rate": 4.359482627879663e-08, "loss": 0.0012190258130431175, "num_tokens": 242449569.0, "reward": 2.360107421875, "reward_std": 0.495420902967453, "rewards/code_complexity_reward/mean": 0.9791015386581421, "rewards/code_complexity_reward/std": 0.11056151241064072, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1662, "step_time": 47.58469477947801 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 362.0, "completions/max_terminated_length": 362.0, "completions/mean_length": 118.953125, "completions/mean_terminated_length": 118.953125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24686267622746527, "epoch": 0.9475783475783476, "frac_reward_zero_std": 0.265625, "grad_norm": 0.09526114910840988, "kl": 0.2530980184674263, "learning_rate": 4.267484449222703e-08, "loss": 0.0012658273335546255, "num_tokens": 242581081.0, "reward": 2.3506345748901367, "reward_std": 0.500528872013092, "rewards/code_complexity_reward/mean": 0.9747070074081421, "rewards/code_complexity_reward/std": 0.12041965126991272, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1663, "step_time": 38.56666318979114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 333.0, "completions/max_terminated_length": 333.0, "completions/mean_length": 118.388671875, "completions/mean_terminated_length": 118.388671875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2431907244026661, "epoch": 0.9481481481481482, "frac_reward_zero_std": 0.203125, "grad_norm": 0.09157686680555344, "kl": 0.22591457073576748, "learning_rate": 4.176459013239598e-08, "loss": 0.001130102202296257, "num_tokens": 242710992.0, "reward": 2.310253858566284, "reward_std": 0.4900151193141937, "rewards/code_complexity_reward/mean": 0.9709961414337158, "rewards/code_complexity_reward/std": 0.13309349119663239, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1664, "step_time": 37.56535746343434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 113.1171875, "completions/mean_terminated_length": 113.1171875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24223545007407665, "epoch": 0.9487179487179487, "frac_reward_zero_std": 0.375, "grad_norm": 0.08562693744897842, "kl": 0.23456409620121121, "learning_rate": 4.08640668025842e-08, "loss": 0.001172977383248508, "num_tokens": 242835660.0, "reward": 2.3568849563598633, "reward_std": 0.5062116384506226, "rewards/code_complexity_reward/mean": 0.973925769329071, "rewards/code_complexity_reward/std": 0.12632396817207336, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1665, "step_time": 38.08482921682298 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 307.0, "completions/max_terminated_length": 307.0, "completions/mean_length": 116.546875, "completions/mean_terminated_length": 116.546875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.25578550226055086, "epoch": 0.9492877492877493, "frac_reward_zero_std": 0.40625, "grad_norm": 0.10016804933547974, "kl": 0.24551987322047353, "learning_rate": 3.9973278067552134e-08, "loss": 0.0012280060909688473, "num_tokens": 242964180.0, "reward": 2.2922849655151367, "reward_std": 0.46448639035224915, "rewards/code_complexity_reward/mean": 0.9764648675918579, "rewards/code_complexity_reward/std": 0.11823806166648865, "rewards/code_execution_reward/mean": 0.22265625, "rewards/code_execution_reward/std": 0.41643625497817993, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1666, "step_time": 35.421149510890245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 473.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 112.8984375, "completions/mean_terminated_length": 112.8984375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2302362045738846, "epoch": 0.9498575498575499, "frac_reward_zero_std": 0.3125, "grad_norm": 0.09721994400024414, "kl": 0.24047122430056334, "learning_rate": 3.9092227453524925e-08, "loss": 0.001202890183776617, "num_tokens": 243089200.0, "reward": 2.4343748092651367, "reward_std": 0.4951031804084778, "rewards/code_complexity_reward/mean": 0.9886718988418579, "rewards/code_complexity_reward/std": 0.06766003370285034, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1667, "step_time": 46.37356133013964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 297.0, "completions/max_terminated_length": 297.0, "completions/mean_length": 113.544921875, "completions/mean_terminated_length": 113.544921875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2510780915617943, "epoch": 0.9504273504273504, "frac_reward_zero_std": 0.28125, "grad_norm": 0.11645244061946869, "kl": 0.23819637554697692, "learning_rate": 3.82209184481791e-08, "loss": 0.0011913441121578217, "num_tokens": 243215575.0, "reward": 2.3685545921325684, "reward_std": 0.4930708110332489, "rewards/code_complexity_reward/mean": 0.98046875, "rewards/code_complexity_reward/std": 0.10198314487934113, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1668, "step_time": 51.90254649706185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 111.408203125, "completions/mean_terminated_length": 111.408203125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.25128316693007946, "epoch": 0.950997150997151, "frac_reward_zero_std": 0.46875, "grad_norm": 0.08772800862789154, "kl": 0.2720765923149884, "learning_rate": 3.7359354500628716e-08, "loss": 0.00136069324798882, "num_tokens": 243343592.0, "reward": 2.3370118141174316, "reward_std": 0.4624154567718506, "rewards/code_complexity_reward/mean": 0.9850585460662842, "rewards/code_complexity_reward/std": 0.09080201387405396, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1669, "step_time": 69.53057502768934 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 377.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 113.38671875, "completions/mean_terminated_length": 113.38671875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.24976722453720868, "epoch": 0.9515669515669516, "frac_reward_zero_std": 0.390625, "grad_norm": 0.0871535912156105, "kl": 0.23445947538129985, "learning_rate": 3.650753902141119e-08, "loss": 0.001172569114714861, "num_tokens": 243468390.0, "reward": 2.3876953125, "reward_std": 0.49544399976730347, "rewards/code_complexity_reward/mean": 0.9820312261581421, "rewards/code_complexity_reward/std": 0.1008252426981926, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1670, "step_time": 48.84779101610184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 116.732421875, "completions/mean_terminated_length": 115.95890045166016, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24112956668250263, "epoch": 0.9521367521367521, "frac_reward_zero_std": 0.265625, "grad_norm": 0.10287943482398987, "kl": 0.23325035674497485, "learning_rate": 3.566547538247506e-08, "loss": 0.0011667516082525253, "num_tokens": 243598261.0, "reward": 2.376708984375, "reward_std": 0.513626217842102, "rewards/code_complexity_reward/mean": 0.9742187261581421, "rewards/code_complexity_reward/std": 0.12574386596679688, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1671, "step_time": 53.770586238242686 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 281.0, "completions/max_terminated_length": 281.0, "completions/mean_length": 112.55859375, "completions/mean_terminated_length": 112.55859375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2436845637857914, "epoch": 0.9527065527065527, "frac_reward_zero_std": 0.390625, "grad_norm": 0.08625148981809616, "kl": 0.237091958289966, "learning_rate": 3.4833166917164766e-08, "loss": 0.0011850579176098108, "num_tokens": 243723739.0, "reward": 2.3841795921325684, "reward_std": 0.5160664319992065, "rewards/code_complexity_reward/mean": 0.9755859375, "rewards/code_complexity_reward/std": 0.1253942847251892, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1672, "step_time": 33.772884737700224 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 257.0, "completions/mean_length": 108.9375, "completions/mean_terminated_length": 108.14872741699219, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24180269800126553, "epoch": 0.9532763532763533, "frac_reward_zero_std": 0.375, "grad_norm": 0.1060832217335701, "kl": 0.2488896723370999, "learning_rate": 3.4010616920209516e-08, "loss": 0.0012449431233108044, "num_tokens": 243845163.0, "reward": 2.4769043922424316, "reward_std": 0.5433056950569153, "rewards/code_complexity_reward/mean": 0.974804699420929, "rewards/code_complexity_reward/std": 0.12555146217346191, "rewards/code_execution_reward/mean": 0.41015625, "rewards/code_execution_reward/std": 0.49234291911125183, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1673, "step_time": 48.26884107571095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 489.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 115.068359375, "completions/mean_terminated_length": 115.068359375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2412599897943437, "epoch": 0.9538461538461539, "frac_reward_zero_std": 0.375, "grad_norm": 0.0986233726143837, "kl": 0.24190578539855778, "learning_rate": 3.3197828647708316e-08, "loss": 0.0012103037443012, "num_tokens": 243974510.0, "reward": 2.36865234375, "reward_std": 0.48032599687576294, "rewards/code_complexity_reward/mean": 0.9815429449081421, "rewards/code_complexity_reward/std": 0.09250885993242264, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1674, "step_time": 53.5940457098186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 257.0, "completions/max_terminated_length": 257.0, "completions/mean_length": 107.25, "completions/mean_terminated_length": 107.25, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2478078247513622, "epoch": 0.9544159544159544, "frac_reward_zero_std": 0.390625, "grad_norm": 0.10649050027132034, "kl": 0.2463888693600893, "learning_rate": 3.2394805317118025e-08, "loss": 0.0012324275448918343, "num_tokens": 244097118.0, "reward": 2.386962890625, "reward_std": 0.5026856660842896, "rewards/code_complexity_reward/mean": 0.9827148914337158, "rewards/code_complexity_reward/std": 0.10867036879062653, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1675, "step_time": 32.05275264568627 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 119.619140625, "completions/mean_terminated_length": 118.85127258300781, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23299633874557912, "epoch": 0.954985754985755, "frac_reward_zero_std": 0.3125, "grad_norm": 0.07632996141910553, "kl": 0.2500879399012774, "learning_rate": 3.160155010724142e-08, "loss": 0.0012507661012932658, "num_tokens": 244229035.0, "reward": 2.3536133766174316, "reward_std": 0.4796026647090912, "rewards/code_complexity_reward/mean": 0.9808593988418579, "rewards/code_complexity_reward/std": 0.09218400716781616, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1676, "step_time": 51.68209268338978 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 109.640625, "completions/mean_terminated_length": 109.640625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2360074056778103, "epoch": 0.9555555555555556, "frac_reward_zero_std": 0.28125, "grad_norm": 0.10429877787828445, "kl": 0.24785100296139717, "learning_rate": 3.081806615821276e-08, "loss": 0.0012393281795084476, "num_tokens": 244352307.0, "reward": 2.4334959983825684, "reward_std": 0.5408607721328735, "rewards/code_complexity_reward/mean": 0.97314453125, "rewards/code_complexity_reward/std": 0.13277290761470795, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1677, "step_time": 37.030754558742046 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 477.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 116.62890625, "completions/mean_terminated_length": 116.62890625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2497057963628322, "epoch": 0.9561253561253561, "frac_reward_zero_std": 0.265625, "grad_norm": 0.12712234258651733, "kl": 0.2542129154317081, "learning_rate": 3.0044356571486964e-08, "loss": 0.0012718134094029665, "num_tokens": 244482221.0, "reward": 2.38232421875, "reward_std": 0.5108366012573242, "rewards/code_complexity_reward/mean": 0.9727539420127869, "rewards/code_complexity_reward/std": 0.1197471022605896, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1678, "step_time": 54.34221118502319 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 302.0, "completions/mean_length": 111.203125, "completions/mean_terminated_length": 110.41878509521484, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2414062328170985, "epoch": 0.9566951566951567, "frac_reward_zero_std": 0.421875, "grad_norm": 0.0938640832901001, "kl": 0.26346524711698294, "learning_rate": 2.9280424409826315e-08, "loss": 0.0013172437902539968, "num_tokens": 244608301.0, "reward": 2.385449171066284, "reward_std": 0.5179797410964966, "rewards/code_complexity_reward/mean": 0.9775390625, "rewards/code_complexity_reward/std": 0.1252918243408203, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1679, "step_time": 49.75224886648357 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 414.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 111.56640625, "completions/mean_terminated_length": 111.56640625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23554193042218685, "epoch": 0.9572649572649573, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09343074262142181, "kl": 0.237145192688331, "learning_rate": 2.852627269728958e-08, "loss": 0.0011858136858791113, "num_tokens": 244734863.0, "reward": 2.4919919967651367, "reward_std": 0.5299395322799683, "rewards/code_complexity_reward/mean": 0.9808593988418579, "rewards/code_complexity_reward/std": 0.10128742456436157, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1680, "step_time": 43.58377313334495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 301.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 114.623046875, "completions/mean_terminated_length": 114.623046875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24556115386076272, "epoch": 0.9578347578347578, "frac_reward_zero_std": 0.203125, "grad_norm": 0.1194952204823494, "kl": 0.2557187704369426, "learning_rate": 2.7781904419217632e-08, "loss": 0.0012790132313966751, "num_tokens": 244863174.0, "reward": 2.3329100608825684, "reward_std": 0.46406421065330505, "rewards/code_complexity_reward/mean": 0.983593761920929, "rewards/code_complexity_reward/std": 0.09084499627351761, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1681, "step_time": 34.36834750883281 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 326.0, "completions/max_terminated_length": 326.0, "completions/mean_length": 112.6796875, "completions/mean_terminated_length": 112.6796875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24400615971535444, "epoch": 0.9584045584045584, "frac_reward_zero_std": 0.28125, "grad_norm": 0.08646760880947113, "kl": 0.2903874556068331, "learning_rate": 2.7047322522224807e-08, "loss": 0.0014523647259920835, "num_tokens": 244988378.0, "reward": 2.3812499046325684, "reward_std": 0.5706795454025269, "rewards/code_complexity_reward/mean": 0.958984375, "rewards/code_complexity_reward/std": 0.17449212074279785, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1682, "step_time": 36.160043560899794 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 118.048828125, "completions/mean_terminated_length": 116.5039291381836, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2509991507977247, "epoch": 0.958974358974359, "frac_reward_zero_std": 0.3125, "grad_norm": 0.09498438239097595, "kl": 0.24057710426859558, "learning_rate": 2.632252991418477e-08, "loss": 0.0012032820377498865, "num_tokens": 245117595.0, "reward": 2.2774412631988525, "reward_std": 0.5245829820632935, "rewards/code_complexity_reward/mean": 0.9591796398162842, "rewards/code_complexity_reward/std": 0.1753210872411728, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1683, "step_time": 76.13957065995783 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 111.3046875, "completions/mean_terminated_length": 111.3046875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2391811916604638, "epoch": 0.9595441595441595, "frac_reward_zero_std": 0.53125, "grad_norm": 0.07542479038238525, "kl": 0.2506517115980387, "learning_rate": 2.560752946421996e-08, "loss": 0.001253487542271614, "num_tokens": 245242023.0, "reward": 2.3514647483825684, "reward_std": 0.4702320694923401, "rewards/code_complexity_reward/mean": 0.986328125, "rewards/code_complexity_reward/std": 0.09021931141614914, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 1684, "step_time": 37.11726716533303 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 118.69921875, "completions/mean_terminated_length": 117.92955017089844, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24224996520206332, "epoch": 0.9601139601139601, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09698266535997391, "kl": 0.23580612987279892, "learning_rate": 2.4902324002690492e-08, "loss": 0.0011795489117503166, "num_tokens": 245369829.0, "reward": 2.3388185501098633, "reward_std": 0.512219250202179, "rewards/code_complexity_reward/mean": 0.97021484375, "rewards/code_complexity_reward/std": 0.1394585222005844, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1685, "step_time": 56.3857378968969 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 336.0, "completions/max_terminated_length": 336.0, "completions/mean_length": 113.314453125, "completions/mean_terminated_length": 113.314453125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24558756360784173, "epoch": 0.9606837606837607, "frac_reward_zero_std": 0.375, "grad_norm": 0.08301832526922226, "kl": 0.25560035021044314, "learning_rate": 2.4206916321182217e-08, "loss": 0.0012782355770468712, "num_tokens": 245496062.0, "reward": 2.347119092941284, "reward_std": 0.536903977394104, "rewards/code_complexity_reward/mean": 0.96533203125, "rewards/code_complexity_reward/std": 0.15908338129520416, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1686, "step_time": 43.991345243528485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 300.0, "completions/max_terminated_length": 300.0, "completions/mean_length": 112.416015625, "completions/mean_terminated_length": 112.416015625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2480104558635503, "epoch": 0.9612535612535612, "frac_reward_zero_std": 0.390625, "grad_norm": 0.10631856322288513, "kl": 0.24274367606267333, "learning_rate": 2.3521309172496177e-08, "loss": 0.0012142385821789503, "num_tokens": 245625115.0, "reward": 2.368945360183716, "reward_std": 0.508405327796936, "rewards/code_complexity_reward/mean": 0.9779296517372131, "rewards/code_complexity_reward/std": 0.1248529851436615, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1687, "step_time": 35.449187946505845 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 112.515625, "completions/mean_terminated_length": 112.515625, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.2468067486770451, "epoch": 0.9618233618233618, "frac_reward_zero_std": 0.34375, "grad_norm": 0.1108422577381134, "kl": 0.2501549383159727, "learning_rate": 2.28455052706375e-08, "loss": 0.0012514109257608652, "num_tokens": 245750547.0, "reward": 2.2981443405151367, "reward_std": 0.474920392036438, "rewards/code_complexity_reward/mean": 0.9754883050918579, "rewards/code_complexity_reward/std": 0.12706130743026733, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1688, "step_time": 38.44844731222838 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 323.0, "completions/max_terminated_length": 323.0, "completions/mean_length": 113.53125, "completions/mean_terminated_length": 113.53125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2472858231049031, "epoch": 0.9623931623931624, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09597176313400269, "kl": 0.2414932178799063, "learning_rate": 2.217950729080487e-08, "loss": 0.001207580091431737, "num_tokens": 245876675.0, "reward": 2.3667478561401367, "reward_std": 0.47259432077407837, "rewards/code_complexity_reward/mean": 0.9867187738418579, "rewards/code_complexity_reward/std": 0.07919219136238098, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1689, "step_time": 37.60329850111157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 268.0, "completions/max_terminated_length": 268.0, "completions/mean_length": 111.833984375, "completions/mean_terminated_length": 111.833984375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24622282898053527, "epoch": 0.9629629629629629, "frac_reward_zero_std": 0.375, "grad_norm": 0.09614239633083344, "kl": 0.2446889285929501, "learning_rate": 2.1523317869379945e-08, "loss": 0.0012238838244229555, "num_tokens": 246002414.0, "reward": 2.364941120147705, "reward_std": 0.5225880146026611, "rewards/code_complexity_reward/mean": 0.9719725847244263, "rewards/code_complexity_reward/std": 0.13982312381267548, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1690, "step_time": 37.02309064287692 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 113.25, "completions/mean_terminated_length": 113.25, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.25649915053509176, "epoch": 0.9635327635327635, "frac_reward_zero_std": 0.375, "grad_norm": 0.08928929269313812, "kl": 0.24776445026509464, "learning_rate": 2.0876939603916567e-08, "loss": 0.0012394496006891131, "num_tokens": 246128742.0, "reward": 2.3004393577575684, "reward_std": 0.4810051918029785, "rewards/code_complexity_reward/mean": 0.9772460460662842, "rewards/code_complexity_reward/std": 0.13221296668052673, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1691, "step_time": 37.259585263207555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 111.267578125, "completions/mean_terminated_length": 111.267578125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2498257129918784, "epoch": 0.9641025641025641, "frac_reward_zero_std": 0.375, "grad_norm": 0.09284526854753494, "kl": 0.23096192814409733, "learning_rate": 2.0240375053130755e-08, "loss": 0.001155316480435431, "num_tokens": 246254175.0, "reward": 2.3492674827575684, "reward_std": 0.4868532419204712, "rewards/code_complexity_reward/mean": 0.9801757335662842, "rewards/code_complexity_reward/std": 0.10998384654521942, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1692, "step_time": 39.355539133772254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 225.0, "completions/max_terminated_length": 225.0, "completions/mean_length": 114.736328125, "completions/mean_terminated_length": 114.736328125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24282618472352624, "epoch": 0.9646723646723647, "frac_reward_zero_std": 0.390625, "grad_norm": 0.0978085994720459, "kl": 0.24112857854925096, "learning_rate": 1.9613626736890433e-08, "loss": 0.0012061640154570341, "num_tokens": 246380792.0, "reward": 2.3292479515075684, "reward_std": 0.4916178584098816, "rewards/code_complexity_reward/mean": 0.9775390625, "rewards/code_complexity_reward/std": 0.12540891766548157, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1693, "step_time": 38.28296219278127 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 330.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 113.091796875, "completions/mean_terminated_length": 113.091796875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.236777959857136, "epoch": 0.9652421652421652, "frac_reward_zero_std": 0.3125, "grad_norm": 0.0913449227809906, "kl": 0.255605666898191, "learning_rate": 1.8996697136205443e-08, "loss": 0.0012781221885234118, "num_tokens": 246505031.0, "reward": 2.406445264816284, "reward_std": 0.5288889408111572, "rewards/code_complexity_reward/mean": 0.9705078601837158, "rewards/code_complexity_reward/std": 0.12725207209587097, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1694, "step_time": 36.25972953811288 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 229.0, "completions/mean_length": 109.265625, "completions/mean_terminated_length": 108.47749328613281, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24112101597711444, "epoch": 0.9658119658119658, "frac_reward_zero_std": 0.40625, "grad_norm": 0.1093076542019844, "kl": 0.2463461661245674, "learning_rate": 1.8389588693218384e-08, "loss": 0.0012322681723162532, "num_tokens": 246632111.0, "reward": 2.2805662155151367, "reward_std": 0.4688650071620941, "rewards/code_complexity_reward/mean": 0.9771484136581421, "rewards/code_complexity_reward/std": 0.13232555985450745, "rewards/code_execution_reward/mean": 0.212890625, "rewards/code_execution_reward/std": 0.409751296043396, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1695, "step_time": 66.63737570121884 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 278.0, "completions/max_terminated_length": 278.0, "completions/mean_length": 111.861328125, "completions/mean_terminated_length": 111.861328125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23890719003975391, "epoch": 0.9663817663817664, "frac_reward_zero_std": 0.296875, "grad_norm": 0.09785637259483337, "kl": 0.26426368462853134, "learning_rate": 1.7792303811193513e-08, "loss": 0.0013217454543337226, "num_tokens": 246757528.0, "reward": 2.37841796875, "reward_std": 0.5208359360694885, "rewards/code_complexity_reward/mean": 0.9747070074081421, "rewards/code_complexity_reward/std": 0.1329328715801239, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1696, "step_time": 41.65929834358394 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 294.0, "completions/max_terminated_length": 294.0, "completions/mean_length": 110.19140625, "completions/mean_terminated_length": 110.19140625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23149393033236265, "epoch": 0.9669515669515669, "frac_reward_zero_std": 0.359375, "grad_norm": 0.07500239461660385, "kl": 0.24132779170759022, "learning_rate": 1.7204844854509238e-08, "loss": 0.001207007560878992, "num_tokens": 246880442.0, "reward": 2.3558592796325684, "reward_std": 0.47488269209861755, "rewards/code_complexity_reward/mean": 0.982421875, "rewards/code_complexity_reward/std": 0.09169843792915344, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1697, "step_time": 34.607611963525414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 367.0, "completions/max_terminated_length": 367.0, "completions/mean_length": 110.94140625, "completions/mean_terminated_length": 110.94140625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2428890981245786, "epoch": 0.9675213675213675, "frac_reward_zero_std": 0.40625, "grad_norm": 0.08990396559238434, "kl": 0.2585985909681767, "learning_rate": 1.6627214148646486e-08, "loss": 0.0012933843536302447, "num_tokens": 247008324.0, "reward": 2.445507764816284, "reward_std": 0.5151514410972595, "rewards/code_complexity_reward/mean": 0.9832030534744263, "rewards/code_complexity_reward/std": 0.10024965554475784, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1698, "step_time": 45.9287584265694 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 119.794921875, "completions/mean_terminated_length": 119.794921875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.2530367646832019, "epoch": 0.9680911680911681, "frac_reward_zero_std": 0.3125, "grad_norm": 0.10064608603715897, "kl": 0.257652168860659, "learning_rate": 1.605941398018118e-08, "loss": 0.001288226107135415, "num_tokens": 247140563.0, "reward": 2.357666015625, "reward_std": 0.4983152449131012, "rewards/code_complexity_reward/mean": 0.9778319597244263, "rewards/code_complexity_reward/std": 0.1172991394996643, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1699, "step_time": 41.14171962440014 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 113.115234375, "completions/mean_terminated_length": 113.115234375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.24494162038899958, "epoch": 0.9686609686609686, "frac_reward_zero_std": 0.296875, "grad_norm": 0.1169181764125824, "kl": 0.258592784171924, "learning_rate": 1.5501446596774827e-08, "loss": 0.0012935937847942114, "num_tokens": 247265574.0, "reward": 2.3753905296325684, "reward_std": 0.4951722323894501, "rewards/code_complexity_reward/mean": 0.9794921875, "rewards/code_complexity_reward/std": 0.10145390033721924, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1700, "step_time": 40.135359023697674 }, { "epoch": 0.9686609686609686, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 156.18, "eval_completions/max_terminated_length": 156.18, "eval_completions/mean_length": 115.80625, "eval_completions/mean_terminated_length": 115.80625, "eval_completions/min_length": 87.41, "eval_completions/min_terminated_length": 87.41, "eval_entropy": 0.23891292050480842, "eval_frac_reward_zero_std": 0.36, "eval_kl": 0.23775239586830138, "eval_loss": 0.0011894424678757787, "eval_num_tokens": 247265574.0, "eval_reward": 2.3218749356269837, "eval_reward_std": 0.23853125374764203, "eval_rewards/code_complexity_reward/mean": 0.9687499988079071, "eval_rewards/code_complexity_reward/std": 0.05913862569257617, "eval_rewards/code_execution_reward/mean": 0.26375, "eval_rewards/code_execution_reward/std": 0.16458675742149353, "eval_rewards/code_syntax_reward/mean": 0.489375, "eval_rewards/code_syntax_reward/std": 0.024894515872001647, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.5, "eval_rewards/xmlcount_reward_func/std": 0.0, "eval_runtime": 719.7883, "eval_samples_per_second": 0.139, "eval_steps_per_second": 0.018, "step": 1700 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 113.578125, "completions/mean_terminated_length": 112.79843139648438, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.25466715707443655, "epoch": 0.9692307692307692, "frac_reward_zero_std": 0.40625, "grad_norm": 0.09974861890077591, "kl": 0.2528182282112539, "learning_rate": 1.4953314207164783e-08, "loss": 0.001264794496819377, "num_tokens": 247394166.0, "reward": 2.3709473609924316, "reward_std": 0.5273066759109497, "rewards/code_complexity_reward/mean": 0.9732421636581421, "rewards/code_complexity_reward/std": 0.13928399980068207, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 1701, "step_time": 51.122677938081324 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 279.0, "completions/max_terminated_length": 279.0, "completions/mean_length": 112.92578125, "completions/mean_terminated_length": 112.92578125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2505303039215505, "epoch": 0.9698005698005698, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09425324946641922, "kl": 0.25324742728844285, "learning_rate": 1.4415018981157047e-08, "loss": 0.001267195213586092, "num_tokens": 247522056.0, "reward": 2.2910642623901367, "reward_std": 0.4223043918609619, "rewards/code_complexity_reward/mean": 0.9852539300918579, "rewards/code_complexity_reward/std": 0.07970305532217026, "rewards/code_execution_reward/mean": 0.208984375, "rewards/code_execution_reward/std": 0.40698084235191345, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1702, "step_time": 39.87690868973732 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 110.814453125, "completions/mean_terminated_length": 110.814453125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2408555648289621, "epoch": 0.9703703703703703, "frac_reward_zero_std": 0.4375, "grad_norm": 0.0836089476943016, "kl": 0.25211879378184676, "learning_rate": 1.388656304961572e-08, "loss": 0.0012610855046659708, "num_tokens": 247648241.0, "reward": 2.382519245147705, "reward_std": 0.48253461718559265, "rewards/code_complexity_reward/mean": 0.986621081829071, "rewards/code_complexity_reward/std": 0.08515962213277817, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1703, "step_time": 38.44205625541508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 311.0, "completions/max_terminated_length": 311.0, "completions/mean_length": 111.23046875, "completions/mean_terminated_length": 111.23046875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23428315622732043, "epoch": 0.9709401709401709, "frac_reward_zero_std": 0.390625, "grad_norm": 0.20180480182170868, "kl": 0.4225056718569249, "learning_rate": 1.3367948504456607e-08, "loss": 0.002110927365720272, "num_tokens": 247772863.0, "reward": 2.43359375, "reward_std": 0.4982595443725586, "rewards/code_complexity_reward/mean": 0.9869140386581421, "rewards/code_complexity_reward/std": 0.07916299253702164, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1704, "step_time": 37.39001183398068 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 295.0, "completions/max_terminated_length": 295.0, "completions/mean_length": 116.16015625, "completions/mean_terminated_length": 116.16015625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24948852276429534, "epoch": 0.9715099715099715, "frac_reward_zero_std": 0.375, "grad_norm": 0.08528417348861694, "kl": 0.23775973008014262, "learning_rate": 1.2859177398637235e-08, "loss": 0.0011890314053744078, "num_tokens": 247903865.0, "reward": 2.2958984375, "reward_std": 0.4720311164855957, "rewards/code_complexity_reward/mean": 0.97607421875, "rewards/code_complexity_reward/std": 0.12503939867019653, "rewards/code_execution_reward/mean": 0.228515625, "rewards/code_execution_reward/std": 0.4202871024608612, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1705, "step_time": 53.14745965600014 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 283.0, "completions/max_terminated_length": 283.0, "completions/mean_length": 110.296875, "completions/mean_terminated_length": 110.296875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.23837458877824247, "epoch": 0.972079772079772, "frac_reward_zero_std": 0.34375, "grad_norm": 0.1054973229765892, "kl": 0.2390367235057056, "learning_rate": 1.2360251746150187e-08, "loss": 0.0011955315712839365, "num_tokens": 248028713.0, "reward": 2.3893065452575684, "reward_std": 0.4949837923049927, "rewards/code_complexity_reward/mean": 0.984082043170929, "rewards/code_complexity_reward/std": 0.10012485831975937, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1706, "step_time": 34.20159657485783 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 115.0078125, "completions/mean_terminated_length": 115.0078125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.25079614855349064, "epoch": 0.9726495726495726, "frac_reward_zero_std": 0.46875, "grad_norm": 0.08582890033721924, "kl": 0.23781169089488685, "learning_rate": 1.1871173522013946e-08, "loss": 0.001189434784464538, "num_tokens": 248156757.0, "reward": 2.296093702316284, "reward_std": 0.4833877980709076, "rewards/code_complexity_reward/mean": 0.972460925579071, "rewards/code_complexity_reward/std": 0.13528038561344147, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1707, "step_time": 43.837696368806064 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 111.599609375, "completions/mean_terminated_length": 111.599609375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23720307578332722, "epoch": 0.9732193732193732, "frac_reward_zero_std": 0.375, "grad_norm": 0.09874870628118515, "kl": 0.23649233928881586, "learning_rate": 1.1391944662265398e-08, "loss": 0.0011820397339761257, "num_tokens": 248280992.0, "reward": 2.3882811069488525, "reward_std": 0.5205458998680115, "rewards/code_complexity_reward/mean": 0.975781261920929, "rewards/code_complexity_reward/std": 0.12523706257343292, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1708, "step_time": 35.66940994746983 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 112.44921875, "completions/mean_terminated_length": 112.44921875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.240047593601048, "epoch": 0.9737891737891737, "frac_reward_zero_std": 0.375, "grad_norm": 0.09891968965530396, "kl": 0.2562828306108713, "learning_rate": 1.0922567063952338e-08, "loss": 0.0012818749528378248, "num_tokens": 248407566.0, "reward": 2.333935499191284, "reward_std": 0.46493688225746155, "rewards/code_complexity_reward/mean": 0.984179675579071, "rewards/code_complexity_reward/std": 0.09105658531188965, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1709, "step_time": 40.532994873821735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 231.0, "completions/max_terminated_length": 231.0, "completions/mean_length": 112.38671875, "completions/mean_terminated_length": 112.38671875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23652938986197114, "epoch": 0.9743589743589743, "frac_reward_zero_std": 0.3125, "grad_norm": 0.09554249048233032, "kl": 0.28610371542163193, "learning_rate": 1.0463042585126538e-08, "loss": 0.0014309673570096493, "num_tokens": 248532612.0, "reward": 2.4090332984924316, "reward_std": 0.5022870302200317, "rewards/code_complexity_reward/mean": 0.9823242425918579, "rewards/code_complexity_reward/std": 0.10046403855085373, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1710, "step_time": 31.008726792410016 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 376.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 112.3984375, "completions/mean_terminated_length": 112.3984375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24037361447699368, "epoch": 0.9749287749287749, "frac_reward_zero_std": 0.46875, "grad_norm": 0.08321024477481842, "kl": 0.2768084378913045, "learning_rate": 1.0013373044835128e-08, "loss": 0.0013845828361809254, "num_tokens": 248664888.0, "reward": 2.3641111850738525, "reward_std": 0.5545111894607544, "rewards/code_complexity_reward/mean": 0.9647461175918579, "rewards/code_complexity_reward/std": 0.1649358868598938, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 1711, "step_time": 47.19532722234726 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 110.931640625, "completions/mean_terminated_length": 110.931640625, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.23444117256440222, "epoch": 0.9754985754985755, "frac_reward_zero_std": 0.390625, "grad_norm": 0.09134510159492493, "kl": 0.23335667443461716, "learning_rate": 9.573560223113953e-09, "loss": 0.0011671993415802717, "num_tokens": 248790677.0, "reward": 2.362060546875, "reward_std": 0.4855266511440277, "rewards/code_complexity_reward/mean": 0.982226550579071, "rewards/code_complexity_reward/std": 0.10076286643743515, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1712, "step_time": 81.87954989820719 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 413.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 118.60546875, "completions/mean_terminated_length": 118.60546875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2499518778640777, "epoch": 0.976068376068376, "frac_reward_zero_std": 0.484375, "grad_norm": 0.09812954068183899, "kl": 0.24863347713835537, "learning_rate": 9.143605860981175e-09, "loss": 0.0012440988793969154, "num_tokens": 248923147.0, "reward": 2.3251466751098633, "reward_std": 0.47505658864974976, "rewards/code_complexity_reward/mean": 0.9812500476837158, "rewards/code_complexity_reward/std": 0.10907904803752899, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1713, "step_time": 49.79117651004344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 223.0, "completions/max_terminated_length": 223.0, "completions/mean_length": 106.732421875, "completions/mean_terminated_length": 106.732421875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.25399200106039643, "epoch": 0.9766381766381766, "frac_reward_zero_std": 0.328125, "grad_norm": 0.11024760454893112, "kl": 0.23818062874488533, "learning_rate": 8.723511660429229e-09, "loss": 0.0011914168717339635, "num_tokens": 249046426.0, "reward": 2.3645505905151367, "reward_std": 0.5218214392662048, "rewards/code_complexity_reward/mean": 0.9715820550918579, "rewards/code_complexity_reward/std": 0.1396390199661255, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1714, "step_time": 30.219928804785013 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 315.0, "completions/max_terminated_length": 315.0, "completions/mean_length": 112.0390625, "completions/mean_terminated_length": 112.0390625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2500213794410229, "epoch": 0.9772079772079773, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09490318596363068, "kl": 0.2576726458501071, "learning_rate": 8.313279284419273e-09, "loss": 0.0012891341466456652, "num_tokens": 249169094.0, "reward": 2.4072751998901367, "reward_std": 0.5379366278648376, "rewards/code_complexity_reward/mean": 0.9715820550918579, "rewards/code_complexity_reward/std": 0.13925309479236603, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1715, "step_time": 35.63325234968215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 301.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 115.30859375, "completions/mean_terminated_length": 115.30859375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2388848348055035, "epoch": 0.9777777777777777, "frac_reward_zero_std": 0.375, "grad_norm": 0.10692448914051056, "kl": 0.2257627248764038, "learning_rate": 7.912910356873416e-09, "loss": 0.0011290812399238348, "num_tokens": 249296244.0, "reward": 2.302294969558716, "reward_std": 0.4793442487716675, "rewards/code_complexity_reward/mean": 0.9761718511581421, "rewards/code_complexity_reward/std": 0.12728765606880188, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1716, "step_time": 35.64723670296371 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 435.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 110.0703125, "completions/mean_terminated_length": 110.0703125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2375319895800203, "epoch": 0.9783475783475784, "frac_reward_zero_std": 0.40625, "grad_norm": 0.09780952334403992, "kl": 0.23915586457587779, "learning_rate": 7.522406462669718e-09, "loss": 0.001195969758555293, "num_tokens": 249421384.0, "reward": 2.4014647006988525, "reward_std": 0.4920530915260315, "rewards/code_complexity_reward/mean": 0.9840819835662842, "rewards/code_complexity_reward/std": 0.08592502027750015, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1717, "step_time": 45.66790789179504 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 412.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 118.79296875, "completions/mean_terminated_length": 118.79296875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.23609367478638887, "epoch": 0.978917378917379, "frac_reward_zero_std": 0.359375, "grad_norm": 0.08971916884183884, "kl": 0.23719938658177853, "learning_rate": 7.14176914763387e-09, "loss": 0.0011863703839480877, "num_tokens": 249550630.0, "reward": 2.373242139816284, "reward_std": 0.5356625318527222, "rewards/code_complexity_reward/mean": 0.9656250476837158, "rewards/code_complexity_reward/std": 0.14668549597263336, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1718, "step_time": 56.96515004616231 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 113.302734375, "completions/mean_terminated_length": 113.302734375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.238754082005471, "epoch": 0.9794871794871794, "frac_reward_zero_std": 0.296875, "grad_norm": 0.10936617106199265, "kl": 0.25068459613248706, "learning_rate": 6.770999918535581e-09, "loss": 0.0012537792790681124, "num_tokens": 249676617.0, "reward": 2.427929639816284, "reward_std": 0.5113785862922668, "rewards/code_complexity_reward/mean": 0.9812500476837158, "rewards/code_complexity_reward/std": 0.10145709663629532, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1719, "step_time": 49.021375415846705 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 237.0, "completions/max_terminated_length": 237.0, "completions/mean_length": 106.4453125, "completions/mean_terminated_length": 106.4453125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24692555028013885, "epoch": 0.98005698005698, "frac_reward_zero_std": 0.375, "grad_norm": 0.07488406449556351, "kl": 0.26641771337017417, "learning_rate": 6.4101002430802525e-09, "loss": 0.0013325728941708803, "num_tokens": 249799365.0, "reward": 2.4449217319488525, "reward_std": 0.5273104906082153, "rewards/code_complexity_reward/mean": 0.9806640148162842, "rewards/code_complexity_reward/std": 0.11752981692552567, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1720, "step_time": 52.053451117128134 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 305.0, "completions/max_terminated_length": 305.0, "completions/mean_length": 113.75390625, "completions/mean_terminated_length": 113.75390625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23982750833965838, "epoch": 0.9806267806267807, "frac_reward_zero_std": 0.265625, "grad_norm": 0.10485323518514633, "kl": 0.2361449976451695, "learning_rate": 6.059071549904816e-09, "loss": 0.0011808264534920454, "num_tokens": 249926255.0, "reward": 2.405078172683716, "reward_std": 0.5146328210830688, "rewards/code_complexity_reward/mean": 0.9789062738418579, "rewards/code_complexity_reward/std": 0.11070127040147781, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1721, "step_time": 36.482109397649765 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 328.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 113.1328125, "completions/mean_terminated_length": 113.1328125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.25533042871393263, "epoch": 0.9811965811965812, "frac_reward_zero_std": 0.40625, "grad_norm": 0.0943119004368782, "kl": 0.24115686467848718, "learning_rate": 5.717915228571625e-09, "loss": 0.0012061001034453511, "num_tokens": 250052115.0, "reward": 2.3030271530151367, "reward_std": 0.4904624819755554, "rewards/code_complexity_reward/mean": 0.9725585579872131, "rewards/code_complexity_reward/std": 0.13948428630828857, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1722, "step_time": 45.94547493662685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 319.0, "completions/max_terminated_length": 319.0, "completions/mean_length": 111.59375, "completions/mean_terminated_length": 111.59375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.25048534036614, "epoch": 0.9817663817663818, "frac_reward_zero_std": 0.296875, "grad_norm": 0.10961824655532837, "kl": 0.25299224769696593, "learning_rate": 5.386632629562072e-09, "loss": 0.0012654135935008526, "num_tokens": 250179459.0, "reward": 2.368115186691284, "reward_std": 0.5498533844947815, "rewards/code_complexity_reward/mean": 0.962890625, "rewards/code_complexity_reward/std": 0.15946905314922333, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1723, "step_time": 48.0605193907395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 282.0, "completions/mean_length": 118.123046875, "completions/mean_terminated_length": 117.35224914550781, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24629138363525271, "epoch": 0.9823361823361824, "frac_reward_zero_std": 0.265625, "grad_norm": 0.09740828722715378, "kl": 0.23748833127319813, "learning_rate": 5.065225064272982e-09, "loss": 0.0011875627096742392, "num_tokens": 250308834.0, "reward": 2.3287596702575684, "reward_std": 0.5871081352233887, "rewards/code_complexity_reward/mean": 0.94775390625, "rewards/code_complexity_reward/std": 0.20245516300201416, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1724, "step_time": 53.44404618255794 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 378.0, "completions/max_terminated_length": 378.0, "completions/mean_length": 114.76171875, "completions/mean_terminated_length": 114.76171875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2365132626146078, "epoch": 0.9829059829059829, "frac_reward_zero_std": 0.328125, "grad_norm": 0.09157314896583557, "kl": 0.2568170577287674, "learning_rate": 4.753693805009674e-09, "loss": 0.0012846351601183414, "num_tokens": 250436640.0, "reward": 2.3782224655151367, "reward_std": 0.479358047246933, "rewards/code_complexity_reward/mean": 0.9842773675918579, "rewards/code_complexity_reward/std": 0.08128051459789276, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1725, "step_time": 48.723787458613515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 111.87109375, "completions/mean_terminated_length": 111.87109375, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.23164500249549747, "epoch": 0.9834757834757835, "frac_reward_zero_std": 0.359375, "grad_norm": 0.11606588214635849, "kl": 0.25831234687939286, "learning_rate": 4.452040084982068e-09, "loss": 0.0012917355634272099, "num_tokens": 250562110.0, "reward": 2.4189453125, "reward_std": 0.49527356028556824, "rewards/code_complexity_reward/mean": 0.9859374761581421, "rewards/code_complexity_reward/std": 0.07887107133865356, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1726, "step_time": 36.97635305300355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 265.0, "completions/mean_length": 114.291015625, "completions/mean_terminated_length": 113.5127182006836, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2461884052027017, "epoch": 0.9840455840455841, "frac_reward_zero_std": 0.375, "grad_norm": 0.10237487405538559, "kl": 0.24874266772530973, "learning_rate": 4.160265098299421e-09, "loss": 0.001244159648194909, "num_tokens": 250688179.0, "reward": 2.2767579555511475, "reward_std": 0.46887919306755066, "rewards/code_complexity_reward/mean": 0.97509765625, "rewards/code_complexity_reward/std": 0.13245384395122528, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1727, "step_time": 48.52388248126954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 115.822265625, "completions/mean_terminated_length": 115.822265625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2444087751209736, "epoch": 0.9846153846153847, "frac_reward_zero_std": 0.453125, "grad_norm": 0.08902976661920547, "kl": 0.24474081443622708, "learning_rate": 3.878369999965048e-09, "loss": 0.0012240058276802301, "num_tokens": 250817592.0, "reward": 2.3568358421325684, "reward_std": 0.49636855721473694, "rewards/code_complexity_reward/mean": 0.9785156846046448, "rewards/code_complexity_reward/std": 0.11860797554254532, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1728, "step_time": 41.307073006406426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 113.2109375, "completions/mean_terminated_length": 113.2109375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.25147146568633616, "epoch": 0.9851851851851852, "frac_reward_zero_std": 0.390625, "grad_norm": 0.0964927077293396, "kl": 0.23495840141549706, "learning_rate": 3.606355905873271e-09, "loss": 0.0011748683173209429, "num_tokens": 250946620.0, "reward": 2.3594727516174316, "reward_std": 0.48629260063171387, "rewards/code_complexity_reward/mean": 0.981152355670929, "rewards/code_complexity_reward/std": 0.10165577381849289, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1729, "step_time": 34.79360942449421 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 111.62109375, "completions/mean_terminated_length": 111.62109375, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.24319412000477314, "epoch": 0.9857549857549858, "frac_reward_zero_std": 0.390625, "grad_norm": 0.09364761412143707, "kl": 0.2558152354322374, "learning_rate": 3.3442238928030335e-09, "loss": 0.0012795613147318363, "num_tokens": 251071178.0, "reward": 2.385937452316284, "reward_std": 0.5231598019599915, "rewards/code_complexity_reward/mean": 0.978320300579071, "rewards/code_complexity_reward/std": 0.1321903020143509, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1730, "step_time": 40.067273918539286 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 111.732421875, "completions/mean_terminated_length": 110.9491195678711, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24955532932654023, "epoch": 0.9863247863247864, "frac_reward_zero_std": 0.296875, "grad_norm": 0.09955208003520966, "kl": 0.25661757634952664, "learning_rate": 3.091974998415681e-09, "loss": 0.0012832063948735595, "num_tokens": 251197193.0, "reward": 2.4055662155151367, "reward_std": 0.5343400835990906, "rewards/code_complexity_reward/mean": 0.9712890386581421, "rewards/code_complexity_reward/std": 0.13376198709011078, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1731, "step_time": 62.151282829232514 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 227.0, "completions/max_terminated_length": 227.0, "completions/mean_length": 110.01171875, "completions/mean_terminated_length": 110.01171875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24395787506364286, "epoch": 0.9868945868945869, "frac_reward_zero_std": 0.390625, "grad_norm": 0.09737244993448257, "kl": 0.23577213403768837, "learning_rate": 2.849610221248855e-09, "loss": 0.0011792309815064073, "num_tokens": 251322831.0, "reward": 2.387988328933716, "reward_std": 0.48845502734184265, "rewards/code_complexity_reward/mean": 0.9852539300918579, "rewards/code_complexity_reward/std": 0.09034796059131622, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1732, "step_time": 38.14650648459792 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 270.0, "completions/max_terminated_length": 270.0, "completions/mean_length": 111.86328125, "completions/mean_terminated_length": 111.86328125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23559318319894373, "epoch": 0.9874643874643875, "frac_reward_zero_std": 0.3125, "grad_norm": 0.08994467556476593, "kl": 0.25616392213851213, "learning_rate": 2.617130520714273e-09, "loss": 0.0012813166249543428, "num_tokens": 251450585.0, "reward": 2.362011432647705, "reward_std": 0.5186324715614319, "rewards/code_complexity_reward/mean": 0.9729492664337158, "rewards/code_complexity_reward/std": 0.13935023546218872, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1733, "step_time": 42.52515659946948 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 319.0, "completions/mean_length": 110.86328125, "completions/mean_terminated_length": 110.07827758789062, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24044357612729073, "epoch": 0.9880341880341881, "frac_reward_zero_std": 0.34375, "grad_norm": 0.10358155518770218, "kl": 0.26039011660031974, "learning_rate": 2.3945368170921745e-09, "loss": 0.001302691176533699, "num_tokens": 251576459.0, "reward": 2.3556151390075684, "reward_std": 0.4659060835838318, "rewards/code_complexity_reward/mean": 0.9873046875, "rewards/code_complexity_reward/std": 0.07978052645921707, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1734, "step_time": 48.17079260200262 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 336.0, "completions/mean_length": 112.85546875, "completions/mean_terminated_length": 112.0743637084961, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.24320611264556646, "epoch": 0.9886039886039886, "frac_reward_zero_std": 0.28125, "grad_norm": 0.12445064634084702, "kl": 0.237098973011598, "learning_rate": 2.1818299915296603e-09, "loss": 0.0011859429068863392, "num_tokens": 251701793.0, "reward": 2.393798828125, "reward_std": 0.5000653862953186, "rewards/code_complexity_reward/mean": 0.9825195074081421, "rewards/code_complexity_reward/std": 0.09971632808446884, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1735, "step_time": 48.744710221886635 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 114.849609375, "completions/mean_terminated_length": 114.849609375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.25476735085248947, "epoch": 0.9891737891737892, "frac_reward_zero_std": 0.34375, "grad_norm": 0.09459994733333588, "kl": 0.24297293717972934, "learning_rate": 1.979010886035693e-09, "loss": 0.0012152192648500204, "num_tokens": 251827852.0, "reward": 2.3047852516174316, "reward_std": 0.5004904866218567, "rewards/code_complexity_reward/mean": 0.969433605670929, "rewards/code_complexity_reward/std": 0.1470128893852234, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1736, "step_time": 40.487632968463004 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 372.0, "completions/max_terminated_length": 372.0, "completions/mean_length": 109.673828125, "completions/mean_terminated_length": 109.673828125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2361544561572373, "epoch": 0.9897435897435898, "frac_reward_zero_std": 0.453125, "grad_norm": 0.08504188805818558, "kl": 0.2627506775315851, "learning_rate": 1.7860803034785989e-09, "loss": 0.0013143252581357956, "num_tokens": 251951173.0, "reward": 2.42431640625, "reward_std": 0.5024981498718262, "rewards/code_complexity_reward/mean": 0.9864257574081421, "rewards/code_complexity_reward/std": 0.09053179621696472, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1737, "step_time": 39.99162161163986 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 117.2890625, "completions/mean_terminated_length": 116.51663208007812, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23750139377079904, "epoch": 0.9903133903133903, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09683092683553696, "kl": 0.24151420150883496, "learning_rate": 1.6030390075821856e-09, "loss": 0.0012080154847353697, "num_tokens": 252079441.0, "reward": 2.2923340797424316, "reward_std": 0.4939171075820923, "rewards/code_complexity_reward/mean": 0.968945324420929, "rewards/code_complexity_reward/std": 0.14582400023937225, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1738, "step_time": 48.59561058040708 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 314.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 118.244140625, "completions/mean_terminated_length": 118.244140625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2554116905666888, "epoch": 0.9908831908831909, "frac_reward_zero_std": 0.3125, "grad_norm": 0.07461263239383698, "kl": 0.23984426399692893, "learning_rate": 1.4298877229229623e-09, "loss": 0.0011996899265795946, "num_tokens": 252205286.0, "reward": 2.282470703125, "reward_std": 0.4843152165412903, "rewards/code_complexity_reward/mean": 0.9661133289337158, "rewards/code_complexity_reward/std": 0.14262403547763824, "rewards/code_execution_reward/mean": 0.2265625, "rewards/code_execution_reward/std": 0.4190165400505066, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1739, "step_time": 44.255554099567235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 112.248046875, "completions/mean_terminated_length": 112.248046875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2296392461284995, "epoch": 0.9914529914529915, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09893419593572617, "kl": 0.24685219558887184, "learning_rate": 1.2666271349282e-09, "loss": 0.0012345574796199799, "num_tokens": 252329645.0, "reward": 2.4175782203674316, "reward_std": 0.5244544148445129, "rewards/code_complexity_reward/mean": 0.9767577648162842, "rewards/code_complexity_reward/std": 0.11922300606966019, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1740, "step_time": 49.19994237739593 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 115.330078125, "completions/mean_terminated_length": 115.330078125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2327532097697258, "epoch": 0.992022792022792, "frac_reward_zero_std": 0.328125, "grad_norm": 0.0864955261349678, "kl": 0.23838193737901747, "learning_rate": 1.1132578898714884e-09, "loss": 0.0011920544784516096, "num_tokens": 252457574.0, "reward": 2.455517530441284, "reward_std": 0.511989414691925, "rewards/code_complexity_reward/mean": 0.98291015625, "rewards/code_complexity_reward/std": 0.09345471858978271, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1741, "step_time": 52.03790449351072 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 110.498046875, "completions/mean_terminated_length": 110.498046875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2398973668459803, "epoch": 0.9925925925925926, "frac_reward_zero_std": 0.28125, "grad_norm": 0.105681411921978, "kl": 0.2619665074162185, "learning_rate": 9.697805948716278e-10, "loss": 0.0013104794779792428, "num_tokens": 252583629.0, "reward": 2.3726561069488525, "reward_std": 0.5498214960098267, "rewards/code_complexity_reward/mean": 0.965039074420929, "rewards/code_complexity_reward/std": 0.1592804193496704, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1742, "step_time": 54.66792021505535 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 279.0, "completions/max_terminated_length": 279.0, "completions/mean_length": 115.212890625, "completions/mean_terminated_length": 115.212890625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24939421354793012, "epoch": 0.9931623931623932, "frac_reward_zero_std": 0.234375, "grad_norm": 0.11363429576158524, "kl": 0.24776997067965567, "learning_rate": 8.36195817889851e-10, "loss": 0.0012390433112159371, "num_tokens": 252710882.0, "reward": 2.3337888717651367, "reward_std": 0.5245078802108765, "rewards/code_complexity_reward/mean": 0.9681640863418579, "rewards/code_complexity_reward/std": 0.1522568315267563, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1743, "step_time": 40.57259862497449 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 238.0, "completions/max_terminated_length": 238.0, "completions/mean_length": 112.869140625, "completions/mean_terminated_length": 112.869140625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24900209228508174, "epoch": 0.9937321937321937, "frac_reward_zero_std": 0.375, "grad_norm": 0.08413927257061005, "kl": 0.23976012389175594, "learning_rate": 7.125040877273282e-10, "loss": 0.0011991772335022688, "num_tokens": 252836215.0, "reward": 2.3418946266174316, "reward_std": 0.5236460566520691, "rewards/code_complexity_reward/mean": 0.9704101085662842, "rewards/code_complexity_reward/std": 0.15150396525859833, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1744, "step_time": 40.421582685783505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 114.234375, "completions/mean_terminated_length": 114.234375, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.2407744110096246, "epoch": 0.9943019943019943, "frac_reward_zero_std": 0.359375, "grad_norm": 0.11520750820636749, "kl": 0.24873684835620224, "learning_rate": 5.987058940226665e-10, "loss": 0.0012442308943718672, "num_tokens": 252963223.0, "reward": 2.369824171066284, "reward_std": 0.4946806728839874, "rewards/code_complexity_reward/mean": 0.9778320789337158, "rewards/code_complexity_reward/std": 0.10204267501831055, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1745, "step_time": 47.47392059955746 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 299.0, "completions/max_terminated_length": 299.0, "completions/mean_length": 113.287109375, "completions/mean_terminated_length": 113.287109375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24745793803595006, "epoch": 0.9948717948717949, "frac_reward_zero_std": 0.328125, "grad_norm": 0.10606041550636292, "kl": 0.24292968935333192, "learning_rate": 4.94801687250801e-10, "loss": 0.0012147929519414902, "num_tokens": 253089570.0, "reward": 2.429931640625, "reward_std": 0.5348142385482788, "rewards/code_complexity_reward/mean": 0.9766601324081421, "rewards/code_complexity_reward/std": 0.12550164759159088, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1746, "step_time": 42.08667399454862 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 378.0, "completions/max_terminated_length": 378.0, "completions/mean_length": 117.005859375, "completions/mean_terminated_length": 117.005859375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23931660898961127, "epoch": 0.9954415954415955, "frac_reward_zero_std": 0.265625, "grad_norm": 0.11526336520910263, "kl": 0.2457624429371208, "learning_rate": 4.0079187872160696e-10, "loss": 0.0012295612832531333, "num_tokens": 253217893.0, "reward": 2.4231443405151367, "reward_std": 0.5055721402168274, "rewards/code_complexity_reward/mean": 0.9774414300918579, "rewards/code_complexity_reward/std": 0.09297168254852295, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1747, "step_time": 40.944787873886526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 336.0, "completions/max_terminated_length": 336.0, "completions/mean_length": 115.361328125, "completions/mean_terminated_length": 115.361328125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.24301472259685397, "epoch": 0.996011396011396, "frac_reward_zero_std": 0.421875, "grad_norm": 0.09310657531023026, "kl": 0.24742679204791784, "learning_rate": 3.16676840576291e-10, "loss": 0.0012374324724078178, "num_tokens": 253344646.0, "reward": 2.36474609375, "reward_std": 0.5059260725975037, "rewards/code_complexity_reward/mean": 0.9776366949081421, "rewards/code_complexity_reward/std": 0.12532885372638702, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1748, "step_time": 40.722092469222844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 338.0, "completions/max_terminated_length": 338.0, "completions/mean_length": 119.30859375, "completions/mean_terminated_length": 119.30859375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2486997228115797, "epoch": 0.9965811965811966, "frac_reward_zero_std": 0.359375, "grad_norm": 0.12708552181720734, "kl": 0.24493482639081776, "learning_rate": 2.424569057882242e-10, "loss": 0.0012255040928721428, "num_tokens": 253475436.0, "reward": 2.33984375, "reward_std": 0.482972651720047, "rewards/code_complexity_reward/mean": 0.9751952886581421, "rewards/code_complexity_reward/std": 0.10547610372304916, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1749, "step_time": 44.691528510302305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 399.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 113.126953125, "completions/mean_terminated_length": 113.126953125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2374540634918958, "epoch": 0.9971509971509972, "frac_reward_zero_std": 0.28125, "grad_norm": 0.09568040072917938, "kl": 0.22795722330920398, "learning_rate": 1.7813236815988898e-10, "loss": 0.001140194246545434, "num_tokens": 253601213.0, "reward": 2.4458985328674316, "reward_std": 0.5243883728981018, "rewards/code_complexity_reward/mean": 0.978710949420929, "rewards/code_complexity_reward/std": 0.10964226722717285, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 1750, "step_time": 41.06569554191083 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 113.671875, "completions/mean_terminated_length": 113.671875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.23735808674246073, "epoch": 0.9977207977207977, "frac_reward_zero_std": 0.359375, "grad_norm": 0.10009981691837311, "kl": 0.24695266759954393, "learning_rate": 1.2370348232315644e-10, "loss": 0.0012350832112133503, "num_tokens": 253726725.0, "reward": 2.3895020484924316, "reward_std": 0.4950769245624542, "rewards/code_complexity_reward/mean": 0.9813476800918579, "rewards/code_complexity_reward/std": 0.09310232847929001, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1751, "step_time": 44.92010587267578 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 316.0, "completions/mean_length": 115.134765625, "completions/mean_terminated_length": 114.35812377929688, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.24540540925227106, "epoch": 0.9982905982905983, "frac_reward_zero_std": 0.328125, "grad_norm": 0.09632980823516846, "kl": 0.24674632516689599, "learning_rate": 7.917046373651094e-11, "loss": 0.0012343438575044274, "num_tokens": 253856162.0, "reward": 2.3809568881988525, "reward_std": 0.4969192445278168, "rewards/code_complexity_reward/mean": 0.9779297113418579, "rewards/code_complexity_reward/std": 0.10237497836351395, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 1752, "step_time": 57.40056773275137 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 253.0, "completions/mean_length": 112.72265625, "completions/mean_terminated_length": 111.94129180908203, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.24426230695098639, "epoch": 0.9988603988603989, "frac_reward_zero_std": 0.359375, "grad_norm": 0.09384270012378693, "kl": 0.24587000021710992, "learning_rate": 4.453348868643792e-11, "loss": 0.0012299632653594017, "num_tokens": 253982188.0, "reward": 2.360790967941284, "reward_std": 0.5053161382675171, "rewards/code_complexity_reward/mean": 0.9749023914337158, "rewards/code_complexity_reward/std": 0.11984983086585999, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1753, "step_time": 48.369906444102526 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 111.853515625, "completions/mean_terminated_length": 111.853515625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23749089636839926, "epoch": 0.9994301994301994, "frac_reward_zero_std": 0.34375, "grad_norm": 0.08599209785461426, "kl": 0.24734876863658428, "learning_rate": 1.9792694284370695e-11, "loss": 0.0012371060438454151, "num_tokens": 254107889.0, "reward": 2.3826658725738525, "reward_std": 0.5099102258682251, "rewards/code_complexity_reward/mean": 0.9754883050918579, "rewards/code_complexity_reward/std": 0.11841151863336563, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1754, "step_time": 47.781810813583434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 120.533203125, "completions/mean_terminated_length": 119.76712036132812, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.24181423080153763, "epoch": 1.0, "frac_reward_zero_std": 0.3125, "grad_norm": 0.10063210874795914, "kl": 0.23375988146290183, "learning_rate": 4.948178468078269e-12, "loss": 0.0011692731641232967, "num_tokens": 254239242.0, "reward": 2.3416502475738525, "reward_std": 0.5240309834480286, "rewards/code_complexity_reward/mean": 0.9616210460662842, "rewards/code_complexity_reward/std": 0.1475723683834076, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 1755, "step_time": 57.00852202437818 }, { "epoch": 1.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.00125, "eval_completions/max_length": 159.12, "eval_completions/max_terminated_length": 155.91, "eval_completions/mean_length": 116.00625, "eval_completions/mean_terminated_length": 115.55482147216797, "eval_completions/min_length": 87.65, "eval_completions/min_terminated_length": 87.65, "eval_entropy": 0.23991589844226838, "eval_frac_reward_zero_std": 0.32, "eval_kl": 0.24852377265691758, "eval_loss": 0.0012423543957993388, "eval_num_tokens": 254239242.0, "eval_reward": 2.3418749356269837, "eval_reward_std": 0.24072861278429628, "eval_rewards/code_complexity_reward/mean": 0.9715624988079071, "eval_rewards/code_complexity_reward/std": 0.054564117956906556, "eval_rewards/code_execution_reward/mean": 0.28, "eval_rewards/code_execution_reward/std": 0.17363928526639938, "eval_rewards/code_syntax_reward/mean": 0.490625, "eval_rewards/code_syntax_reward/std": 0.022853553146123886, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.4996875, "eval_rewards/xmlcount_reward_func/std": 0.000883883461356163, "eval_runtime": 732.5805, "eval_samples_per_second": 0.137, "eval_steps_per_second": 0.018, "step": 1755 } ], "logging_steps": 1, "max_steps": 1755, "num_input_tokens_seen": 254239242, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 1, "trial_name": null, "trial_params": null }