{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 100, "global_step": 877, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 377.796875, "completions/mean_terminated_length": 342.75860595703125, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.28149473504163325, "epoch": 0.0011402508551881414, "frac_reward_zero_std": 0.015625, "grad_norm": 0.013365611433982849, "kl": 0.0, "learning_rate": 0.0, "loss": -6.51925802230835e-09, "num_tokens": 262056.0, "reward": 1.1783690452575684, "reward_std": 0.9925503134727478, "rewards/code_complexity_reward/mean": 0.46796873211860657, "rewards/code_complexity_reward/std": 0.4073677659034729, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.2919921875, "rewards/code_syntax_reward/std": 0.246689110994339, "rewards/reasoning_present_reward_func/mean": 0.03339843451976776, "rewards/reasoning_present_reward_func/std": 0.047209545969963074, "rewards/xmlcount_reward_func/mean": 0.154541015625, "rewards/xmlcount_reward_func/std": 0.20515048503875732, "step": 1, "step_time": 85.39325274899602 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.193359375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 374.267578125, "completions/mean_terminated_length": 341.2518310546875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.285196773475036, "epoch": 0.002280501710376283, "frac_reward_zero_std": 0.015625, "grad_norm": 0.014123172499239445, "kl": 0.0, "learning_rate": 5.681818181818182e-08, "loss": 1.076841726899147e-08, "num_tokens": 522913.0, "reward": 1.1459472179412842, "reward_std": 0.9580074548721313, "rewards/code_complexity_reward/mean": 0.4676758050918579, "rewards/code_complexity_reward/std": 0.409721702337265, "rewards/code_execution_reward/mean": 0.201171875, "rewards/code_execution_reward/std": 0.4012683033943176, "rewards/code_syntax_reward/mean": 0.2890625, "rewards/code_syntax_reward/std": 0.24717088043689728, "rewards/reasoning_present_reward_func/mean": 0.03398437798023224, "rewards/reasoning_present_reward_func/std": 0.04741192236542702, "rewards/xmlcount_reward_func/mean": 0.154052734375, "rewards/xmlcount_reward_func/std": 0.20773786306381226, "step": 2, "step_time": 74.93634564243257 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 367.19140625, "completions/mean_terminated_length": 335.471435546875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2805805513635278, "epoch": 0.0034207525655644243, "frac_reward_zero_std": 0.0, "grad_norm": 0.01401860173791647, "kl": 0.0008799334054856445, "learning_rate": 1.1363636363636364e-07, "loss": 4.401808837428689e-06, "num_tokens": 780863.0, "reward": 1.1578125953674316, "reward_std": 0.9575390815734863, "rewards/code_complexity_reward/mean": 0.4736328125, "rewards/code_complexity_reward/std": 0.4047275483608246, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.2958984375, "rewards/code_syntax_reward/std": 0.24599088728427887, "rewards/reasoning_present_reward_func/mean": 0.02988281287252903, "rewards/reasoning_present_reward_func/std": 0.045819200575351715, "rewards/xmlcount_reward_func/mean": 0.1416015625, "rewards/xmlcount_reward_func/std": 0.19960276782512665, "step": 3, "step_time": 70.137695110403 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 372.986328125, "completions/mean_terminated_length": 339.2451477050781, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.29070385452359915, "epoch": 0.004561003420752566, "frac_reward_zero_std": 0.03125, "grad_norm": 0.01328803040087223, "kl": 0.0009660195537435357, "learning_rate": 1.7045454545454545e-07, "loss": 4.828150849789381e-06, "num_tokens": 1040932.0, "reward": 1.149999976158142, "reward_std": 0.9581698775291443, "rewards/code_complexity_reward/mean": 0.47539061307907104, "rewards/code_complexity_reward/std": 0.408795028924942, "rewards/code_execution_reward/mean": 0.20703125, "rewards/code_execution_reward/std": 0.40557438135147095, "rewards/code_syntax_reward/mean": 0.29296875, "rewards/code_syntax_reward/std": 0.24652054905891418, "rewards/reasoning_present_reward_func/mean": 0.03203125298023224, "rewards/reasoning_present_reward_func/std": 0.04670529440045357, "rewards/xmlcount_reward_func/mean": 0.142578125, "rewards/xmlcount_reward_func/std": 0.20240987837314606, "step": 4, "step_time": 70.56261961907148 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 371.453125, "completions/mean_terminated_length": 340.66668701171875, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.2881651141215116, "epoch": 0.005701254275940707, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.014521763660013676, "kl": 0.0009996680328185903, "learning_rate": 2.2727272727272729e-07, "loss": 4.989269655197859e-06, "num_tokens": 1297340.0, "reward": 1.1011230945587158, "reward_std": 0.9791977405548096, "rewards/code_complexity_reward/mean": 0.4317382872104645, "rewards/code_complexity_reward/std": 0.402442991733551, "rewards/code_execution_reward/mean": 0.203125, "rewards/code_execution_reward/std": 0.4027182459831238, "rewards/code_syntax_reward/mean": 0.275390625, "rewards/code_syntax_reward/std": 0.2489505261182785, "rewards/reasoning_present_reward_func/mean": 0.03437499701976776, "rewards/reasoning_present_reward_func/std": 0.04754234105348587, "rewards/xmlcount_reward_func/mean": 0.156494140625, "rewards/xmlcount_reward_func/std": 0.2130538374185562, "step": 5, "step_time": 70.19602769613266 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.220703125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 366.447265625, "completions/mean_terminated_length": 325.2255554199219, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2781517843250185, "epoch": 0.0068415051311288486, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.014394187368452549, "kl": 0.0009583225273672724, "learning_rate": 2.840909090909091e-07, "loss": 4.780551535077393e-06, "num_tokens": 1555021.0, "reward": 1.05078125, "reward_std": 0.9463517069816589, "rewards/code_complexity_reward/mean": 0.4292968511581421, "rewards/code_complexity_reward/std": 0.40943029522895813, "rewards/code_execution_reward/mean": 0.171875, "rewards/code_execution_reward/std": 0.3776407241821289, "rewards/code_syntax_reward/mean": 0.2685546875, "rewards/code_syntax_reward/std": 0.2495543211698532, "rewards/reasoning_present_reward_func/mean": 0.03261718899011612, "rewards/reasoning_present_reward_func/std": 0.04692695289850235, "rewards/xmlcount_reward_func/mean": 0.1484375, "rewards/xmlcount_reward_func/std": 0.20675364136695862, "step": 6, "step_time": 77.54880525916815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 375.65234375, "completions/mean_terminated_length": 341.7317199707031, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.2825922288466245, "epoch": 0.00798175598631699, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.01325511746108532, "kl": 0.0009835812197707128, "learning_rate": 3.409090909090909e-07, "loss": 4.909459676127881e-06, "num_tokens": 1817323.0, "reward": 1.1300780773162842, "reward_std": 0.967909038066864, "rewards/code_complexity_reward/mean": 0.4659179449081421, "rewards/code_complexity_reward/std": 0.4039621949195862, "rewards/code_execution_reward/mean": 0.205078125, "rewards/code_execution_reward/std": 0.4041535556316376, "rewards/code_syntax_reward/mean": 0.2919921875, "rewards/code_syntax_reward/std": 0.246689110994339, "rewards/reasoning_present_reward_func/mean": 0.02988281100988388, "rewards/reasoning_present_reward_func/std": 0.045819200575351715, "rewards/xmlcount_reward_func/mean": 0.13720703125, "rewards/xmlcount_reward_func/std": 0.2032572627067566, "step": 7, "step_time": 71.14855664223433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2109375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 373.3828125, "completions/mean_terminated_length": 336.32672119140625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.2805709319654852, "epoch": 0.009122006841505131, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.013729007914662361, "kl": 0.000979447864665417, "learning_rate": 3.9772727272727276e-07, "loss": 4.8983783926814795e-06, "num_tokens": 2075835.0, "reward": 1.1439942121505737, "reward_std": 0.9428232312202454, "rewards/code_complexity_reward/mean": 0.4642578065395355, "rewards/code_complexity_reward/std": 0.4068058729171753, "rewards/code_execution_reward/mean": 0.197265625, "rewards/code_execution_reward/std": 0.3983237147331238, "rewards/code_syntax_reward/mean": 0.291015625, "rewards/code_syntax_reward/std": 0.24685366451740265, "rewards/reasoning_present_reward_func/mean": 0.03398437798023224, "rewards/reasoning_present_reward_func/std": 0.04741192236542702, "rewards/xmlcount_reward_func/mean": 0.157470703125, "rewards/xmlcount_reward_func/std": 0.20855386555194855, "step": 8, "step_time": 53.32210203167051 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.220703125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 378.18359375, "completions/mean_terminated_length": 340.28570556640625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.27819288382306695, "epoch": 0.010262257696693273, "frac_reward_zero_std": 0.0, "grad_norm": 0.014202114194631577, "kl": 0.0009611887207938707, "learning_rate": 4.5454545454545457e-07, "loss": 4.79813024867326e-06, "num_tokens": 2337349.0, "reward": 1.1053221225738525, "reward_std": 0.9561763405799866, "rewards/code_complexity_reward/mean": 0.4537109136581421, "rewards/code_complexity_reward/std": 0.40325960516929626, "rewards/code_execution_reward/mean": 0.1953125, "rewards/code_execution_reward/std": 0.3968288004398346, "rewards/code_syntax_reward/mean": 0.28515625, "rewards/code_syntax_reward/std": 0.24775780737400055, "rewards/reasoning_present_reward_func/mean": 0.03125, "rewards/reasoning_present_reward_func/std": 0.04639657214283943, "rewards/xmlcount_reward_func/mean": 0.139892578125, "rewards/xmlcount_reward_func/std": 0.2014906257390976, "step": 9, "step_time": 59.08156055957079 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 376.1328125, "completions/mean_terminated_length": 338.0899963378906, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.2870701430365443, "epoch": 0.011402508551881414, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.012802721932530403, "kl": 0.0009960319075617008, "learning_rate": 5.113636363636364e-07, "loss": 4.991947207599878e-06, "num_tokens": 2598437.0, "reward": 1.1334960460662842, "reward_std": 0.9378294348716736, "rewards/code_complexity_reward/mean": 0.4647460877895355, "rewards/code_complexity_reward/std": 0.40656590461730957, "rewards/code_execution_reward/mean": 0.1953125, "rewards/code_execution_reward/std": 0.3968288004398346, "rewards/code_syntax_reward/mean": 0.291015625, "rewards/code_syntax_reward/std": 0.24685366451740265, "rewards/reasoning_present_reward_func/mean": 0.03300781175494194, "rewards/reasoning_present_reward_func/std": 0.04707008972764015, "rewards/xmlcount_reward_func/mean": 0.1494140625, "rewards/xmlcount_reward_func/std": 0.2038094848394394, "step": 10, "step_time": 63.22964357677847 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.208984375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 373.890625, "completions/mean_terminated_length": 337.4024963378906, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.2782894466072321, "epoch": 0.012542759407069556, "frac_reward_zero_std": 0.015625, "grad_norm": 0.01448128279298544, "kl": 0.0009485021082582534, "learning_rate": 5.681818181818182e-07, "loss": 4.7879875637590885e-06, "num_tokens": 2860189.0, "reward": 1.062109351158142, "reward_std": 0.9320557117462158, "rewards/code_complexity_reward/mean": 0.44316408038139343, "rewards/code_complexity_reward/std": 0.4074431359767914, "rewards/code_execution_reward/mean": 0.16015625, "rewards/code_execution_reward/std": 0.3671095669269562, "rewards/code_syntax_reward/mean": 0.27734375, "rewards/code_syntax_reward/std": 0.24874316155910492, "rewards/reasoning_present_reward_func/mean": 0.03203125298023224, "rewards/reasoning_present_reward_func/std": 0.046705298125743866, "rewards/xmlcount_reward_func/mean": 0.1494140625, "rewards/xmlcount_reward_func/std": 0.20634421706199646, "step": 11, "step_time": 60.237005112692714 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.216796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 372.2734375, "completions/mean_terminated_length": 333.59600830078125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2803159242030233, "epoch": 0.013683010262257697, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.013282102532684803, "kl": 0.0009683300104370574, "learning_rate": 6.25e-07, "loss": 4.844958311878145e-06, "num_tokens": 3120053.0, "reward": 1.1748535633087158, "reward_std": 0.9849745631217957, "rewards/code_complexity_reward/mean": 0.4774414002895355, "rewards/code_complexity_reward/std": 0.4056401550769806, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.296875, "rewards/code_syntax_reward/std": 0.24580632150173187, "rewards/reasoning_present_reward_func/mean": 0.03164062649011612, "rewards/reasoning_present_reward_func/std": 0.04655282944440842, "rewards/xmlcount_reward_func/mean": 0.144287109375, "rewards/xmlcount_reward_func/std": 0.20503169298171997, "step": 12, "step_time": 65.72911863680929 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.224609375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 375.4296875, "completions/mean_terminated_length": 335.8690185546875, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.2778543457388878, "epoch": 0.014823261117445839, "frac_reward_zero_std": 0.0, "grad_norm": 0.013369815424084663, "kl": 0.000939276397730282, "learning_rate": 6.818181818181818e-07, "loss": 4.687346518039703e-06, "num_tokens": 3379129.0, "reward": 1.1634767055511475, "reward_std": 0.9708163142204285, "rewards/code_complexity_reward/mean": 0.46210935711860657, "rewards/code_complexity_reward/std": 0.40476593375205994, "rewards/code_execution_reward/mean": 0.21875, "rewards/code_execution_reward/std": 0.41380295157432556, "rewards/code_syntax_reward/mean": 0.2919921875, "rewards/code_syntax_reward/std": 0.246689110994339, "rewards/reasoning_present_reward_func/mean": 0.03437499701976776, "rewards/reasoning_present_reward_func/std": 0.04754234105348587, "rewards/xmlcount_reward_func/mean": 0.15625, "rewards/xmlcount_reward_func/std": 0.20719683170318604, "step": 13, "step_time": 53.16662717796862 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 375.083984375, "completions/mean_terminated_length": 340.183837890625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.27878254652023315, "epoch": 0.01596351197263398, "frac_reward_zero_std": 0.015625, "grad_norm": 0.013684618286788464, "kl": 0.0009379438324685907, "learning_rate": 7.386363636363638e-07, "loss": 4.674628144130111e-06, "num_tokens": 3637796.0, "reward": 1.1271483898162842, "reward_std": 0.9663835167884827, "rewards/code_complexity_reward/mean": 0.4532226622104645, "rewards/code_complexity_reward/std": 0.4046867787837982, "rewards/code_execution_reward/mean": 0.208984375, "rewards/code_execution_reward/std": 0.40698084235191345, "rewards/code_syntax_reward/mean": 0.2841796875, "rewards/code_syntax_reward/std": 0.24789467453956604, "rewards/reasoning_present_reward_func/mean": 0.03281250223517418, "rewards/reasoning_present_reward_func/std": 0.04699898138642311, "rewards/xmlcount_reward_func/mean": 0.14794921875, "rewards/xmlcount_reward_func/std": 0.204578697681427, "step": 14, "step_time": 60.15144179482013 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 365.126953125, "completions/mean_terminated_length": 335.4765319824219, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2894662204198539, "epoch": 0.01710376282782212, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.013838808983564377, "kl": 0.000986451634162222, "learning_rate": 7.954545454545455e-07, "loss": 4.916393663734198e-06, "num_tokens": 3891489.0, "reward": 1.1912109851837158, "reward_std": 0.970281720161438, "rewards/code_complexity_reward/mean": 0.48359373211860657, "rewards/code_complexity_reward/std": 0.40992382168769836, "rewards/code_execution_reward/mean": 0.203125, "rewards/code_execution_reward/std": 0.4027182459831238, "rewards/code_syntax_reward/mean": 0.296875, "rewards/code_syntax_reward/std": 0.24580632150173187, "rewards/reasoning_present_reward_func/mean": 0.03769531100988388, "rewards/reasoning_present_reward_func/std": 0.04850969836115837, "rewards/xmlcount_reward_func/mean": 0.169921875, "rewards/xmlcount_reward_func/std": 0.21358256042003632, "step": 15, "step_time": 68.94674453139305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.173828125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 359.015625, "completions/mean_terminated_length": 326.8274230957031, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2795557521749288, "epoch": 0.018244013683010263, "frac_reward_zero_std": 0.015625, "grad_norm": 0.014217556454241276, "kl": 0.000958997588895727, "learning_rate": 8.522727272727273e-07, "loss": 4.798639565706253e-06, "num_tokens": 4145561.0, "reward": 1.161279320716858, "reward_std": 0.9835143089294434, "rewards/code_complexity_reward/mean": 0.4756835997104645, "rewards/code_complexity_reward/std": 0.40838751196861267, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.29296875, "rewards/code_syntax_reward/std": 0.24652054905891418, "rewards/reasoning_present_reward_func/mean": 0.02910156361758709, "rewards/reasoning_present_reward_func/std": 0.045467495918273926, "rewards/xmlcount_reward_func/mean": 0.133056640625, "rewards/xmlcount_reward_func/std": 0.20338928699493408, "step": 16, "step_time": 78.94620764814317 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.220703125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 381.62109375, "completions/mean_terminated_length": 344.6967468261719, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.29028880619443953, "epoch": 0.019384264538198404, "frac_reward_zero_std": 0.015625, "grad_norm": 0.013359704986214638, "kl": 0.0009993666290029068, "learning_rate": 9.090909090909091e-07, "loss": 5.0129624469263945e-06, "num_tokens": 4408119.0, "reward": 1.105078101158142, "reward_std": 0.9533035755157471, "rewards/code_complexity_reward/mean": 0.4345703125, "rewards/code_complexity_reward/std": 0.4063057601451874, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.2734375, "rewards/code_syntax_reward/std": 0.2491423636674881, "rewards/reasoning_present_reward_func/mean": 0.03281249850988388, "rewards/reasoning_present_reward_func/std": 0.04699898138642311, "rewards/xmlcount_reward_func/mean": 0.1533203125, "rewards/xmlcount_reward_func/std": 0.2086467742919922, "step": 17, "step_time": 60.83864306751639 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.181640625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 367.611328125, "completions/mean_terminated_length": 335.5632629394531, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.28304107091389596, "epoch": 0.020524515393386546, "frac_reward_zero_std": 0.0, "grad_norm": 0.014577304013073444, "kl": 0.0009889285065582953, "learning_rate": 9.65909090909091e-07, "loss": 4.9258305807597935e-06, "num_tokens": 4665468.0, "reward": 1.176611304283142, "reward_std": 0.9571231603622437, "rewards/code_complexity_reward/mean": 0.47685545682907104, "rewards/code_complexity_reward/std": 0.40343013405799866, "rewards/code_execution_reward/mean": 0.212890625, "rewards/code_execution_reward/std": 0.409751296043396, "rewards/code_syntax_reward/mean": 0.2978515625, "rewards/code_syntax_reward/std": 0.24561770260334015, "rewards/reasoning_present_reward_func/mean": 0.03398437798023224, "rewards/reasoning_present_reward_func/std": 0.04741192236542702, "rewards/xmlcount_reward_func/mean": 0.155029296875, "rewards/xmlcount_reward_func/std": 0.20686091482639313, "step": 18, "step_time": 61.24952935799956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.212890625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 368.54296875, "completions/mean_terminated_length": 329.741943359375, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.2884885563980788, "epoch": 0.021664766248574687, "frac_reward_zero_std": 0.0, "grad_norm": 0.014295939356088638, "kl": 0.0009555920341881574, "learning_rate": 1.0227272727272729e-06, "loss": 4.756904672831297e-06, "num_tokens": 4922894.0, "reward": 1.1525390148162842, "reward_std": 1.0030766725540161, "rewards/code_complexity_reward/mean": 0.45751953125, "rewards/code_complexity_reward/std": 0.40844449400901794, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.28515625, "rewards/code_syntax_reward/std": 0.24775780737400055, "rewards/reasoning_present_reward_func/mean": 0.03144531324505806, "rewards/reasoning_present_reward_func/std": 0.04647517949342728, "rewards/xmlcount_reward_func/mean": 0.14404296875, "rewards/xmlcount_reward_func/std": 0.20121611654758453, "step": 19, "step_time": 59.42103870399296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.232421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 374.89453125, "completions/mean_terminated_length": 333.379150390625, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.29008942283689976, "epoch": 0.02280501710376283, "frac_reward_zero_std": 0.015625, "grad_norm": 0.014065018855035305, "kl": 0.0010680274544938584, "learning_rate": 1.0795454545454546e-06, "loss": 5.333286026143469e-06, "num_tokens": 5184444.0, "reward": 1.0583007335662842, "reward_std": 0.9375872611999512, "rewards/code_complexity_reward/mean": 0.4374023675918579, "rewards/code_complexity_reward/std": 0.40710723400115967, "rewards/code_execution_reward/mean": 0.1796875, "rewards/code_execution_reward/std": 0.38430243730545044, "rewards/code_syntax_reward/mean": 0.2744140625, "rewards/code_syntax_reward/std": 0.2490483820438385, "rewards/reasoning_present_reward_func/mean": 0.03007812425494194, "rewards/reasoning_present_reward_func/std": 0.045904628932476044, "rewards/xmlcount_reward_func/mean": 0.13671875, "rewards/xmlcount_reward_func/std": 0.19810594618320465, "step": 20, "step_time": 84.8450934337452 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.189453125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 369.40234375, "completions/mean_terminated_length": 336.0722961425781, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.2834486239589751, "epoch": 0.02394526795895097, "frac_reward_zero_std": 0.0, "grad_norm": 0.014015935361385345, "kl": 0.0009578018816682743, "learning_rate": 1.1363636363636364e-06, "loss": 4.809349775314331e-06, "num_tokens": 5440230.0, "reward": 1.1206055879592896, "reward_std": 0.9672399759292603, "rewards/code_complexity_reward/mean": 0.453125, "rewards/code_complexity_reward/std": 0.4106936752796173, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.2802734375, "rewards/code_syntax_reward/std": 0.24840296804904938, "rewards/reasoning_present_reward_func/mean": 0.03125, "rewards/reasoning_present_reward_func/std": 0.04639657214283943, "rewards/xmlcount_reward_func/mean": 0.13916015625, "rewards/xmlcount_reward_func/std": 0.2025272697210312, "step": 21, "step_time": 61.56270793173462 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 364.044921875, "completions/mean_terminated_length": 326.3309020996094, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.28403640422038734, "epoch": 0.02508551881413911, "frac_reward_zero_std": 0.046875, "grad_norm": 0.013677509501576424, "kl": 0.0009866129430520232, "learning_rate": 1.1931818181818183e-06, "loss": 4.92902472615242e-06, "num_tokens": 5695017.0, "reward": 1.0594239234924316, "reward_std": 0.9750999212265015, "rewards/code_complexity_reward/mean": 0.4263671934604645, "rewards/code_complexity_reward/std": 0.41218963265419006, "rewards/code_execution_reward/mean": 0.205078125, "rewards/code_execution_reward/std": 0.4041535556316376, "rewards/code_syntax_reward/mean": 0.263671875, "rewards/code_syntax_reward/std": 0.24987001717090607, "rewards/reasoning_present_reward_func/mean": 0.029296875, "rewards/reasoning_present_reward_func/std": 0.0455569326877594, "rewards/xmlcount_reward_func/mean": 0.135009765625, "rewards/xmlcount_reward_func/std": 0.19734948873519897, "step": 22, "step_time": 69.40462993457913 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.240234375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 380.26953125, "completions/mean_terminated_length": 338.6169738769531, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.281135382829234, "epoch": 0.026225769669327253, "frac_reward_zero_std": 0.0, "grad_norm": 0.014724337495863438, "kl": 0.0009815347229960025, "learning_rate": 1.25e-06, "loss": 4.924804670736194e-06, "num_tokens": 5957539.0, "reward": 1.072607398033142, "reward_std": 0.9458467960357666, "rewards/code_complexity_reward/mean": 0.44990235567092896, "rewards/code_complexity_reward/std": 0.4068096876144409, "rewards/code_execution_reward/mean": 0.16796875, "rewards/code_execution_reward/std": 0.374204158782959, "rewards/code_syntax_reward/mean": 0.28125, "rewards/code_syntax_reward/std": 0.24828176200389862, "rewards/reasoning_present_reward_func/mean": 0.03066406026482582, "rewards/reasoning_present_reward_func/std": 0.04615497961640358, "rewards/xmlcount_reward_func/mean": 0.142822265625, "rewards/xmlcount_reward_func/std": 0.2021617293357849, "step": 23, "step_time": 69.01056973356754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 362.298828125, "completions/mean_terminated_length": 330.3720397949219, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.28397320094518363, "epoch": 0.027366020524515394, "frac_reward_zero_std": 0.015625, "grad_norm": 0.01336432620882988, "kl": 0.0009717307375467499, "learning_rate": 1.3068181818181819e-06, "loss": 4.866975359618664e-06, "num_tokens": 6212664.0, "reward": 1.188574194908142, "reward_std": 0.9959145188331604, "rewards/code_complexity_reward/mean": 0.4666992425918579, "rewards/code_complexity_reward/std": 0.4087466597557068, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.2900390625, "rewards/code_syntax_reward/std": 0.24701425433158875, "rewards/reasoning_present_reward_func/mean": 0.03632812574505806, "rewards/reasoning_present_reward_func/std": 0.04814152419567108, "rewards/xmlcount_reward_func/mean": 0.1650390625, "rewards/xmlcount_reward_func/std": 0.21008896827697754, "step": 24, "step_time": 64.33458342496306 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.201171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 371.38671875, "completions/mean_terminated_length": 335.9755554199219, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.2804835834540427, "epoch": 0.028506271379703536, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.014218461699783802, "kl": 0.0009985056503865053, "learning_rate": 1.3636363636363636e-06, "loss": 5.054695066064596e-06, "num_tokens": 6472246.0, "reward": 1.1318359375, "reward_std": 0.9764833450317383, "rewards/code_complexity_reward/mean": 0.4520507752895355, "rewards/code_complexity_reward/std": 0.4095970690250397, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.28125, "rewards/code_syntax_reward/std": 0.24828176200389862, "rewards/reasoning_present_reward_func/mean": 0.03085937350988388, "rewards/reasoning_present_reward_func/std": 0.04623647779226303, "rewards/xmlcount_reward_func/mean": 0.14306640625, "rewards/xmlcount_reward_func/std": 0.2028195858001709, "step": 25, "step_time": 71.90071750059724 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.189453125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 373.931640625, "completions/mean_terminated_length": 341.6602478027344, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2799784913659096, "epoch": 0.029646522234891677, "frac_reward_zero_std": 0.03125, "grad_norm": 0.014111863449215889, "kl": 0.000998425640318601, "learning_rate": 1.4204545454545458e-06, "loss": 4.987205102224834e-06, "num_tokens": 6733167.0, "reward": 1.116845726966858, "reward_std": 0.9404235482215881, "rewards/code_complexity_reward/mean": 0.4775390326976776, "rewards/code_complexity_reward/std": 0.40814051032066345, "rewards/code_execution_reward/mean": 0.169921875, "rewards/code_execution_reward/std": 0.3759314715862274, "rewards/code_syntax_reward/mean": 0.2939453125, "rewards/code_syntax_reward/std": 0.24634800851345062, "rewards/reasoning_present_reward_func/mean": 0.03261718899011612, "rewards/reasoning_present_reward_func/std": 0.04692695289850235, "rewards/xmlcount_reward_func/mean": 0.142822265625, "rewards/xmlcount_reward_func/std": 0.2004910707473755, "step": 26, "step_time": 64.65371398627758 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.201171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 366.107421875, "completions/mean_terminated_length": 329.36676025390625, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.289378899615258, "epoch": 0.03078677309007982, "frac_reward_zero_std": 0.015625, "grad_norm": 0.01375313475728035, "kl": 0.0010233193115709582, "learning_rate": 1.4772727272727275e-06, "loss": 5.123424671182875e-06, "num_tokens": 6989038.0, "reward": 1.146484375, "reward_std": 0.9889642000198364, "rewards/code_complexity_reward/mean": 0.4609375, "rewards/code_complexity_reward/std": 0.4112977683544159, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.28515625, "rewards/code_syntax_reward/std": 0.24775780737400055, "rewards/reasoning_present_reward_func/mean": 0.033203125, "rewards/reasoning_present_reward_func/std": 0.047140274196863174, "rewards/xmlcount_reward_func/mean": 0.15234375, "rewards/xmlcount_reward_func/std": 0.20745493471622467, "step": 27, "step_time": 61.143425012007356 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 369.05078125, "completions/mean_terminated_length": 335.2125549316406, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.28816690016537905, "epoch": 0.03192702394526796, "frac_reward_zero_std": 0.0, "grad_norm": 0.014523481018841267, "kl": 0.0009744747285367339, "learning_rate": 1.5340909090909093e-06, "loss": 4.873611032962799e-06, "num_tokens": 7246160.0, "reward": 1.1372559070587158, "reward_std": 0.9989891648292542, "rewards/code_complexity_reward/mean": 0.44462889432907104, "rewards/code_complexity_reward/std": 0.41116446256637573, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.275390625, "rewards/code_syntax_reward/std": 0.2489505261182785, "rewards/reasoning_present_reward_func/mean": 0.033203125, "rewards/reasoning_present_reward_func/std": 0.047140270471572876, "rewards/xmlcount_reward_func/mean": 0.149658203125, "rewards/xmlcount_reward_func/std": 0.20712865889072418, "step": 28, "step_time": 69.89847231283784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 377.90625, "completions/mean_terminated_length": 343.7254943847656, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.28332001296803355, "epoch": 0.0330672748004561, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.013679190538823605, "kl": 0.0009913389003486373, "learning_rate": 1.590909090909091e-06, "loss": 4.942237865179777e-06, "num_tokens": 7510488.0, "reward": 1.109521508216858, "reward_std": 0.9447304606437683, "rewards/code_complexity_reward/mean": 0.46337890625, "rewards/code_complexity_reward/std": 0.4080425798892975, "rewards/code_execution_reward/mean": 0.1953125, "rewards/code_execution_reward/std": 0.3968288004398346, "rewards/code_syntax_reward/mean": 0.287109375, "rewards/code_syntax_reward/std": 0.24747224152088165, "rewards/reasoning_present_reward_func/mean": 0.02968750149011612, "rewards/reasoning_present_reward_func/std": 0.045732785016298294, "rewards/xmlcount_reward_func/mean": 0.134033203125, "rewards/xmlcount_reward_func/std": 0.19646507501602173, "step": 29, "step_time": 75.6580112921074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.205078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 367.640625, "completions/mean_terminated_length": 330.3980407714844, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.2801074022427201, "epoch": 0.03420752565564424, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.014952288009226322, "kl": 0.0009608122281861142, "learning_rate": 1.6477272727272728e-06, "loss": 4.832047125091776e-06, "num_tokens": 7767060.0, "reward": 1.172021508216858, "reward_std": 1.001180648803711, "rewards/code_complexity_reward/mean": 0.4720703363418579, "rewards/code_complexity_reward/std": 0.4092870056629181, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.29296875, "rewards/code_syntax_reward/std": 0.24652054905891418, "rewards/reasoning_present_reward_func/mean": 0.03125, "rewards/reasoning_present_reward_func/std": 0.04639657586812973, "rewards/xmlcount_reward_func/mean": 0.141357421875, "rewards/xmlcount_reward_func/std": 0.2040916085243225, "step": 30, "step_time": 62.548960007727146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 377.955078125, "completions/mean_terminated_length": 340.4224853515625, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.27586913062259555, "epoch": 0.03534777651083238, "frac_reward_zero_std": 0.03125, "grad_norm": 0.013199426233768463, "kl": 0.0009895948978737579, "learning_rate": 1.7045454545454546e-06, "loss": 4.947598426952027e-06, "num_tokens": 8028021.0, "reward": 1.129052758216858, "reward_std": 0.9908062815666199, "rewards/code_complexity_reward/mean": 0.4610351324081421, "rewards/code_complexity_reward/std": 0.40795692801475525, "rewards/code_execution_reward/mean": 0.220703125, "rewards/code_execution_reward/std": 0.4151262938976288, "rewards/code_syntax_reward/mean": 0.287109375, "rewards/code_syntax_reward/std": 0.24747224152088165, "rewards/reasoning_present_reward_func/mean": 0.02812499925494194, "rewards/reasoning_present_reward_func/std": 0.045004893094301224, "rewards/xmlcount_reward_func/mean": 0.132080078125, "rewards/xmlcount_reward_func/std": 0.19871141016483307, "step": 31, "step_time": 65.62863153591752 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.177734375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 368.046875, "completions/mean_terminated_length": 336.9311218261719, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.2856355074327439, "epoch": 0.036488027366020526, "frac_reward_zero_std": 0.0, "grad_norm": 0.013928210362792015, "kl": 0.0009481179922659067, "learning_rate": 1.7613636363636365e-06, "loss": 4.742294549942017e-06, "num_tokens": 8283653.0, "reward": 1.1265137195587158, "reward_std": 0.9859514236450195, "rewards/code_complexity_reward/mean": 0.4500976800918579, "rewards/code_complexity_reward/std": 0.4111282229423523, "rewards/code_execution_reward/mean": 0.21484375, "rewards/code_execution_reward/std": 0.4111155867576599, "rewards/code_syntax_reward/mean": 0.27734375, "rewards/code_syntax_reward/std": 0.24874316155910492, "rewards/reasoning_present_reward_func/mean": 0.03359375149011612, "rewards/reasoning_present_reward_func/std": 0.04727790877223015, "rewards/xmlcount_reward_func/mean": 0.150634765625, "rewards/xmlcount_reward_func/std": 0.20641814172267914, "step": 32, "step_time": 70.84146134834737 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.193359375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 362.89453125, "completions/mean_terminated_length": 327.1525573730469, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.2860108418390155, "epoch": 0.037628278221208664, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.015222957357764244, "kl": 0.0010168497519771336, "learning_rate": 1.8181818181818183e-06, "loss": 5.059322575107217e-06, "num_tokens": 8537903.0, "reward": 1.1905274391174316, "reward_std": 0.9839927554130554, "rewards/code_complexity_reward/mean": 0.47802734375, "rewards/code_complexity_reward/std": 0.4043194055557251, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.298828125, "rewards/code_syntax_reward/std": 0.2454250603914261, "rewards/reasoning_present_reward_func/mean": 0.03281249850988388, "rewards/reasoning_present_reward_func/std": 0.04699898138642311, "rewards/xmlcount_reward_func/mean": 0.1484375, "rewards/xmlcount_reward_func/std": 0.20287199318408966, "step": 33, "step_time": 79.2272175848484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 378.48046875, "completions/mean_terminated_length": 345.263427734375, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.27466969774104655, "epoch": 0.03876852907639681, "frac_reward_zero_std": 0.0, "grad_norm": 0.01477984618395567, "kl": 0.0009653461538619013, "learning_rate": 1.8750000000000003e-06, "loss": 4.860048647969961e-06, "num_tokens": 8800057.0, "reward": 1.169335961341858, "reward_std": 0.9263138771057129, "rewards/code_complexity_reward/mean": 0.48193359375, "rewards/code_complexity_reward/std": 0.3953382670879364, "rewards/code_execution_reward/mean": 0.20703125, "rewards/code_execution_reward/std": 0.40557438135147095, "rewards/code_syntax_reward/mean": 0.306640625, "rewards/code_syntax_reward/std": 0.24373729526996613, "rewards/reasoning_present_reward_func/mean": 0.03164062649011612, "rewards/reasoning_present_reward_func/std": 0.046552833169698715, "rewards/xmlcount_reward_func/mean": 0.14208984375, "rewards/xmlcount_reward_func/std": 0.20184673368930817, "step": 34, "step_time": 58.43496016878635 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 377.86328125, "completions/mean_terminated_length": 348.48095703125, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "entropy": 0.28262723819352686, "epoch": 0.039908779931584946, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.013748189434409142, "kl": 0.001027341820190486, "learning_rate": 1.931818181818182e-06, "loss": 5.115187377668917e-06, "num_tokens": 9061795.0, "reward": 1.079443335533142, "reward_std": 0.963082492351532, "rewards/code_complexity_reward/mean": 0.4453125, "rewards/code_complexity_reward/std": 0.4134214520454407, "rewards/code_execution_reward/mean": 0.177734375, "rewards/code_execution_reward/std": 0.3826628625392914, "rewards/code_syntax_reward/mean": 0.275390625, "rewards/code_syntax_reward/std": 0.2489505261182785, "rewards/reasoning_present_reward_func/mean": 0.03281250223517418, "rewards/reasoning_present_reward_func/std": 0.04699898138642311, "rewards/xmlcount_reward_func/mean": 0.148193359375, "rewards/xmlcount_reward_func/std": 0.20342686772346497, "step": 35, "step_time": 53.91715769004077 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 368.9609375, "completions/mean_terminated_length": 335.1014404296875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2761817085556686, "epoch": 0.04104903078677309, "frac_reward_zero_std": 0.015625, "grad_norm": 0.013064015656709671, "kl": 0.0009477249850533553, "learning_rate": 1.9886363636363638e-06, "loss": 4.789006197825074e-06, "num_tokens": 9318067.0, "reward": 1.1952147483825684, "reward_std": 0.9612942934036255, "rewards/code_complexity_reward/mean": 0.4833984375, "rewards/code_complexity_reward/std": 0.40000948309898376, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.3037109375, "rewards/code_syntax_reward/std": 0.24440090358257294, "rewards/reasoning_present_reward_func/mean": 0.03261718899011612, "rewards/reasoning_present_reward_func/std": 0.04692695289850235, "rewards/xmlcount_reward_func/mean": 0.14306640625, "rewards/xmlcount_reward_func/std": 0.2019129991531372, "step": 36, "step_time": 69.92044174857438 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 374.76953125, "completions/mean_terminated_length": 335.4623107910156, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.2815474537201226, "epoch": 0.04218928164196123, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.014988254755735397, "kl": 0.0010082272465297137, "learning_rate": 2.0454545454545457e-06, "loss": 5.050853360444307e-06, "num_tokens": 9577997.0, "reward": 1.1634764671325684, "reward_std": 0.976859986782074, "rewards/code_complexity_reward/mean": 0.4698242247104645, "rewards/code_complexity_reward/std": 0.4027699828147888, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.294921875, "rewards/code_syntax_reward/std": 0.24617145955562592, "rewards/reasoning_present_reward_func/mean": 0.03398437798023224, "rewards/reasoning_present_reward_func/std": 0.04741192236542702, "rewards/xmlcount_reward_func/mean": 0.15380859375, "rewards/xmlcount_reward_func/std": 0.20696093142032623, "step": 37, "step_time": 61.26754767727107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.185546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 367.525390625, "completions/mean_terminated_length": 334.61151123046875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.27616851101629436, "epoch": 0.043329532497149374, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.014095905236899853, "kl": 0.0009541106319375103, "learning_rate": 2.1022727272727277e-06, "loss": 4.7818757593631744e-06, "num_tokens": 9834830.0, "reward": 1.251953125, "reward_std": 0.9799649715423584, "rewards/code_complexity_reward/mean": 0.509765625, "rewards/code_complexity_reward/std": 0.4036675691604614, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.314453125, "rewards/code_syntax_reward/std": 0.2417849749326706, "rewards/reasoning_present_reward_func/mean": 0.0341796875, "rewards/reasoning_present_reward_func/std": 0.04747757688164711, "rewards/xmlcount_reward_func/mean": 0.1513671875, "rewards/xmlcount_reward_func/std": 0.20919561386108398, "step": 38, "step_time": 71.4450958892703 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.166015625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 358.515625, "completions/mean_terminated_length": 327.9625244140625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.28067226172424853, "epoch": 0.04446978335233751, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.013809632509946823, "kl": 0.0009641747647037846, "learning_rate": 2.1590909090909092e-06, "loss": 4.815345164388418e-06, "num_tokens": 10086434.0, "reward": 1.2184569835662842, "reward_std": 0.9711739420890808, "rewards/code_complexity_reward/mean": 0.4942382872104645, "rewards/code_complexity_reward/std": 0.40237918496131897, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.306640625, "rewards/code_syntax_reward/std": 0.24373729526996613, "rewards/reasoning_present_reward_func/mean": 0.03378906100988388, "rewards/reasoning_present_reward_func/std": 0.0473453663289547, "rewards/xmlcount_reward_func/mean": 0.1513671875, "rewards/xmlcount_reward_func/std": 0.2062515765428543, "step": 39, "step_time": 70.21939959283918 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 366.158203125, "completions/mean_terminated_length": 334.2119140625, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.28038592962548137, "epoch": 0.04561003420752566, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.01421199832111597, "kl": 0.0010326188239559997, "learning_rate": 2.2159090909090912e-06, "loss": 5.179375875741243e-06, "num_tokens": 10341835.0, "reward": 1.209326148033142, "reward_std": 0.9622699618339539, "rewards/code_complexity_reward/mean": 0.482421875, "rewards/code_complexity_reward/std": 0.40337809920310974, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.30078125, "rewards/code_syntax_reward/std": 0.2450276017189026, "rewards/reasoning_present_reward_func/mean": 0.03476562723517418, "rewards/reasoning_present_reward_func/std": 0.047669194638729095, "rewards/xmlcount_reward_func/mean": 0.160888671875, "rewards/xmlcount_reward_func/std": 0.2100399285554886, "step": 40, "step_time": 52.55774412397295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 359.322265625, "completions/mean_terminated_length": 328.5, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.28061610157601535, "epoch": 0.046750285062713795, "frac_reward_zero_std": 0.015625, "grad_norm": 0.013518421910703182, "kl": 0.000983570136668277, "learning_rate": 2.2727272727272728e-06, "loss": 4.935718607157469e-06, "num_tokens": 10594924.0, "reward": 1.228417992591858, "reward_std": 0.9853876233100891, "rewards/code_complexity_reward/mean": 0.48779296875, "rewards/code_complexity_reward/std": 0.40405529737472534, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.302734375, "rewards/code_syntax_reward/std": 0.2446138858795166, "rewards/reasoning_present_reward_func/mean": 0.03359375149011612, "rewards/reasoning_present_reward_func/std": 0.04727790877223015, "rewards/xmlcount_reward_func/mean": 0.154296875, "rewards/xmlcount_reward_func/std": 0.21055464446544647, "step": 41, "step_time": 84.1496867351234 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.232421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 365.806640625, "completions/mean_terminated_length": 321.5394287109375, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.2822035853751004, "epoch": 0.04789053591790194, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.017206845805048943, "kl": 0.0009675700657680864, "learning_rate": 2.3295454545454547e-06, "loss": 4.832050763070583e-06, "num_tokens": 10849857.0, "reward": 1.07080078125, "reward_std": 0.9367846846580505, "rewards/code_complexity_reward/mean": 0.43896484375, "rewards/code_complexity_reward/std": 0.4106582701206207, "rewards/code_execution_reward/mean": 0.17578125, "rewards/code_execution_reward/std": 0.3810062110424042, "rewards/code_syntax_reward/mean": 0.2724609375, "rewards/code_syntax_reward/std": 0.2492324709892273, "rewards/reasoning_present_reward_func/mean": 0.033203125, "rewards/reasoning_present_reward_func/std": 0.047140274196863174, "rewards/xmlcount_reward_func/mean": 0.150390625, "rewards/xmlcount_reward_func/std": 0.2074088603258133, "step": 42, "step_time": 65.60187099594623 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.173828125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 364.583984375, "completions/mean_terminated_length": 333.5673828125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.27988922595977783, "epoch": 0.04903078677309008, "frac_reward_zero_std": 0.0, "grad_norm": 0.014663632959127426, "kl": 0.0009591370426278445, "learning_rate": 2.3863636363636367e-06, "loss": 4.7817593440413475e-06, "num_tokens": 11105892.0, "reward": 1.090478539466858, "reward_std": 0.9524049758911133, "rewards/code_complexity_reward/mean": 0.44921875, "rewards/code_complexity_reward/std": 0.4086986780166626, "rewards/code_execution_reward/mean": 0.1796875, "rewards/code_execution_reward/std": 0.38430243730545044, "rewards/code_syntax_reward/mean": 0.2802734375, "rewards/code_syntax_reward/std": 0.24840296804904938, "rewards/reasoning_present_reward_func/mean": 0.03261718899011612, "rewards/reasoning_present_reward_func/std": 0.04692695289850235, "rewards/xmlcount_reward_func/mean": 0.148681640625, "rewards/xmlcount_reward_func/std": 0.2056134194135666, "step": 43, "step_time": 64.99101581145078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.203125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 376.35546875, "completions/mean_terminated_length": 341.7794189453125, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.27801523939706385, "epoch": 0.05017103762827822, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.014363840222358704, "kl": 0.0009413715679329471, "learning_rate": 2.4431818181818182e-06, "loss": 4.689674824476242e-06, "num_tokens": 11366926.0, "reward": 1.140380859375, "reward_std": 0.9478392601013184, "rewards/code_complexity_reward/mean": 0.45634764432907104, "rewards/code_complexity_reward/std": 0.40001681447029114, "rewards/code_execution_reward/mean": 0.20703125, "rewards/code_execution_reward/std": 0.40557438135147095, "rewards/code_syntax_reward/mean": 0.2900390625, "rewards/code_syntax_reward/std": 0.24701425433158875, "rewards/reasoning_present_reward_func/mean": 0.03339843824505806, "rewards/reasoning_present_reward_func/std": 0.047209545969963074, "rewards/xmlcount_reward_func/mean": 0.153564453125, "rewards/xmlcount_reward_func/std": 0.2058839499950409, "step": 44, "step_time": 63.53531844820827 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.24609375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 379.66796875, "completions/mean_terminated_length": 336.47149658203125, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.2909051973838359, "epoch": 0.05131128848346636, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.015234177932143211, "kl": 0.0009669850396676338, "learning_rate": 2.5e-06, "loss": 4.835892468690872e-06, "num_tokens": 11628440.0, "reward": 1.068017601966858, "reward_std": 0.9726237058639526, "rewards/code_complexity_reward/mean": 0.4276367127895355, "rewards/code_complexity_reward/std": 0.40669888257980347, "rewards/code_execution_reward/mean": 0.177734375, "rewards/code_execution_reward/std": 0.3826628625392914, "rewards/code_syntax_reward/mean": 0.26953125, "rewards/code_syntax_reward/std": 0.24947965145111084, "rewards/reasoning_present_reward_func/mean": 0.0341796875, "rewards/reasoning_present_reward_func/std": 0.04747757688164711, "rewards/xmlcount_reward_func/mean": 0.158935546875, "rewards/xmlcount_reward_func/std": 0.204916313290596, "step": 45, "step_time": 74.42429666407406 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.189453125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 365.421875, "completions/mean_terminated_length": 331.16143798828125, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.284686072031036, "epoch": 0.052451539338654506, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.014166286215186119, "kl": 0.001023317860926909, "learning_rate": 2.556818181818182e-06, "loss": 5.1130191422998905e-06, "num_tokens": 11885216.0, "reward": 1.119531273841858, "reward_std": 0.9774935245513916, "rewards/code_complexity_reward/mean": 0.44599607586860657, "rewards/code_complexity_reward/std": 0.40958258509635925, "rewards/code_execution_reward/mean": 0.220703125, "rewards/code_execution_reward/std": 0.4151262938976288, "rewards/code_syntax_reward/mean": 0.27734375, "rewards/code_syntax_reward/std": 0.24874316155910492, "rewards/reasoning_present_reward_func/mean": 0.03242187574505806, "rewards/reasoning_present_reward_func/std": 0.04685399681329727, "rewards/xmlcount_reward_func/mean": 0.14306640625, "rewards/xmlcount_reward_func/std": 0.2034217268228531, "step": 46, "step_time": 97.46727702859789 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 373.416015625, "completions/mean_terminated_length": 334.61248779296875, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.2835312730167061, "epoch": 0.053591790193842644, "frac_reward_zero_std": 0.0, "grad_norm": 0.013812029734253883, "kl": 0.0009713213567010825, "learning_rate": 2.6136363636363637e-06, "loss": 4.840083420276642e-06, "num_tokens": 12145481.0, "reward": 1.1296875476837158, "reward_std": 0.9507670402526855, "rewards/code_complexity_reward/mean": 0.4569335877895355, "rewards/code_complexity_reward/std": 0.40738746523857117, "rewards/code_execution_reward/mean": 0.20703125, "rewards/code_execution_reward/std": 0.40557438135147095, "rewards/code_syntax_reward/mean": 0.2861328125, "rewards/code_syntax_reward/std": 0.2476169914007187, "rewards/reasoning_present_reward_func/mean": 0.03164062649011612, "rewards/reasoning_present_reward_func/std": 0.046552833169698715, "rewards/xmlcount_reward_func/mean": 0.14794921875, "rewards/xmlcount_reward_func/std": 0.2063644677400589, "step": 47, "step_time": 65.39909795019776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 376.80859375, "completions/mean_terminated_length": 338.9549865722656, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.2824942769948393, "epoch": 0.05473204104903079, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.014452755451202393, "kl": 0.0013801649865854415, "learning_rate": 2.6704545454545457e-06, "loss": 6.922404281795025e-06, "num_tokens": 12408031.0, "reward": 1.104638695716858, "reward_std": 0.9380332231521606, "rewards/code_complexity_reward/mean": 0.4613281190395355, "rewards/code_complexity_reward/std": 0.40325072407722473, "rewards/code_execution_reward/mean": 0.189453125, "rewards/code_execution_reward/std": 0.3922513723373413, "rewards/code_syntax_reward/mean": 0.2900390625, "rewards/code_syntax_reward/std": 0.24701425433158875, "rewards/reasoning_present_reward_func/mean": 0.029296875, "rewards/reasoning_present_reward_func/std": 0.0455569326877594, "rewards/xmlcount_reward_func/mean": 0.134521484375, "rewards/xmlcount_reward_func/std": 0.2015131562948227, "step": 48, "step_time": 60.34925615694374 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 357.4765625, "completions/mean_terminated_length": 323.6285705566406, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.2877321292180568, "epoch": 0.055872291904218926, "frac_reward_zero_std": 0.015625, "grad_norm": 0.016017019748687744, "kl": 0.0009878035898509552, "learning_rate": 2.7272727272727272e-06, "loss": 4.961679223924875e-06, "num_tokens": 12659259.0, "reward": 1.207763671875, "reward_std": 0.9581505060195923, "rewards/code_complexity_reward/mean": 0.49794918298721313, "rewards/code_complexity_reward/std": 0.4111112058162689, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.3037109375, "rewards/code_syntax_reward/std": 0.24440090358257294, "rewards/reasoning_present_reward_func/mean": 0.03574218600988388, "rewards/reasoning_present_reward_func/std": 0.04797092452645302, "rewards/xmlcount_reward_func/mean": 0.159423828125, "rewards/xmlcount_reward_func/std": 0.2104308009147644, "step": 49, "step_time": 64.13916902057827 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.23828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 373.306640625, "completions/mean_terminated_length": 329.9205322265625, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.2750046057626605, "epoch": 0.05701254275940707, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.015110653825104237, "kl": 0.000940132748837641, "learning_rate": 2.784090909090909e-06, "loss": 4.684668965637684e-06, "num_tokens": 12916928.0, "reward": 1.1237304210662842, "reward_std": 0.98036789894104, "rewards/code_complexity_reward/mean": 0.45283204317092896, "rewards/code_complexity_reward/std": 0.4097357392311096, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.2822265625, "rewards/code_syntax_reward/std": 0.24815665185451508, "rewards/reasoning_present_reward_func/mean": 0.03125, "rewards/reasoning_present_reward_func/std": 0.04639657214283943, "rewards/xmlcount_reward_func/mean": 0.146484375, "rewards/xmlcount_reward_func/std": 0.2074088603258133, "step": 50, "step_time": 61.240202845074236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 360.748046875, "completions/mean_terminated_length": 329.35614013671875, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.2833573401439935, "epoch": 0.05815279361459521, "frac_reward_zero_std": 0.0, "grad_norm": 0.014915142208337784, "kl": 0.001023878869091277, "learning_rate": 2.8409090909090916e-06, "loss": 5.108420737087727e-06, "num_tokens": 13170559.0, "reward": 1.23974609375, "reward_std": 0.9595971703529358, "rewards/code_complexity_reward/mean": 0.5035156011581421, "rewards/code_complexity_reward/std": 0.4082171320915222, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.3076171875, "rewards/code_syntax_reward/std": 0.24350784718990326, "rewards/reasoning_present_reward_func/mean": 0.03457031399011612, "rewards/reasoning_present_reward_func/std": 0.047606211155653, "rewards/xmlcount_reward_func/mean": 0.15966796875, "rewards/xmlcount_reward_func/std": 0.2094438374042511, "step": 51, "step_time": 62.846952254883945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 355.556640625, "completions/mean_terminated_length": 329.9568176269531, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.2786765194032341, "epoch": 0.059293044469783354, "frac_reward_zero_std": 0.0, "grad_norm": 0.01618802733719349, "kl": 0.0009953689914254937, "learning_rate": 2.897727272727273e-06, "loss": 4.950561560690403e-06, "num_tokens": 13421600.0, "reward": 1.3297852277755737, "reward_std": 1.006546139717102, "rewards/code_complexity_reward/mean": 0.519824206829071, "rewards/code_complexity_reward/std": 0.40482965111732483, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.31640625, "rewards/code_syntax_reward/std": 0.24125482141971588, "rewards/reasoning_present_reward_func/mean": 0.03457031399011612, "rewards/reasoning_present_reward_func/std": 0.0476062074303627, "rewards/xmlcount_reward_func/mean": 0.158203125, "rewards/xmlcount_reward_func/std": 0.20792421698570251, "step": 52, "step_time": 76.01567516569048 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.201171875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 357.609375, "completions/mean_terminated_length": 318.7286071777344, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2729928463231772, "epoch": 0.06043329532497149, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.014687403105199337, "kl": 0.0009446707708775648, "learning_rate": 2.954545454545455e-06, "loss": 4.73059481009841e-06, "num_tokens": 13673876.0, "reward": 1.303466796875, "reward_std": 0.9945294857025146, "rewards/code_complexity_reward/mean": 0.515917956829071, "rewards/code_complexity_reward/std": 0.40212926268577576, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.3173828125, "rewards/code_syntax_reward/std": 0.24098336696624756, "rewards/reasoning_present_reward_func/mean": 0.03535156324505806, "rewards/reasoning_present_reward_func/std": 0.047852855175733566, "rewards/xmlcount_reward_func/mean": 0.163330078125, "rewards/xmlcount_reward_func/std": 0.21465234458446503, "step": 53, "step_time": 62.168058067560196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.169921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 364.744140625, "completions/mean_terminated_length": 334.5999755859375, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.2870727239642292, "epoch": 0.06157354618015964, "frac_reward_zero_std": 0.015625, "grad_norm": 0.014258396811783314, "kl": 0.000999507419692236, "learning_rate": 3.0113636363636366e-06, "loss": 5.007044819649309e-06, "num_tokens": 13928317.0, "reward": 1.2185547351837158, "reward_std": 0.9847043752670288, "rewards/code_complexity_reward/mean": 0.47880858182907104, "rewards/code_complexity_reward/std": 0.403719425201416, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.2978515625, "rewards/code_syntax_reward/std": 0.24561770260334015, "rewards/reasoning_present_reward_func/mean": 0.03515625, "rewards/reasoning_present_reward_func/std": 0.04779251292347908, "rewards/xmlcount_reward_func/mean": 0.16064453125, "rewards/xmlcount_reward_func/std": 0.2075188308954239, "step": 54, "step_time": 64.79033876303583 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 366.50390625, "completions/mean_terminated_length": 334.6333312988281, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.27985313907265663, "epoch": 0.06271379703534778, "frac_reward_zero_std": 0.015625, "grad_norm": 0.01661447249352932, "kl": 0.0009907272096825182, "learning_rate": 3.0681818181818186e-06, "loss": 4.935020115226507e-06, "num_tokens": 14183567.0, "reward": 1.126953125, "reward_std": 0.990498423576355, "rewards/code_complexity_reward/mean": 0.4569336175918579, "rewards/code_complexity_reward/std": 0.4129813611507416, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.28125, "rewards/code_syntax_reward/std": 0.24828176200389862, "rewards/reasoning_present_reward_func/mean": 0.03183593600988388, "rewards/reasoning_present_reward_func/std": 0.04662953317165375, "rewards/xmlcount_reward_func/mean": 0.14599609375, "rewards/xmlcount_reward_func/std": 0.20672070980072021, "step": 55, "step_time": 56.96609140653163 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 360.4296875, "completions/mean_terminated_length": 327.22857666015625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.28779523982666433, "epoch": 0.06385404789053592, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.014352328144013882, "kl": 0.000986859302429366, "learning_rate": 3.125e-06, "loss": 4.931556759402156e-06, "num_tokens": 14434507.0, "reward": 1.1491210460662842, "reward_std": 1.0000741481781006, "rewards/code_complexity_reward/mean": 0.4473632574081421, "rewards/code_complexity_reward/std": 0.4127824008464813, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.275390625, "rewards/code_syntax_reward/std": 0.2489505261182785, "rewards/reasoning_present_reward_func/mean": 0.03378906100988388, "rewards/reasoning_present_reward_func/std": 0.0473453663289547, "rewards/xmlcount_reward_func/mean": 0.154296875, "rewards/xmlcount_reward_func/std": 0.20585507154464722, "step": 56, "step_time": 60.35307946521789 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.21484375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 373.625, "completions/mean_terminated_length": 335.7611999511719, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.281193706439808, "epoch": 0.06499429874572406, "frac_reward_zero_std": 0.015625, "grad_norm": 0.015468799509108067, "kl": 0.0010013665323640453, "learning_rate": 3.181818181818182e-06, "loss": 4.974863259121776e-06, "num_tokens": 14694303.0, "reward": 1.1971192359924316, "reward_std": 0.9771347045898438, "rewards/code_complexity_reward/mean": 0.47587889432907104, "rewards/code_complexity_reward/std": 0.40465638041496277, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.296875, "rewards/code_syntax_reward/std": 0.24580632150173187, "rewards/reasoning_present_reward_func/mean": 0.03593750298023224, "rewards/reasoning_present_reward_func/std": 0.048028651624917984, "rewards/xmlcount_reward_func/mean": 0.163818359375, "rewards/xmlcount_reward_func/std": 0.2139936238527298, "step": 57, "step_time": 72.51619333587587 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.232421875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 384.4375, "completions/mean_terminated_length": 345.81170654296875, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.27765323664061725, "epoch": 0.0661345496009122, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.014310355298221111, "kl": 0.0009935715388564859, "learning_rate": 3.2386363636363637e-06, "loss": 4.978093784302473e-06, "num_tokens": 14958503.0, "reward": 1.1775879859924316, "reward_std": 0.9784377217292786, "rewards/code_complexity_reward/mean": 0.4710937738418579, "rewards/code_complexity_reward/std": 0.4122563898563385, "rewards/code_execution_reward/mean": 0.212890625, "rewards/code_execution_reward/std": 0.409751296043396, "rewards/code_syntax_reward/mean": 0.2900390625, "rewards/code_syntax_reward/std": 0.24701425433158875, "rewards/reasoning_present_reward_func/mean": 0.03730468824505806, "rewards/reasoning_present_reward_func/std": 0.048408739268779755, "rewards/xmlcount_reward_func/mean": 0.166259765625, "rewards/xmlcount_reward_func/std": 0.20919546484947205, "step": 58, "step_time": 58.27725547738373 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 358.2109375, "completions/mean_terminated_length": 333.0454406738281, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2801101829390973, "epoch": 0.06727480045610035, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.0160534605383873, "kl": 0.0009846615266724257, "learning_rate": 3.2954545454545456e-06, "loss": 4.9390800995752215e-06, "num_tokens": 15210243.0, "reward": 1.2475097179412842, "reward_std": 0.9617679119110107, "rewards/code_complexity_reward/mean": 0.508496105670929, "rewards/code_complexity_reward/std": 0.4017830491065979, "rewards/code_execution_reward/mean": 0.220703125, "rewards/code_execution_reward/std": 0.4151262938976288, "rewards/code_syntax_reward/mean": 0.3134765625, "rewards/code_syntax_reward/std": 0.24204368889331818, "rewards/reasoning_present_reward_func/mean": 0.037109375, "rewards/reasoning_present_reward_func/std": 0.04835699871182442, "rewards/xmlcount_reward_func/mean": 0.167724609375, "rewards/xmlcount_reward_func/std": 0.21424798667430878, "step": 59, "step_time": 62.41233399417251 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 362.291015625, "completions/mean_terminated_length": 331.2193298339844, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.28345848876051605, "epoch": 0.06841505131128849, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.01615319773554802, "kl": 0.0009591266407369403, "learning_rate": 3.352272727272727e-06, "loss": 4.7908997657941654e-06, "num_tokens": 15463956.0, "reward": 1.23828125, "reward_std": 0.998988926410675, "rewards/code_complexity_reward/mean": 0.4786132872104645, "rewards/code_complexity_reward/std": 0.41063007712364197, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.2939453125, "rewards/code_syntax_reward/std": 0.24634800851345062, "rewards/reasoning_present_reward_func/mean": 0.03945312649011612, "rewards/reasoning_present_reward_func/std": 0.04892278090119362, "rewards/xmlcount_reward_func/mean": 0.17822265625, "rewards/xmlcount_reward_func/std": 0.2155282199382782, "step": 60, "step_time": 69.94953529257327 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 371.2421875, "completions/mean_terminated_length": 337.9226989746094, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.28552410565316677, "epoch": 0.06955530216647662, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.016158215701580048, "kl": 0.0009610243250790518, "learning_rate": 3.409090909090909e-06, "loss": 4.799105226993561e-06, "num_tokens": 15722480.0, "reward": 1.158837914466858, "reward_std": 0.9360228180885315, "rewards/code_complexity_reward/mean": 0.4769531488418579, "rewards/code_complexity_reward/std": 0.40296855568885803, "rewards/code_execution_reward/mean": 0.181640625, "rewards/code_execution_reward/std": 0.38592514395713806, "rewards/code_syntax_reward/mean": 0.296875, "rewards/code_syntax_reward/std": 0.24580632150173187, "rewards/reasoning_present_reward_func/mean": 0.037109375, "rewards/reasoning_present_reward_func/std": 0.04835699871182442, "rewards/xmlcount_reward_func/mean": 0.166259765625, "rewards/xmlcount_reward_func/std": 0.21425020694732666, "step": 61, "step_time": 61.81370337307453 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.169921875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 366.26171875, "completions/mean_terminated_length": 336.42822265625, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.2882582324091345, "epoch": 0.07069555302166476, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.0157513078302145, "kl": 0.0009997224178732722, "learning_rate": 3.4659090909090915e-06, "loss": 5.001522367820144e-06, "num_tokens": 15979662.0, "reward": 1.1808106899261475, "reward_std": 0.9473565816879272, "rewards/code_complexity_reward/mean": 0.482421875, "rewards/code_complexity_reward/std": 0.40048113465309143, "rewards/code_execution_reward/mean": 0.205078125, "rewards/code_execution_reward/std": 0.4041535556316376, "rewards/code_syntax_reward/mean": 0.3017578125, "rewards/code_syntax_reward/std": 0.24482278525829315, "rewards/reasoning_present_reward_func/mean": 0.03359375149011612, "rewards/reasoning_present_reward_func/std": 0.04727790877223015, "rewards/xmlcount_reward_func/mean": 0.157958984375, "rewards/xmlcount_reward_func/std": 0.21326228976249695, "step": 62, "step_time": 85.1151889488101 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.173828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 360.509765625, "completions/mean_terminated_length": 328.63592529296875, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.2850562618114054, "epoch": 0.07183580387685291, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.015727490186691284, "kl": 0.0009967936757675488, "learning_rate": 3.522727272727273e-06, "loss": 4.9617665354162455e-06, "num_tokens": 16233147.0, "reward": 1.1357421875, "reward_std": 0.9666398763656616, "rewards/code_complexity_reward/mean": 0.4498046636581421, "rewards/code_complexity_reward/std": 0.4073925018310547, "rewards/code_execution_reward/mean": 0.197265625, "rewards/code_execution_reward/std": 0.3983237147331238, "rewards/code_syntax_reward/mean": 0.28125, "rewards/code_syntax_reward/std": 0.24828176200389862, "rewards/reasoning_present_reward_func/mean": 0.03750000149011612, "rewards/reasoning_present_reward_func/std": 0.04845963791012764, "rewards/xmlcount_reward_func/mean": 0.169921875, "rewards/xmlcount_reward_func/std": 0.2121460884809494, "step": 63, "step_time": 71.38107294403017 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.234375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 378.453125, "completions/mean_terminated_length": 337.5714111328125, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.28826083964668214, "epoch": 0.07297605473204105, "frac_reward_zero_std": 0.015625, "grad_norm": 0.014837370254099369, "kl": 0.0010065444021165604, "learning_rate": 3.579545454545455e-06, "loss": 5.0591479521244764e-06, "num_tokens": 16497331.0, "reward": 1.080322265625, "reward_std": 0.9315490126609802, "rewards/code_complexity_reward/mean": 0.44501951336860657, "rewards/code_complexity_reward/std": 0.400439590215683, "rewards/code_execution_reward/mean": 0.158203125, "rewards/code_execution_reward/std": 0.36528825759887695, "rewards/code_syntax_reward/mean": 0.2861328125, "rewards/code_syntax_reward/std": 0.2476169914007187, "rewards/reasoning_present_reward_func/mean": 0.03496094048023224, "rewards/reasoning_present_reward_func/std": 0.04773129150271416, "rewards/xmlcount_reward_func/mean": 0.156005859375, "rewards/xmlcount_reward_func/std": 0.20833726227283478, "step": 64, "step_time": 93.39880618173629 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 356.41015625, "completions/mean_terminated_length": 320.50482177734375, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.2884065436664969, "epoch": 0.07411630558722919, "frac_reward_zero_std": 0.0, "grad_norm": 0.015228786505758762, "kl": 0.0010305400655852281, "learning_rate": 3.6363636363636366e-06, "loss": 5.162553861737251e-06, "num_tokens": 16749613.0, "reward": 1.242040991783142, "reward_std": 0.982266366481781, "rewards/code_complexity_reward/mean": 0.49345701932907104, "rewards/code_complexity_reward/std": 0.4119914472103119, "rewards/code_execution_reward/mean": 0.216796875, "rewards/code_execution_reward/std": 0.4124660789966583, "rewards/code_syntax_reward/mean": 0.2998046875, "rewards/code_syntax_reward/std": 0.2452283650636673, "rewards/reasoning_present_reward_func/mean": 0.04179687425494194, "rewards/reasoning_present_reward_func/std": 0.04937073215842247, "rewards/xmlcount_reward_func/mean": 0.190185546875, "rewards/xmlcount_reward_func/std": 0.21571142971515656, "step": 65, "step_time": 86.24972351361066 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 363.646484375, "completions/mean_terminated_length": 327.63836669921875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2756996266543865, "epoch": 0.07525655644241733, "frac_reward_zero_std": 0.0, "grad_norm": 0.016822809353470802, "kl": 0.0010261716652166797, "learning_rate": 3.6931818181818186e-06, "loss": 5.135720130056143e-06, "num_tokens": 17004920.0, "reward": 1.1920897960662842, "reward_std": 0.9488024115562439, "rewards/code_complexity_reward/mean": 0.48798829317092896, "rewards/code_complexity_reward/std": 0.4028606712818146, "rewards/code_execution_reward/mean": 0.208984375, "rewards/code_execution_reward/std": 0.40698084235191345, "rewards/code_syntax_reward/mean": 0.3037109375, "rewards/code_syntax_reward/std": 0.24440090358257294, "rewards/reasoning_present_reward_func/mean": 0.03515625, "rewards/reasoning_present_reward_func/std": 0.04779251292347908, "rewards/xmlcount_reward_func/mean": 0.15625, "rewards/xmlcount_reward_func/std": 0.20954474806785583, "step": 66, "step_time": 70.16751716565341 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.181640625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 357.9140625, "completions/mean_terminated_length": 323.713623046875, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.2957018355373293, "epoch": 0.07639680729760548, "frac_reward_zero_std": 0.015625, "grad_norm": 0.016034342348575592, "kl": 0.0010463861435709987, "learning_rate": 3.7500000000000005e-06, "loss": 5.2028626669198275e-06, "num_tokens": 17256648.0, "reward": 1.1475098133087158, "reward_std": 0.9655159711837769, "rewards/code_complexity_reward/mean": 0.4522460997104645, "rewards/code_complexity_reward/std": 0.41235780715942383, "rewards/code_execution_reward/mean": 0.201171875, "rewards/code_execution_reward/std": 0.4012683033943176, "rewards/code_syntax_reward/mean": 0.279296875, "rewards/code_syntax_reward/std": 0.24852026998996735, "rewards/reasoning_present_reward_func/mean": 0.03925781697034836, "rewards/reasoning_present_reward_func/std": 0.04888017848134041, "rewards/xmlcount_reward_func/mean": 0.175537109375, "rewards/xmlcount_reward_func/std": 0.21065886318683624, "step": 67, "step_time": 64.76391999050975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.208984375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 369.3984375, "completions/mean_terminated_length": 331.7234802246094, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.2873107688501477, "epoch": 0.07753705815279362, "frac_reward_zero_std": 0.015625, "grad_norm": 0.014966202899813652, "kl": 0.0009916756789607462, "learning_rate": 3.806818181818182e-06, "loss": 4.95504355058074e-06, "num_tokens": 17515784.0, "reward": 1.1913573741912842, "reward_std": 0.9973949790000916, "rewards/code_complexity_reward/mean": 0.46259766817092896, "rewards/code_complexity_reward/std": 0.40800753235816956, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.2880859375, "rewards/code_syntax_reward/std": 0.24732354283332825, "rewards/reasoning_present_reward_func/mean": 0.037109375, "rewards/reasoning_present_reward_func/std": 0.04835699871182442, "rewards/xmlcount_reward_func/mean": 0.173095703125, "rewards/xmlcount_reward_func/std": 0.21339108049869537, "step": 68, "step_time": 70.04756858199835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 377.373046875, "completions/mean_terminated_length": 345.50482177734375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.27495517767965794, "epoch": 0.07867730900798175, "frac_reward_zero_std": 0.015625, "grad_norm": 0.015875844284892082, "kl": 0.0009901858211378567, "learning_rate": 3.863636363636364e-06, "loss": 4.916801117360592e-06, "num_tokens": 17776627.0, "reward": 1.2225587368011475, "reward_std": 1.0065059661865234, "rewards/code_complexity_reward/mean": 0.48095703125, "rewards/code_complexity_reward/std": 0.40782901644706726, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.296875, "rewards/code_syntax_reward/std": 0.24580632150173187, "rewards/reasoning_present_reward_func/mean": 0.03652343899011612, "rewards/reasoning_present_reward_func/std": 0.04819667339324951, "rewards/xmlcount_reward_func/mean": 0.162109375, "rewards/xmlcount_reward_func/std": 0.20755623281002045, "step": 69, "step_time": 61.17579519562423 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 368.921875, "completions/mean_terminated_length": 334.1941833496094, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.2903327760286629, "epoch": 0.07981755986316989, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.015484836883842945, "kl": 0.0010266920735375606, "learning_rate": 3.9204545454545456e-06, "loss": 5.129229975864291e-06, "num_tokens": 18035075.0, "reward": 1.20556640625, "reward_std": 0.937781035900116, "rewards/code_complexity_reward/mean": 0.5033203363418579, "rewards/code_complexity_reward/std": 0.3981718420982361, "rewards/code_execution_reward/mean": 0.1953125, "rewards/code_execution_reward/std": 0.3968288004398346, "rewards/code_syntax_reward/mean": 0.3154296875, "rewards/code_syntax_reward/std": 0.24152201414108276, "rewards/reasoning_present_reward_func/mean": 0.03281249850988388, "rewards/reasoning_present_reward_func/std": 0.04699898138642311, "rewards/xmlcount_reward_func/mean": 0.15869140625, "rewards/xmlcount_reward_func/std": 0.20696093142032623, "step": 70, "step_time": 62.38496912177652 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 359.5859375, "completions/mean_terminated_length": 328.81689453125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2818322095554322, "epoch": 0.08095781071835804, "frac_reward_zero_std": 0.0, "grad_norm": 0.01642449013888836, "kl": 0.001059103893567226, "learning_rate": 3.9772727272727275e-06, "loss": 5.280773621052504e-06, "num_tokens": 18287907.0, "reward": 1.2520995140075684, "reward_std": 0.959206759929657, "rewards/code_complexity_reward/mean": 0.505175769329071, "rewards/code_complexity_reward/std": 0.4003148674964905, "rewards/code_execution_reward/mean": 0.23046875, "rewards/code_execution_reward/std": 0.42154473066329956, "rewards/code_syntax_reward/mean": 0.314453125, "rewards/code_syntax_reward/std": 0.2417849749326706, "rewards/reasoning_present_reward_func/mean": 0.03574218600988388, "rewards/reasoning_present_reward_func/std": 0.04797092452645302, "rewards/xmlcount_reward_func/mean": 0.166259765625, "rewards/xmlcount_reward_func/std": 0.2096334993839264, "step": 71, "step_time": 55.76003402937204 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20703125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 369.146484375, "completions/mean_terminated_length": 331.8497619628906, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.28052554256282747, "epoch": 0.08209806157354618, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.015851378440856934, "kl": 0.00099183809288661, "learning_rate": 4.0340909090909095e-06, "loss": 4.996167263016105e-06, "num_tokens": 18543906.0, "reward": 1.251708984375, "reward_std": 0.99410480260849, "rewards/code_complexity_reward/mean": 0.4881836175918579, "rewards/code_complexity_reward/std": 0.40395793318748474, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.3037109375, "rewards/code_syntax_reward/std": 0.24440090358257294, "rewards/reasoning_present_reward_func/mean": 0.03769531473517418, "rewards/reasoning_present_reward_func/std": 0.04850969836115837, "rewards/xmlcount_reward_func/mean": 0.166259765625, "rewards/xmlcount_reward_func/std": 0.20816978812217712, "step": 72, "step_time": 68.80537414457649 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 358.310546875, "completions/mean_terminated_length": 332.3447265625, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.27390758180990815, "epoch": 0.08323831242873432, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.015863608568906784, "kl": 0.0010362500224800897, "learning_rate": 4.0909090909090915e-06, "loss": 5.181500455364585e-06, "num_tokens": 18794389.0, "reward": 1.265722632408142, "reward_std": 0.9620524644851685, "rewards/code_complexity_reward/mean": 0.505175769329071, "rewards/code_complexity_reward/std": 0.40529465675354004, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.310546875, "rewards/code_syntax_reward/std": 0.24279458820819855, "rewards/reasoning_present_reward_func/mean": 0.03789062425494194, "rewards/reasoning_present_reward_func/std": 0.04855892062187195, "rewards/xmlcount_reward_func/mean": 0.16796875, "rewards/xmlcount_reward_func/std": 0.2103821337223053, "step": 73, "step_time": 66.68856343533844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19921875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 361.751953125, "completions/mean_terminated_length": 324.3731689453125, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.27169977175071836, "epoch": 0.08437856328392246, "frac_reward_zero_std": 0.015625, "grad_norm": 0.01592276059091091, "kl": 0.0010253260397803388, "learning_rate": 4.1477272727272734e-06, "loss": 5.118199624121189e-06, "num_tokens": 19046618.0, "reward": 1.2314453125, "reward_std": 0.9653854370117188, "rewards/code_complexity_reward/mean": 0.5, "rewards/code_complexity_reward/std": 0.40207386016845703, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.3095703125, "rewards/code_syntax_reward/std": 0.24303650856018066, "rewards/reasoning_present_reward_func/mean": 0.0322265625, "rewards/reasoning_present_reward_func/std": 0.046780116856098175, "rewards/xmlcount_reward_func/mean": 0.1494140625, "rewards/xmlcount_reward_func/std": 0.20693610608577728, "step": 74, "step_time": 70.05793157499284 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.22265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 383.24609375, "completions/mean_terminated_length": 346.3668212890625, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.2776493956334889, "epoch": 0.08551881413911061, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.01711234636604786, "kl": 0.0011006820896000136, "learning_rate": 4.204545454545455e-06, "loss": 5.509064067155123e-06, "num_tokens": 19312380.0, "reward": 1.167822241783142, "reward_std": 0.9723808169364929, "rewards/code_complexity_reward/mean": 0.4754882752895355, "rewards/code_complexity_reward/std": 0.4006962478160858, "rewards/code_execution_reward/mean": 0.20703125, "rewards/code_execution_reward/std": 0.40557438135147095, "rewards/code_syntax_reward/mean": 0.2998046875, "rewards/code_syntax_reward/std": 0.2452283650636673, "rewards/reasoning_present_reward_func/mean": 0.03242187574505806, "rewards/reasoning_present_reward_func/std": 0.04685399681329727, "rewards/xmlcount_reward_func/mean": 0.153076171875, "rewards/xmlcount_reward_func/std": 0.2087530791759491, "step": 75, "step_time": 66.9906646842137 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.212890625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 365.55078125, "completions/mean_terminated_length": 325.9404296875, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.2882778476923704, "epoch": 0.08665906499429875, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.016617856919765472, "kl": 0.0010415409115012153, "learning_rate": 4.2613636363636365e-06, "loss": 5.2041723392903805e-06, "num_tokens": 19566426.0, "reward": 1.2540526390075684, "reward_std": 0.9877154231071472, "rewards/code_complexity_reward/mean": 0.49853515625, "rewards/code_complexity_reward/std": 0.40522855520248413, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.306640625, "rewards/code_syntax_reward/std": 0.24373729526996613, "rewards/reasoning_present_reward_func/mean": 0.03750000149011612, "rewards/reasoning_present_reward_func/std": 0.04845963791012764, "rewards/xmlcount_reward_func/mean": 0.169189453125, "rewards/xmlcount_reward_func/std": 0.21265992522239685, "step": 76, "step_time": 59.77638092637062 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.173828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 352.904296875, "completions/mean_terminated_length": 319.4302673339844, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.2831700989045203, "epoch": 0.08779931584948689, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.01750069670379162, "kl": 0.0010438848439662252, "learning_rate": 4.3181818181818185e-06, "loss": 5.216366844251752e-06, "num_tokens": 19814349.0, "reward": 1.311132788658142, "reward_std": 0.9941515326499939, "rewards/code_complexity_reward/mean": 0.514453113079071, "rewards/code_complexity_reward/std": 0.40630561113357544, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.314453125, "rewards/code_syntax_reward/std": 0.2417849749326706, "rewards/reasoning_present_reward_func/mean": 0.04179687425494194, "rewards/reasoning_present_reward_func/std": 0.04937073215842247, "rewards/xmlcount_reward_func/mean": 0.1923828125, "rewards/xmlcount_reward_func/std": 0.22071774303913116, "step": 77, "step_time": 78.4742631604895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 359.7265625, "completions/mean_terminated_length": 323.68115234375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2898839295376092, "epoch": 0.08893956670467502, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.016802184283733368, "kl": 0.001100616937947052, "learning_rate": 4.3750000000000005e-06, "loss": 5.523426807485521e-06, "num_tokens": 20066529.0, "reward": 1.2547850608825684, "reward_std": 0.931354284286499, "rewards/code_complexity_reward/mean": 0.504687488079071, "rewards/code_complexity_reward/std": 0.39958086609840393, "rewards/code_execution_reward/mean": 0.203125, "rewards/code_execution_reward/std": 0.4027182459831238, "rewards/code_syntax_reward/mean": 0.314453125, "rewards/code_syntax_reward/std": 0.2417849749326706, "rewards/reasoning_present_reward_func/mean": 0.04062499850988388, "rewards/reasoning_present_reward_func/std": 0.049161266535520554, "rewards/xmlcount_reward_func/mean": 0.19189453125, "rewards/xmlcount_reward_func/std": 0.21667344868183136, "step": 78, "step_time": 74.27848668955266 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 357.71484375, "completions/mean_terminated_length": 322.110595703125, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.2840535878203809, "epoch": 0.09007981755986318, "frac_reward_zero_std": 0.0, "grad_norm": 0.02005339413881302, "kl": 0.0011343783580741729, "learning_rate": 4.4318181818181824e-06, "loss": 5.6470562412869185e-06, "num_tokens": 20319735.0, "reward": 1.3021972179412842, "reward_std": 0.9754703640937805, "rewards/code_complexity_reward/mean": 0.513671875, "rewards/code_complexity_reward/std": 0.39764299988746643, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.3193359375, "rewards/code_syntax_reward/std": 0.24042759835720062, "rewards/reasoning_present_reward_func/mean": 0.04218750074505806, "rewards/reasoning_present_reward_func/std": 0.049434177577495575, "rewards/xmlcount_reward_func/mean": 0.192626953125, "rewards/xmlcount_reward_func/std": 0.21679840981960297, "step": 79, "step_time": 62.62821809761226 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 355.68359375, "completions/mean_terminated_length": 323.2405700683594, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.2759528742171824, "epoch": 0.09122006841505131, "frac_reward_zero_std": 0.0, "grad_norm": 0.01737358421087265, "kl": 0.0010695240662244032, "learning_rate": 4.4886363636363636e-06, "loss": 5.361143848858774e-06, "num_tokens": 20570205.0, "reward": 1.3369629383087158, "reward_std": 0.9891224503517151, "rewards/code_complexity_reward/mean": 0.5247070789337158, "rewards/code_complexity_reward/std": 0.39679479598999023, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.3271484375, "rewards/code_syntax_reward/std": 0.2380310446023941, "rewards/reasoning_present_reward_func/mean": 0.041015625, "rewards/reasoning_present_reward_func/std": 0.0492342934012413, "rewards/xmlcount_reward_func/mean": 0.188232421875, "rewards/xmlcount_reward_func/std": 0.21473246812820435, "step": 80, "step_time": 60.800372837111354 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 364.45703125, "completions/mean_terminated_length": 332.13812255859375, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.2790476833470166, "epoch": 0.09236031927023945, "frac_reward_zero_std": 0.015625, "grad_norm": 0.017209086567163467, "kl": 0.0010919747237494448, "learning_rate": 4.5454545454545455e-06, "loss": 5.436333594843745e-06, "num_tokens": 20826203.0, "reward": 1.283203125, "reward_std": 0.957798957824707, "rewards/code_complexity_reward/mean": 0.5116211175918579, "rewards/code_complexity_reward/std": 0.39768990874290466, "rewards/code_execution_reward/mean": 0.21875, "rewards/code_execution_reward/std": 0.41380295157432556, "rewards/code_syntax_reward/mean": 0.3203125, "rewards/code_syntax_reward/std": 0.24014326930046082, "rewards/reasoning_present_reward_func/mean": 0.04160156100988388, "rewards/reasoning_present_reward_func/std": 0.04933782294392586, "rewards/xmlcount_reward_func/mean": 0.19091796875, "rewards/xmlcount_reward_func/std": 0.21963439881801605, "step": 81, "step_time": 58.81491813343018 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 341.435546875, "completions/mean_terminated_length": 307.0023498535156, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2752939711790532, "epoch": 0.09350057012542759, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.01891406439244747, "kl": 0.0011429099040469737, "learning_rate": 4.6022727272727275e-06, "loss": 5.670794053003192e-06, "num_tokens": 21068914.0, "reward": 1.406103491783142, "reward_std": 0.9947848916053772, "rewards/code_complexity_reward/mean": 0.5517578125, "rewards/code_complexity_reward/std": 0.3989551365375519, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.333984375, "rewards/code_syntax_reward/std": 0.23570136725902557, "rewards/reasoning_present_reward_func/mean": 0.04355468600988388, "rewards/reasoning_present_reward_func/std": 0.049631327390670776, "rewards/xmlcount_reward_func/mean": 0.201416015625, "rewards/xmlcount_reward_func/std": 0.21949772536754608, "step": 82, "step_time": 64.12830250803381 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 356.607421875, "completions/mean_terminated_length": 328.6797180175781, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.2936073027085513, "epoch": 0.09464082098061574, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.020900553092360497, "kl": 0.001172297062112193, "learning_rate": 4.6590909090909095e-06, "loss": 5.877809599041939e-06, "num_tokens": 21319237.0, "reward": 1.2363770008087158, "reward_std": 0.9669280052185059, "rewards/code_complexity_reward/mean": 0.47832030057907104, "rewards/code_complexity_reward/std": 0.40579065680503845, "rewards/code_execution_reward/mean": 0.208984375, "rewards/code_execution_reward/std": 0.40698084235191345, "rewards/code_syntax_reward/mean": 0.296875, "rewards/code_syntax_reward/std": 0.24580632150173187, "rewards/reasoning_present_reward_func/mean": 0.044921875, "rewards/reasoning_present_reward_func/std": 0.04979010671377182, "rewards/xmlcount_reward_func/mean": 0.207275390625, "rewards/xmlcount_reward_func/std": 0.22140705585479736, "step": 83, "step_time": 64.5250125369057 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.19140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 359.224609375, "completions/mean_terminated_length": 323.0603942871094, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.2821024931035936, "epoch": 0.09578107183580388, "frac_reward_zero_std": 0.0, "grad_norm": 0.017594926059246063, "kl": 0.001163364625426766, "learning_rate": 4.715909090909091e-06, "loss": 5.86951500736177e-06, "num_tokens": 21570572.0, "reward": 1.3840820789337158, "reward_std": 0.9905352592468262, "rewards/code_complexity_reward/mean": 0.5305664539337158, "rewards/code_complexity_reward/std": 0.4008287191390991, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.3251953125, "rewards/code_syntax_reward/std": 0.23865646123886108, "rewards/reasoning_present_reward_func/mean": 0.0439453125, "rewards/reasoning_present_reward_func/std": 0.049680594354867935, "rewards/xmlcount_reward_func/mean": 0.203125, "rewards/xmlcount_reward_func/std": 0.2184046357870102, "step": 84, "step_time": 73.60944599285722 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.173828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 351.74609375, "completions/mean_terminated_length": 318.02838134765625, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.2865317747928202, "epoch": 0.09692132269099202, "frac_reward_zero_std": 0.0, "grad_norm": 0.018545234575867653, "kl": 0.0012141327861172613, "learning_rate": 4.772727272727273e-06, "loss": 6.064437911845744e-06, "num_tokens": 21819542.0, "reward": 1.33447265625, "reward_std": 0.9824702143669128, "rewards/code_complexity_reward/mean": 0.518261730670929, "rewards/code_complexity_reward/std": 0.4007008373737335, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.3212890625, "rewards/code_syntax_reward/std": 0.2398546040058136, "rewards/reasoning_present_reward_func/mean": 0.04570312798023224, "rewards/reasoning_present_reward_func/std": 0.04986374452710152, "rewards/xmlcount_reward_func/mean": 0.216796875, "rewards/xmlcount_reward_func/std": 0.22447174787521362, "step": 85, "step_time": 62.08776291646063 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 357.04296875, "completions/mean_terminated_length": 322.1961669921875, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.28294387576170266, "epoch": 0.09806157354618016, "frac_reward_zero_std": 0.015625, "grad_norm": 0.018002118915319443, "kl": 0.001278042088415532, "learning_rate": 4.829545454545455e-06, "loss": 6.391317583620548e-06, "num_tokens": 22071260.0, "reward": 1.3639647960662842, "reward_std": 1.0034457445144653, "rewards/code_complexity_reward/mean": 0.51611328125, "rewards/code_complexity_reward/std": 0.39616450667381287, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.3212890625, "rewards/code_syntax_reward/std": 0.2398546040058136, "rewards/reasoning_present_reward_func/mean": 0.04316405951976776, "rewards/reasoning_present_reward_func/std": 0.049578938633203506, "rewards/xmlcount_reward_func/mean": 0.2021484375, "rewards/xmlcount_reward_func/std": 0.21791188418865204, "step": 86, "step_time": 54.02096394170076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.181640625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 357.138671875, "completions/mean_terminated_length": 322.76611328125, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.2859671297483146, "epoch": 0.09920182440136831, "frac_reward_zero_std": 0.0, "grad_norm": 0.018190262839198112, "kl": 0.0013100496498736902, "learning_rate": 4.8863636363636365e-06, "loss": 6.545567885041237e-06, "num_tokens": 22323199.0, "reward": 1.339111328125, "reward_std": 0.9874517917633057, "rewards/code_complexity_reward/mean": 0.5212891101837158, "rewards/code_complexity_reward/std": 0.4030897319316864, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.3203125, "rewards/code_syntax_reward/std": 0.24014326930046082, "rewards/reasoning_present_reward_func/mean": 0.04609375074505806, "rewards/reasoning_present_reward_func/std": 0.04989592730998993, "rewards/xmlcount_reward_func/mean": 0.215087890625, "rewards/xmlcount_reward_func/std": 0.22577469050884247, "step": 87, "step_time": 62.81248601898551 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.138671875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 351.951171875, "completions/mean_terminated_length": 326.1836853027344, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.28674162668175995, "epoch": 0.10034207525655645, "frac_reward_zero_std": 0.0, "grad_norm": 0.01781177707016468, "kl": 0.001381678610414383, "learning_rate": 4.9431818181818184e-06, "loss": 6.899237632751465e-06, "num_tokens": 22572606.0, "reward": 1.314697265625, "reward_std": 0.9675344824790955, "rewards/code_complexity_reward/mean": 0.5055663585662842, "rewards/code_complexity_reward/std": 0.39968588948249817, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.314453125, "rewards/code_syntax_reward/std": 0.2417849749326706, "rewards/reasoning_present_reward_func/mean": 0.04374999925494194, "rewards/reasoning_present_reward_func/std": 0.049656353890895844, "rewards/xmlcount_reward_func/mean": 0.204833984375, "rewards/xmlcount_reward_func/std": 0.2164466232061386, "step": 88, "step_time": 64.02244980633259 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 362.146484375, "completions/mean_terminated_length": 327.5649108886719, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.27856825524941087, "epoch": 0.10148232611174458, "frac_reward_zero_std": 0.0, "grad_norm": 0.017183557152748108, "kl": 0.0013527188693842618, "learning_rate": 5e-06, "loss": 6.759190000593662e-06, "num_tokens": 22827081.0, "reward": 1.350195288658142, "reward_std": 0.9467422962188721, "rewards/code_complexity_reward/mean": 0.5257812738418579, "rewards/code_complexity_reward/std": 0.3963993787765503, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.3271484375, "rewards/code_syntax_reward/std": 0.2380310446023941, "rewards/reasoning_present_reward_func/mean": 0.04707030951976776, "rewards/reasoning_present_reward_func/std": 0.049962908029556274, "rewards/xmlcount_reward_func/mean": 0.2119140625, "rewards/xmlcount_reward_func/std": 0.21787680685520172, "step": 89, "step_time": 65.50574641022831 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 357.052734375, "completions/mean_terminated_length": 321.2956848144531, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.29249624465592206, "epoch": 0.10262257696693272, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.01895245537161827, "kl": 0.001492623918238678, "learning_rate": 4.999980182212003e-06, "loss": 7.48754246160388e-06, "num_tokens": 23079236.0, "reward": 1.2943847179412842, "reward_std": 0.969855785369873, "rewards/code_complexity_reward/mean": 0.4921874701976776, "rewards/code_complexity_reward/std": 0.4036253094673157, "rewards/code_execution_reward/mean": 0.224609375, "rewards/code_execution_reward/std": 0.41773295402526855, "rewards/code_syntax_reward/mean": 0.3076171875, "rewards/code_syntax_reward/std": 0.24350784718990326, "rewards/reasoning_present_reward_func/mean": 0.04902343451976776, "rewards/reasoning_present_reward_func/std": 0.05003935098648071, "rewards/xmlcount_reward_func/mean": 0.220947265625, "rewards/xmlcount_reward_func/std": 0.22017145156860352, "step": 90, "step_time": 63.53884037863463 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.154296875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 348.111328125, "completions/mean_terminated_length": 318.2101745605469, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.2835780919995159, "epoch": 0.10376282782212087, "frac_reward_zero_std": 0.0, "grad_norm": 0.01968037150800228, "kl": 0.0014616533280786825, "learning_rate": 4.999920729162207e-06, "loss": 7.321825250983238e-06, "num_tokens": 23326321.0, "reward": 1.412841796875, "reward_std": 0.9597908854484558, "rewards/code_complexity_reward/mean": 0.5399414300918579, "rewards/code_complexity_reward/std": 0.3869841694831848, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.33984375, "rewards/code_syntax_reward/std": 0.23352646827697754, "rewards/reasoning_present_reward_func/mean": 0.04941406100988388, "rewards/reasoning_present_reward_func/std": 0.05004546418786049, "rewards/xmlcount_reward_func/mean": 0.227783203125, "rewards/xmlcount_reward_func/std": 0.2205519676208496, "step": 91, "step_time": 62.983607821166515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 341.234375, "completions/mean_terminated_length": 318.5663757324219, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.2868186279665679, "epoch": 0.10490307867730901, "frac_reward_zero_std": 0.0, "grad_norm": 0.01947901025414467, "kl": 0.0015654160979465814, "learning_rate": 4.999821641793195e-06, "loss": 7.824855856597424e-06, "num_tokens": 23567973.0, "reward": 1.4456055164337158, "reward_std": 0.978014349937439, "rewards/code_complexity_reward/mean": 0.549609363079071, "rewards/code_complexity_reward/std": 0.3927297294139862, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.3369140625, "rewards/code_syntax_reward/std": 0.23463475704193115, "rewards/reasoning_present_reward_func/mean": 0.05078125, "rewards/reasoning_present_reward_func/std": 0.05004279315471649, "rewards/xmlcount_reward_func/mean": 0.23876953125, "rewards/xmlcount_reward_func/std": 0.21783457696437836, "step": 92, "step_time": 74.67795192450285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.154296875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 347.4609375, "completions/mean_terminated_length": 317.44110107421875, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.28226515697315335, "epoch": 0.10604332953249715, "frac_reward_zero_std": 0.0, "grad_norm": 0.019429588690400124, "kl": 0.0017306460304098437, "learning_rate": 4.999682921675919e-06, "loss": 8.634146070107818e-06, "num_tokens": 23812601.0, "reward": 1.45849609375, "reward_std": 0.9789947271347046, "rewards/code_complexity_reward/mean": 0.553027331829071, "rewards/code_complexity_reward/std": 0.3933841586112976, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.337890625, "rewards/code_syntax_reward/std": 0.23426999151706696, "rewards/reasoning_present_reward_func/mean": 0.05292969197034836, "rewards/reasoning_present_reward_func/std": 0.049962908029556274, "rewards/xmlcount_reward_func/mean": 0.2431640625, "rewards/xmlcount_reward_func/std": 0.22218482196331024, "step": 93, "step_time": 72.18699174560606 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 341.51953125, "completions/mean_terminated_length": 310.88018798828125, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.28236883459612727, "epoch": 0.10718358038768529, "frac_reward_zero_std": 0.0, "grad_norm": 0.026028532534837723, "kl": 0.001800233731046319, "learning_rate": 4.999504571009682e-06, "loss": 8.986098691821098e-06, "num_tokens": 24056383.0, "reward": 1.4964842796325684, "reward_std": 1.0014678239822388, "rewards/code_complexity_reward/mean": 0.555957019329071, "rewards/code_complexity_reward/std": 0.3852701783180237, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.34375, "rewards/code_syntax_reward/std": 0.23198285698890686, "rewards/reasoning_present_reward_func/mean": 0.05429687723517418, "rewards/reasoning_present_reward_func/std": 0.049863748252391815, "rewards/xmlcount_reward_func/mean": 0.24755859375, "rewards/xmlcount_reward_func/std": 0.22364813089370728, "step": 94, "step_time": 74.35079770442098 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.154296875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 352.349609375, "completions/mean_terminated_length": 323.2217102050781, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.28803384955972433, "epoch": 0.10832383124287344, "frac_reward_zero_std": 0.0, "grad_norm": 0.020317409187555313, "kl": 0.0017559669340698747, "learning_rate": 4.999286592622096e-06, "loss": 8.784118108451366e-06, "num_tokens": 24306594.0, "reward": 1.458154320716858, "reward_std": 0.956813633441925, "rewards/code_complexity_reward/mean": 0.55615234375, "rewards/code_complexity_reward/std": 0.38759666681289673, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.34375, "rewards/code_syntax_reward/std": 0.23198285698890686, "rewards/reasoning_present_reward_func/mean": 0.05410156399011612, "rewards/reasoning_present_reward_func/std": 0.049880221486091614, "rewards/xmlcount_reward_func/mean": 0.252197265625, "rewards/xmlcount_reward_func/std": 0.22152139246463776, "step": 95, "step_time": 61.4480580445379 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 352.26171875, "completions/mean_terminated_length": 315.3990478515625, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.2747201514430344, "epoch": 0.10946408209806158, "frac_reward_zero_std": 0.015625, "grad_norm": 0.020548952743411064, "kl": 0.0018623840624059085, "learning_rate": 4.99902898996904e-06, "loss": 9.283947292715311e-06, "num_tokens": 24557108.0, "reward": 1.4430664777755737, "reward_std": 0.971332311630249, "rewards/code_complexity_reward/mean": 0.5582031011581421, "rewards/code_complexity_reward/std": 0.3901691734790802, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.341796875, "rewards/code_syntax_reward/std": 0.23276415467262268, "rewards/reasoning_present_reward_func/mean": 0.05234374850988388, "rewards/reasoning_present_reward_func/std": 0.049993883818387985, "rewards/xmlcount_reward_func/mean": 0.23876953125, "rewards/xmlcount_reward_func/std": 0.22020792961120605, "step": 96, "step_time": 82.37487390916795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 349.072265625, "completions/mean_terminated_length": 317.0957946777344, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.29120792215690017, "epoch": 0.11060433295324971, "frac_reward_zero_std": 0.0, "grad_norm": 0.021802926436066628, "kl": 0.0020911594219796825, "learning_rate": 4.998731767134606e-06, "loss": 1.0519474017200992e-05, "num_tokens": 24802109.0, "reward": 1.5322265625, "reward_std": 0.9706323742866516, "rewards/code_complexity_reward/mean": 0.5730469226837158, "rewards/code_complexity_reward/std": 0.38083502650260925, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.3564453125, "rewards/code_syntax_reward/std": 0.22642776370048523, "rewards/reasoning_present_reward_func/mean": 0.05781250074505806, "rewards/reasoning_present_reward_func/std": 0.049434177577495575, "rewards/xmlcount_reward_func/mean": 0.271484375, "rewards/xmlcount_reward_func/std": 0.2194434404373169, "step": 97, "step_time": 52.06469347793609 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.13671875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 348.47265625, "completions/mean_terminated_length": 322.5746765136719, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.28937540971674025, "epoch": 0.11174458380843785, "frac_reward_zero_std": 0.0, "grad_norm": 0.02171349711716175, "kl": 0.0021226099015621003, "learning_rate": 4.998394928831034e-06, "loss": 1.0591174941509962e-05, "num_tokens": 25047591.0, "reward": 1.5471192598342896, "reward_std": 0.9662670493125916, "rewards/code_complexity_reward/mean": 0.5751953125, "rewards/code_complexity_reward/std": 0.37897273898124695, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.35546875, "rewards/code_syntax_reward/std": 0.22688518464565277, "rewards/reasoning_present_reward_func/mean": 0.05859375, "rewards/reasoning_present_reward_func/std": 0.04930410906672478, "rewards/xmlcount_reward_func/mean": 0.270751953125, "rewards/xmlcount_reward_func/std": 0.21635720133781433, "step": 98, "step_time": 90.3502164920792 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.17578125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 353.28125, "completions/mean_terminated_length": 319.4313049316406, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.29585155728273094, "epoch": 0.11288483466362599, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.020427849143743515, "kl": 0.00275433377828449, "learning_rate": 4.998018480398635e-06, "loss": 1.3773038517683744e-05, "num_tokens": 25296815.0, "reward": 1.4861328601837158, "reward_std": 0.9557520747184753, "rewards/code_complexity_reward/mean": 0.5565429925918579, "rewards/code_complexity_reward/std": 0.39183342456817627, "rewards/code_execution_reward/mean": 0.2421875, "rewards/code_execution_reward/std": 0.42882615327835083, "rewards/code_syntax_reward/mean": 0.3408203125, "rewards/code_syntax_reward/std": 0.23314768075942993, "rewards/reasoning_present_reward_func/mean": 0.06289063394069672, "rewards/reasoning_present_reward_func/std": 0.04835699871182442, "rewards/xmlcount_reward_func/mean": 0.28369140625, "rewards/xmlcount_reward_func/std": 0.21493330597877502, "step": 99, "step_time": 64.85692823305726 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15234375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 342.27734375, "completions/mean_terminated_length": 311.7742004394531, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.29113137419335544, "epoch": 0.11402508551881414, "frac_reward_zero_std": 0.0, "grad_norm": 0.02126418612897396, "kl": 0.002447525559546193, "learning_rate": 4.99760242780571e-06, "loss": 1.2222735676914454e-05, "num_tokens": 25539797.0, "reward": 1.5835449695587158, "reward_std": 0.9543373584747314, "rewards/code_complexity_reward/mean": 0.5891602039337158, "rewards/code_complexity_reward/std": 0.3805761933326721, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.3603515625, "rewards/code_syntax_reward/std": 0.22454623878002167, "rewards/reasoning_present_reward_func/mean": 0.0625, "rewards/reasoning_present_reward_func/std": 0.04845963791012764, "rewards/xmlcount_reward_func/mean": 0.290283203125, "rewards/xmlcount_reward_func/std": 0.21199838817119598, "step": 100, "step_time": 65.66937005147338 }, { "epoch": 0.11402508551881414, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.135, "eval_completions/max_length": 488.4, "eval_completions/max_terminated_length": 445.34, "eval_completions/mean_length": 340.21, "eval_completions/mean_terminated_length": 315.91131713867185, "eval_completions/min_length": 195.5, "eval_completions/min_terminated_length": 195.5, "eval_entropy": 0.2941432127356529, "eval_frac_reward_zero_std": 0.0, "eval_kl": 0.0026192311220802366, "eval_loss": 1.315637109655654e-05, "eval_num_tokens": 25539797.0, "eval_reward": 1.5812500095367432, "eval_reward_std": 0.8673433327674865, "eval_rewards/code_complexity_reward/mean": 0.5882499998807907, "eval_rewards/code_complexity_reward/std": 0.3320083460956812, "eval_rewards/code_execution_reward/mean": 0.2825, "eval_rewards/code_execution_reward/std": 0.37546138644218446, "eval_rewards/code_syntax_reward/mean": 0.3675, "eval_rewards/code_syntax_reward/std": 0.19862963616847992, "eval_rewards/reasoning_present_reward_func/mean": 0.06175000201910734, "eval_rewards/reasoning_present_reward_func/std": 0.04612286478281021, "eval_rewards/xmlcount_reward_func/mean": 0.28125, "eval_rewards/xmlcount_reward_func/std": 0.20750596411526204, "eval_runtime": 1070.0232, "eval_samples_per_second": 0.093, "eval_steps_per_second": 0.012, "step": 100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.18359375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 345.2890625, "completions/mean_terminated_length": 307.7990417480469, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.29240650683641434, "epoch": 0.11516533637400228, "frac_reward_zero_std": 0.015625, "grad_norm": 0.02150583826005459, "kl": 0.0026704321517172502, "learning_rate": 4.9971467776484526e-06, "loss": 1.3303069863468409e-05, "num_tokens": 25786921.0, "reward": 1.5520507097244263, "reward_std": 0.9915990829467773, "rewards/code_complexity_reward/mean": 0.576171875, "rewards/code_complexity_reward/std": 0.3760160207748413, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.357421875, "rewards/code_syntax_reward/std": 0.22596518695354462, "rewards/reasoning_present_reward_func/mean": 0.05839844048023224, "rewards/reasoning_present_reward_func/std": 0.04933782294392586, "rewards/xmlcount_reward_func/mean": 0.27490234375, "rewards/xmlcount_reward_func/std": 0.21697752177715302, "step": 101, "step_time": 64.12120936438441 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14453125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 348.888671875, "completions/mean_terminated_length": 321.3310241699219, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.30048679700121284, "epoch": 0.11630558722919042, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.020123859867453575, "kl": 0.0027580724636209197, "learning_rate": 4.9966515371508445e-06, "loss": 1.3739860150963068e-05, "num_tokens": 26034604.0, "reward": 1.5498046875, "reward_std": 0.9527049660682678, "rewards/code_complexity_reward/mean": 0.5694335699081421, "rewards/code_complexity_reward/std": 0.37939631938934326, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.3544921875, "rewards/code_syntax_reward/std": 0.22733746469020844, "rewards/reasoning_present_reward_func/mean": 0.06484374403953552, "rewards/reasoning_present_reward_func/std": 0.04779251292347908, "rewards/xmlcount_reward_func/mean": 0.29931640625, "rewards/xmlcount_reward_func/std": 0.21014979481697083, "step": 102, "step_time": 65.34954097587615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 335.03125, "completions/mean_terminated_length": 314.16595458984375, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.28849679604172707, "epoch": 0.11744583808437856, "frac_reward_zero_std": 0.0, "grad_norm": 0.02301551029086113, "kl": 0.003208503490895964, "learning_rate": 4.9961167141645435e-06, "loss": 1.6068515833467245e-05, "num_tokens": 26275096.0, "reward": 1.6709473133087158, "reward_std": 0.9258858561515808, "rewards/code_complexity_reward/mean": 0.6021484136581421, "rewards/code_complexity_reward/std": 0.3707706928253174, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.3701171875, "rewards/code_syntax_reward/std": 0.2194673866033554, "rewards/reasoning_present_reward_func/mean": 0.07246094197034836, "rewards/reasoning_present_reward_func/std": 0.044714778661727905, "rewards/xmlcount_reward_func/mean": 0.341064453125, "rewards/xmlcount_reward_func/std": 0.1908549815416336, "step": 103, "step_time": 66.46742796711624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.126953125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 330.84375, "completions/mean_terminated_length": 304.5011291503906, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.288053767522797, "epoch": 0.11858608893956671, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.020980853587388992, "kl": 0.0032619473968225066, "learning_rate": 4.995542317168756e-06, "loss": 1.630352926440537e-05, "num_tokens": 26512680.0, "reward": 1.753564476966858, "reward_std": 0.9200970530509949, "rewards/code_complexity_reward/mean": 0.6359374523162842, "rewards/code_complexity_reward/std": 0.35144028067588806, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.390625, "rewards/code_syntax_reward/std": 0.20690147578716278, "rewards/reasoning_present_reward_func/mean": 0.07050780951976776, "rewards/reasoning_present_reward_func/std": 0.04564535990357399, "rewards/xmlcount_reward_func/mean": 0.330322265625, "rewards/xmlcount_reward_func/std": 0.1994972676038742, "step": 104, "step_time": 63.145911457017064 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 331.283203125, "completions/mean_terminated_length": 299.7821044921875, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.2944441989529878, "epoch": 0.11972633979475485, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.024071892723441124, "kl": 0.0036638019264501054, "learning_rate": 4.994928355270105e-06, "loss": 1.831405097618699e-05, "num_tokens": 26750729.0, "reward": 1.7240722179412842, "reward_std": 0.9266720414161682, "rewards/code_complexity_reward/mean": 0.6262695789337158, "rewards/code_complexity_reward/std": 0.3529139757156372, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.38671875, "rewards/code_syntax_reward/std": 0.2095082700252533, "rewards/reasoning_present_reward_func/mean": 0.07070313394069672, "rewards/reasoning_present_reward_func/std": 0.0455569326877594, "rewards/xmlcount_reward_func/mean": 0.333740234375, "rewards/xmlcount_reward_func/std": 0.19684529304504395, "step": 105, "step_time": 65.98329093493521 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.150390625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 348.376953125, "completions/mean_terminated_length": 319.4137878417969, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.2903658014256507, "epoch": 0.12086659064994298, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.018640384078025818, "kl": 0.0032575417280895635, "learning_rate": 4.994274838202483e-06, "loss": 1.6275029338430613e-05, "num_tokens": 26997850.0, "reward": 1.738867163658142, "reward_std": 0.8888472318649292, "rewards/code_complexity_reward/mean": 0.62744140625, "rewards/code_complexity_reward/std": 0.3400585651397705, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.3955078125, "rewards/code_syntax_reward/std": 0.20349042117595673, "rewards/reasoning_present_reward_func/mean": 0.07382813096046448, "rewards/reasoning_present_reward_func/std": 0.04400002211332321, "rewards/xmlcount_reward_func/mean": 0.33740234375, "rewards/xmlcount_reward_func/std": 0.18863239884376526, "step": 106, "step_time": 65.94735187292099 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 330.888671875, "completions/mean_terminated_length": 301.25225830078125, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.2907344426494092, "epoch": 0.12200684150513112, "frac_reward_zero_std": 0.015625, "grad_norm": 0.021258071064949036, "kl": 0.004063368156494107, "learning_rate": 4.993581776326901e-06, "loss": 2.0318664610385895e-05, "num_tokens": 27234609.0, "reward": 1.6885743141174316, "reward_std": 0.9213221073150635, "rewards/code_complexity_reward/mean": 0.5987304449081421, "rewards/code_complexity_reward/std": 0.36424511671066284, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.373046875, "rewards/code_syntax_reward/std": 0.21783512830734253, "rewards/reasoning_present_reward_func/mean": 0.0751953125, "rewards/reasoning_present_reward_func/std": 0.04323015734553337, "rewards/xmlcount_reward_func/mean": 0.3525390625, "rewards/xmlcount_reward_func/std": 0.18616770207881927, "step": 107, "step_time": 54.05609923880547 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 326.947265625, "completions/mean_terminated_length": 300.51116943359375, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.2974286142271012, "epoch": 0.12314709236031927, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.021794460713863373, "kl": 0.004448721800144995, "learning_rate": 4.9928491806313216e-06, "loss": 2.2234366042539477e-05, "num_tokens": 27470102.0, "reward": 1.7616698741912842, "reward_std": 0.8926076292991638, "rewards/code_complexity_reward/mean": 0.6573241949081421, "rewards/code_complexity_reward/std": 0.3451084792613983, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.3974609375, "rewards/code_syntax_reward/std": 0.2020767778158188, "rewards/reasoning_present_reward_func/mean": 0.07285156100988388, "rewards/reasoning_present_reward_func/std": 0.04451602324843407, "rewards/xmlcount_reward_func/mean": 0.344970703125, "rewards/xmlcount_reward_func/std": 0.19451969861984253, "step": 108, "step_time": 70.2363882791251 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 307.75, "completions/mean_terminated_length": 285.64501953125, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.2820131208281964, "epoch": 0.12428734321550741, "frac_reward_zero_std": 0.015625, "grad_norm": 0.024293646216392517, "kl": 0.0045162201386119705, "learning_rate": 4.992077062730485e-06, "loss": 2.2571533918380737e-05, "num_tokens": 27695046.0, "reward": 1.9524903297424316, "reward_std": 0.8998233079910278, "rewards/code_complexity_reward/mean": 0.683398425579071, "rewards/code_complexity_reward/std": 0.3215632140636444, "rewards/code_execution_reward/mean": 0.41015625, "rewards/code_execution_reward/std": 0.49234291911125183, "rewards/code_syntax_reward/mean": 0.416015625, "rewards/code_syntax_reward/std": 0.1871020793914795, "rewards/reasoning_present_reward_func/mean": 0.07695312052965164, "rewards/reasoning_present_reward_func/std": 0.042154476046562195, "rewards/xmlcount_reward_func/mean": 0.365966796875, "rewards/xmlcount_reward_func/std": 0.18574489653110504, "step": 109, "step_time": 63.135353120975196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 327.427734375, "completions/mean_terminated_length": 297.2249755859375, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.2841554421465844, "epoch": 0.12542759407069556, "frac_reward_zero_std": 0.015625, "grad_norm": 0.02301662042737007, "kl": 0.004606740349117899, "learning_rate": 4.991265434865726e-06, "loss": 2.3005430193734355e-05, "num_tokens": 27930057.0, "reward": 1.7630372047424316, "reward_std": 0.8941671848297119, "rewards/code_complexity_reward/mean": 0.6312500238418579, "rewards/code_complexity_reward/std": 0.3416891396045685, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.3955078125, "rewards/code_syntax_reward/std": 0.20349042117595673, "rewards/reasoning_present_reward_func/mean": 0.08027344197034836, "rewards/reasoning_present_reward_func/std": 0.03983237221837044, "rewards/xmlcount_reward_func/mean": 0.366943359375, "rewards/xmlcount_reward_func/std": 0.18027697503566742, "step": 110, "step_time": 61.38687052484602 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.134765625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 330.8515625, "completions/mean_terminated_length": 302.6365661621094, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.2992855226621032, "epoch": 0.1265678449258837, "frac_reward_zero_std": 0.015625, "grad_norm": 0.021167274564504623, "kl": 0.004746131020510802, "learning_rate": 4.990414309904781e-06, "loss": 2.366863191127777e-05, "num_tokens": 28169741.0, "reward": 1.7827637195587158, "reward_std": 0.8762282729148865, "rewards/code_complexity_reward/mean": 0.638671875, "rewards/code_complexity_reward/std": 0.3383634388446808, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.3984375, "rewards/code_syntax_reward/std": 0.2013591229915619, "rewards/reasoning_present_reward_func/mean": 0.07890625298023224, "rewards/reasoning_present_reward_func/std": 0.04083731025457382, "rewards/xmlcount_reward_func/mean": 0.369873046875, "rewards/xmlcount_reward_func/std": 0.17713426053524017, "step": 111, "step_time": 64.73775178659707 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.103515625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 314.044921875, "completions/mean_terminated_length": 291.1873474121094, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2869799875188619, "epoch": 0.12770809578107184, "frac_reward_zero_std": 0.0, "grad_norm": 0.024537255987524986, "kl": 0.005529759288037894, "learning_rate": 4.98952370134158e-06, "loss": 2.7605135983321816e-05, "num_tokens": 28397472.0, "reward": 1.8639647960662842, "reward_std": 0.9127415418624878, "rewards/code_complexity_reward/mean": 0.648632824420929, "rewards/code_complexity_reward/std": 0.33601218461990356, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4013671875, "rewards/code_syntax_reward/std": 0.1991618573665619, "rewards/reasoning_present_reward_func/mean": 0.080078125, "rewards/reasoning_present_reward_func/std": 0.03998035192489624, "rewards/xmlcount_reward_func/mean": 0.36865234375, "rewards/xmlcount_reward_func/std": 0.17298954725265503, "step": 112, "step_time": 70.27905899193138 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 330.408203125, "completions/mean_terminated_length": 298.75457763671875, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.2891309093683958, "epoch": 0.12884834663625996, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.023192478343844414, "kl": 0.005316921677149367, "learning_rate": 4.988593623296038e-06, "loss": 2.6586261810734868e-05, "num_tokens": 28634757.0, "reward": 1.8051270246505737, "reward_std": 0.8933266997337341, "rewards/code_complexity_reward/mean": 0.6433594226837158, "rewards/code_complexity_reward/std": 0.34551194310188293, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.3955078125, "rewards/code_syntax_reward/std": 0.20349042117595673, "rewards/reasoning_present_reward_func/mean": 0.08437500149011612, "rewards/reasoning_present_reward_func/std": 0.0363447330892086, "rewards/xmlcount_reward_func/mean": 0.392822265625, "rewards/xmlcount_reward_func/std": 0.15290242433547974, "step": 113, "step_time": 64.681621177122 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 317.068359375, "completions/mean_terminated_length": 294.08514404296875, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.2912318736780435, "epoch": 0.12998859749144812, "frac_reward_zero_std": 0.0, "grad_norm": 0.020262105390429497, "kl": 0.005892714427318424, "learning_rate": 4.987624090513825e-06, "loss": 2.9362388886511326e-05, "num_tokens": 28866012.0, "reward": 1.850341796875, "reward_std": 0.8233169913291931, "rewards/code_complexity_reward/mean": 0.6667969226837158, "rewards/code_complexity_reward/std": 0.32073041796684265, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4140625, "rewards/code_syntax_reward/std": 0.18882036209106445, "rewards/reasoning_present_reward_func/mean": 0.08417969197034836, "rewards/reasoning_present_reward_func/std": 0.036528829485177994, "rewards/xmlcount_reward_func/mean": 0.404052734375, "rewards/xmlcount_reward_func/std": 0.15417203307151794, "step": 114, "step_time": 61.13415234722197 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12890625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 332.17578125, "completions/mean_terminated_length": 305.5650329589844, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.28967058076523244, "epoch": 0.13112884834663627, "frac_reward_zero_std": 0.015625, "grad_norm": 0.019244223833084106, "kl": 0.005801345541840419, "learning_rate": 4.986615118366138e-06, "loss": 2.9009534046053886e-05, "num_tokens": 29104666.0, "reward": 1.8229002952575684, "reward_std": 0.8286511301994324, "rewards/code_complexity_reward/mean": 0.654492199420929, "rewards/code_complexity_reward/std": 0.3166905343532562, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.416015625, "rewards/code_syntax_reward/std": 0.1871020793914795, "rewards/reasoning_present_reward_func/mean": 0.08320312201976776, "rewards/reasoning_present_reward_func/std": 0.03742041438817978, "rewards/xmlcount_reward_func/mean": 0.393798828125, "rewards/xmlcount_reward_func/std": 0.15906056761741638, "step": 115, "step_time": 59.1826935056597 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.12109375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 318.28515625, "completions/mean_terminated_length": 291.5955505371094, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.2910815612412989, "epoch": 0.1322690992018244, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.019634844735264778, "kl": 0.006047241444321116, "learning_rate": 4.985566722849454e-06, "loss": 3.0223631256376393e-05, "num_tokens": 29336132.0, "reward": 1.904882788658142, "reward_std": 0.8686331510543823, "rewards/code_complexity_reward/mean": 0.6641601324081421, "rewards/code_complexity_reward/std": 0.3200105130672455, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4140625, "rewards/code_syntax_reward/std": 0.18882036209106445, "rewards/reasoning_present_reward_func/mean": 0.0869140625, "rewards/reasoning_present_reward_func/std": 0.03375763073563576, "rewards/xmlcount_reward_func/mean": 0.40380859375, "rewards/xmlcount_reward_func/std": 0.1515175849199295, "step": 116, "step_time": 73.11432259343565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 319.306640625, "completions/mean_terminated_length": 291.7790222167969, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.291611090535298, "epoch": 0.13340935005701254, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.018146147951483727, "kl": 0.00656001106108306, "learning_rate": 4.984478920585277e-06, "loss": 3.278983058407903e-05, "num_tokens": 29566921.0, "reward": 1.9215821027755737, "reward_std": 0.8235973715782166, "rewards/code_complexity_reward/mean": 0.6833984851837158, "rewards/code_complexity_reward/std": 0.3045123219490051, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.42578125, "rewards/code_syntax_reward/std": 0.17794041335582733, "rewards/reasoning_present_reward_func/mean": 0.08535157144069672, "rewards/reasoning_present_reward_func/std": 0.035393696278333664, "rewards/xmlcount_reward_func/mean": 0.40673828125, "rewards/xmlcount_reward_func/std": 0.15092994272708893, "step": 117, "step_time": 63.83710995037109 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.115234375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 317.54296875, "completions/mean_terminated_length": 292.2163391113281, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.2895487598143518, "epoch": 0.1345496009122007, "frac_reward_zero_std": 0.015625, "grad_norm": 0.022278524935245514, "kl": 0.00777465921419207, "learning_rate": 4.983351728819874e-06, "loss": 3.893449320457876e-05, "num_tokens": 29797663.0, "reward": 1.901269555091858, "reward_std": 0.8443570137023926, "rewards/code_complexity_reward/mean": 0.6770508289337158, "rewards/code_complexity_reward/std": 0.3221832513809204, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4150390625, "rewards/code_syntax_reward/std": 0.1879657357931137, "rewards/reasoning_present_reward_func/mean": 0.08554687350988388, "rewards/reasoning_present_reward_func/std": 0.03519714996218681, "rewards/xmlcount_reward_func/mean": 0.4091796875, "rewards/xmlcount_reward_func/std": 0.14522305130958557, "step": 118, "step_time": 63.68139868136495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08203125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 298.212890625, "completions/mean_terminated_length": 279.1084899902344, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.28941010776907206, "epoch": 0.13568985176738882, "frac_reward_zero_std": 0.0, "grad_norm": 0.020093025639653206, "kl": 0.00747263844095869, "learning_rate": 4.9821851654240025e-06, "loss": 3.736821236088872e-05, "num_tokens": 30017052.0, "reward": 2.039306640625, "reward_std": 0.8142529726028442, "rewards/code_complexity_reward/mean": 0.7181640863418579, "rewards/code_complexity_reward/std": 0.28596001863479614, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.44140625, "rewards/code_syntax_reward/std": 0.16097907721996307, "rewards/reasoning_present_reward_func/mean": 0.08847656846046448, "rewards/reasoning_present_reward_func/std": 0.03196168690919876, "rewards/xmlcount_reward_func/mean": 0.418212890625, "rewards/xmlcount_reward_func/std": 0.14182408154010773, "step": 119, "step_time": 72.60144274868071 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 313.9765625, "completions/mean_terminated_length": 292.5454406738281, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.28645648737438023, "epoch": 0.13683010262257697, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.020160330459475517, "kl": 0.007415076695906464, "learning_rate": 4.98097924889263e-06, "loss": 3.703788388520479e-05, "num_tokens": 30247120.0, "reward": 2.0063962936401367, "reward_std": 0.783872663974762, "rewards/code_complexity_reward/mean": 0.7194335460662842, "rewards/code_complexity_reward/std": 0.2761708199977875, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.443359375, "rewards/code_syntax_reward/std": 0.1586231142282486, "rewards/reasoning_present_reward_func/mean": 0.08945313096046448, "rewards/reasoning_present_reward_func/std": 0.03074568696320057, "rewards/xmlcount_reward_func/mean": 0.422119140625, "rewards/xmlcount_reward_func/std": 0.13749311864376068, "step": 120, "step_time": 81.79685344174504 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.126953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 322.17578125, "completions/mean_terminated_length": 294.5727233886719, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.2906140338163823, "epoch": 0.1379703534777651, "frac_reward_zero_std": 0.03125, "grad_norm": 0.02061237022280693, "kl": 0.007744725888187531, "learning_rate": 4.979733998344632e-06, "loss": 3.8692247471772134e-05, "num_tokens": 30480626.0, "reward": 1.9597656726837158, "reward_std": 0.7813615202903748, "rewards/code_complexity_reward/mean": 0.7027343511581421, "rewards/code_complexity_reward/std": 0.28500068187713623, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4384765625, "rewards/code_syntax_reward/std": 0.16440613567829132, "rewards/reasoning_present_reward_func/mean": 0.09003906697034836, "rewards/reasoning_present_reward_func/std": 0.029977135360240936, "rewards/xmlcount_reward_func/mean": 0.42578125, "rewards/xmlcount_reward_func/std": 0.1298590749502182, "step": 121, "step_time": 58.61201563011855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.064453125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 280.17578125, "completions/mean_terminated_length": 264.20458984375, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.2834449491929263, "epoch": 0.13911060433295325, "frac_reward_zero_std": 0.03125, "grad_norm": 0.02126830816268921, "kl": 0.009573314884619322, "learning_rate": 4.9784494335225e-06, "loss": 4.788313526660204e-05, "num_tokens": 30689968.0, "reward": 2.0772461891174316, "reward_std": 0.7384613752365112, "rewards/code_complexity_reward/mean": 0.7435547113418579, "rewards/code_complexity_reward/std": 0.2503177225589752, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.455078125, "rewards/code_syntax_reward/std": 0.1431187242269516, "rewards/reasoning_present_reward_func/mean": 0.09003905951976776, "rewards/reasoning_present_reward_func/std": 0.029977135360240936, "rewards/xmlcount_reward_func/mean": 0.43701171875, "rewards/xmlcount_reward_func/std": 0.12064225971698761, "step": 122, "step_time": 62.13264080323279 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 304.04296875, "completions/mean_terminated_length": 279.5240173339844, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.2844034265726805, "epoch": 0.1402508551881414, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.021338585764169693, "kl": 0.00921417379504419, "learning_rate": 4.977125574792018e-06, "loss": 4.6085333451628685e-05, "num_tokens": 30914990.0, "reward": 1.9935545921325684, "reward_std": 0.8216477632522583, "rewards/code_complexity_reward/mean": 0.6939453482627869, "rewards/code_complexity_reward/std": 0.29695528745651245, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.43359375, "rewards/code_syntax_reward/std": 0.16985194385051727, "rewards/reasoning_present_reward_func/mean": 0.09062500298023224, "rewards/reasoning_present_reward_func/std": 0.029176566749811172, "rewards/xmlcount_reward_func/mean": 0.42578125, "rewards/xmlcount_reward_func/std": 0.12796148657798767, "step": 123, "step_time": 63.13643799163401 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 291.60546875, "completions/mean_terminated_length": 278.8553466796875, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.29130961024202406, "epoch": 0.14139110604332952, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.02516956813633442, "kl": 0.00943957452182076, "learning_rate": 4.975762443141949e-06, "loss": 4.724979226011783e-05, "num_tokens": 31132316.0, "reward": 2.0867674350738525, "reward_std": 0.7658114433288574, "rewards/code_complexity_reward/mean": 0.7344726324081421, "rewards/code_complexity_reward/std": 0.2643311321735382, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4501953125, "rewards/code_syntax_reward/std": 0.14988566935062408, "rewards/reasoning_present_reward_func/mean": 0.091796875, "rewards/reasoning_present_reward_func/std": 0.02746807038784027, "rewards/xmlcount_reward_func/mean": 0.437255859375, "rewards/xmlcount_reward_func/std": 0.11730185151100159, "step": 124, "step_time": 54.08015895541757 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.091796875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 305.439453125, "completions/mean_terminated_length": 284.5613098144531, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.2822835217230022, "epoch": 0.14253135689851767, "frac_reward_zero_std": 0.0078125, "grad_norm": 0.0204856526106596, "kl": 0.009985148717532866, "learning_rate": 4.9743600601836974e-06, "loss": 4.9919821321964264e-05, "num_tokens": 31357837.0, "reward": 1.9825196266174316, "reward_std": 0.7853677272796631, "rewards/code_complexity_reward/mean": 0.699023425579071, "rewards/code_complexity_reward/std": 0.2881023585796356, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.435546875, "rewards/code_syntax_reward/std": 0.16771192848682404, "rewards/reasoning_present_reward_func/mean": 0.08964844048023224, "rewards/reasoning_present_reward_func/std": 0.030492907389998436, "rewards/xmlcount_reward_func/mean": 0.43017578125, "rewards/xmlcount_reward_func/std": 0.1226842999458313, "step": 125, "step_time": 84.12601159419864 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.072265625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 298.431640625, "completions/mean_terminated_length": 281.7957763671875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2859606989659369, "epoch": 0.14367160775370583, "frac_reward_zero_std": 0.03125, "grad_norm": 0.022951355203986168, "kl": 0.01024541422520997, "learning_rate": 4.9729184481509644e-06, "loss": 5.120855712448247e-05, "num_tokens": 31579382.0, "reward": 2.039794921875, "reward_std": 0.7752767205238342, "rewards/code_complexity_reward/mean": 0.7069335579872131, "rewards/code_complexity_reward/std": 0.2770684063434601, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.44140625, "rewards/code_syntax_reward/std": 0.16097907721996307, "rewards/reasoning_present_reward_func/mean": 0.09433593600988388, "rewards/reasoning_present_reward_func/std": 0.02313806861639023, "rewards/xmlcount_reward_func/mean": 0.447509765625, "rewards/xmlcount_reward_func/std": 0.10086417198181152, "step": 126, "step_time": 66.65049545001239 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.072265625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 288.935546875, "completions/mean_terminated_length": 271.55999755859375, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.28248275560326874, "epoch": 0.14481185860889395, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.020713413134217262, "kl": 0.010512836888665333, "learning_rate": 4.971437629899399e-06, "loss": 5.24992065038532e-05, "num_tokens": 31793457.0, "reward": 2.067187547683716, "reward_std": 0.7841195464134216, "rewards/code_complexity_reward/mean": 0.716015636920929, "rewards/code_complexity_reward/std": 0.27670010924339294, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4423828125, "rewards/code_syntax_reward/std": 0.1598084270954132, "rewards/reasoning_present_reward_func/mean": 0.09140624850988388, "rewards/reasoning_present_reward_func/std": 0.028054583817720413, "rewards/xmlcount_reward_func/mean": 0.4443359375, "rewards/xmlcount_reward_func/std": 0.10842516273260117, "step": 127, "step_time": 61.54929158370942 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 290.466796875, "completions/mean_terminated_length": 274.7091979980469, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.2875898233614862, "epoch": 0.1459521094640821, "frac_reward_zero_std": 0.046875, "grad_norm": 0.021001547574996948, "kl": 0.011194646445801482, "learning_rate": 4.969917628906234e-06, "loss": 5.6056815083138645e-05, "num_tokens": 32012216.0, "reward": 2.023193359375, "reward_std": 0.7245758771896362, "rewards/code_complexity_reward/mean": 0.724902331829071, "rewards/code_complexity_reward/std": 0.26000526547431946, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.451171875, "rewards/code_syntax_reward/std": 0.14856980741024017, "rewards/reasoning_present_reward_func/mean": 0.09296874701976776, "rewards/reasoning_present_reward_func/std": 0.025592297315597534, "rewards/xmlcount_reward_func/mean": 0.449462890625, "rewards/xmlcount_reward_func/std": 0.10334879904985428, "step": 128, "step_time": 78.17812159657478 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 281.619140625, "completions/mean_terminated_length": 265.2322082519531, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.2844657611567527, "epoch": 0.14709236031927023, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.020890671759843826, "kl": 0.009746594238094985, "learning_rate": 4.968358469269917e-06, "loss": 4.869022814091295e-05, "num_tokens": 32226253.0, "reward": 2.071240186691284, "reward_std": 0.7128207087516785, "rewards/code_complexity_reward/mean": 0.7409179210662842, "rewards/code_complexity_reward/std": 0.24656255543231964, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4580078125, "rewards/code_syntax_reward/std": 0.13881781697273254, "rewards/reasoning_present_reward_func/mean": 0.09375, "rewards/reasoning_present_reward_func/std": 0.02422981895506382, "rewards/xmlcount_reward_func/mean": 0.448486328125, "rewards/xmlcount_reward_func/std": 0.09768055379390717, "step": 129, "step_time": 73.33361427672207 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.072265625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 292.7578125, "completions/mean_terminated_length": 275.67999267578125, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.28250889433547854, "epoch": 0.14823261117445838, "frac_reward_zero_std": 0.03125, "grad_norm": 0.021436123177409172, "kl": 0.01172863908141153, "learning_rate": 4.966760175709725e-06, "loss": 5.8604724472388625e-05, "num_tokens": 32445113.0, "reward": 2.03515625, "reward_std": 0.7620344758033752, "rewards/code_complexity_reward/mean": 0.7191405892372131, "rewards/code_complexity_reward/std": 0.268402636051178, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4462890625, "rewards/code_syntax_reward/std": 0.15497584640979767, "rewards/reasoning_present_reward_func/mean": 0.09335938096046448, "rewards/reasoning_present_reward_func/std": 0.02492344006896019, "rewards/xmlcount_reward_func/mean": 0.4482421875, "rewards/xmlcount_reward_func/std": 0.1086716428399086, "step": 130, "step_time": 62.127475623972714 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09765625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 300.658203125, "completions/mean_terminated_length": 277.78570556640625, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.28410631814040244, "epoch": 0.14937286202964653, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.020353209227323532, "kl": 0.010642856650520116, "learning_rate": 4.96512277356537e-06, "loss": 5.3184747230261564e-05, "num_tokens": 32667762.0, "reward": 1.962988257408142, "reward_std": 0.7299805283546448, "rewards/code_complexity_reward/mean": 0.708984375, "rewards/code_complexity_reward/std": 0.2694986164569855, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4462890625, "rewards/code_syntax_reward/std": 0.15497584640979767, "rewards/reasoning_present_reward_func/mean": 0.09238281100988388, "rewards/reasoning_present_reward_func/std": 0.026553237810730934, "rewards/xmlcount_reward_func/mean": 0.44775390625, "rewards/xmlcount_reward_func/std": 0.0996193066239357, "step": 131, "step_time": 63.48883446957916 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.060546875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 282.046875, "completions/mean_terminated_length": 267.22662353515625, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.2814016700722277, "epoch": 0.15051311288483465, "frac_reward_zero_std": 0.015625, "grad_norm": 0.021404700353741646, "kl": 0.011711958359228447, "learning_rate": 4.963446288796605e-06, "loss": 5.852762842550874e-05, "num_tokens": 32880894.0, "reward": 2.173290967941284, "reward_std": 0.7143420577049255, "rewards/code_complexity_reward/mean": 0.7490234375, "rewards/code_complexity_reward/std": 0.2325495183467865, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.4658203125, "rewards/code_syntax_reward/std": 0.12630419433116913, "rewards/reasoning_present_reward_func/mean": 0.09492187201976776, "rewards/reasoning_present_reward_func/std": 0.021976543590426445, "rewards/xmlcount_reward_func/mean": 0.459228515625, "rewards/xmlcount_reward_func/std": 0.0854024663567543, "step": 132, "step_time": 86.69507542438805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 284.0234375, "completions/mean_terminated_length": 270.8346862792969, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.2861393690109253, "epoch": 0.1516533637400228, "frac_reward_zero_std": 0.0546875, "grad_norm": 0.022457338869571686, "kl": 0.012262642660061829, "learning_rate": 4.961730747982804e-06, "loss": 6.124388164607808e-05, "num_tokens": 33095318.0, "reward": 2.1089844703674316, "reward_std": 0.6264433860778809, "rewards/code_complexity_reward/mean": 0.7595703601837158, "rewards/code_complexity_reward/std": 0.20391523838043213, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4765625, "rewards/code_syntax_reward/std": 0.10578890144824982, "rewards/reasoning_present_reward_func/mean": 0.09550781548023224, "rewards/reasoning_present_reward_func/std": 0.020733514800667763, "rewards/xmlcount_reward_func/mean": 0.462890625, "rewards/xmlcount_reward_func/std": 0.08793312311172485, "step": 133, "step_time": 65.39511664677411 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 275.296875, "completions/mean_terminated_length": 259.5166931152344, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.26906505692750216, "epoch": 0.15279361459521096, "frac_reward_zero_std": 0.046875, "grad_norm": 0.021480005234479904, "kl": 0.012739602971123531, "learning_rate": 4.9599761783225465e-06, "loss": 6.377669342327863e-05, "num_tokens": 33303370.0, "reward": 2.1204590797424316, "reward_std": 0.7466363906860352, "rewards/code_complexity_reward/mean": 0.732421875, "rewards/code_complexity_reward/std": 0.2625957727432251, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4501953125, "rewards/code_syntax_reward/std": 0.14988566935062408, "rewards/reasoning_present_reward_func/mean": 0.09628906846046448, "rewards/reasoning_present_reward_func/std": 0.018921468406915665, "rewards/xmlcount_reward_func/mean": 0.464599609375, "rewards/xmlcount_reward_func/std": 0.07292790710926056, "step": 134, "step_time": 70.33186509739608 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.06640625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 284.404296875, "completions/mean_terminated_length": 268.2154846191406, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.2786919444333762, "epoch": 0.15393386545039908, "frac_reward_zero_std": 0.0390625, "grad_norm": 0.025443024933338165, "kl": 0.01163732291024644, "learning_rate": 4.9581826076331854e-06, "loss": 5.816286648041569e-05, "num_tokens": 33516713.0, "reward": 2.1105470657348633, "reward_std": 0.7111212015151978, "rewards/code_complexity_reward/mean": 0.749804675579071, "rewards/code_complexity_reward/std": 0.24079930782318115, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4599609375, "rewards/code_syntax_reward/std": 0.1358397752046585, "rewards/reasoning_present_reward_func/mean": 0.09414062649011612, "rewards/reasoning_present_reward_func/std": 0.023509247228503227, "rewards/xmlcount_reward_func/mean": 0.45703125, "rewards/xmlcount_reward_func/std": 0.08593263477087021, "step": 135, "step_time": 72.35289703216404 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 284.919921875, "completions/mean_terminated_length": 265.67584228515625, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.28163204714655876, "epoch": 0.15507411630558723, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.021894434466958046, "kl": 0.012818514500395395, "learning_rate": 4.956350064350403e-06, "loss": 6.412406219169497e-05, "num_tokens": 33731204.0, "reward": 2.0719728469848633, "reward_std": 0.687714159488678, "rewards/code_complexity_reward/mean": 0.740527331829071, "rewards/code_complexity_reward/std": 0.23736797273159027, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4619140625, "rewards/code_syntax_reward/std": 0.13276617228984833, "rewards/reasoning_present_reward_func/mean": 0.09609375149011612, "rewards/reasoning_present_reward_func/std": 0.0193933192640543, "rewards/xmlcount_reward_func/mean": 0.458984375, "rewards/xmlcount_reward_func/std": 0.08653101325035095, "step": 136, "step_time": 78.34991680365056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 274.7109375, "completions/mean_terminated_length": 263.04095458984375, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.28360952832736075, "epoch": 0.15621436716077536, "frac_reward_zero_std": 0.03125, "grad_norm": 0.02156275138258934, "kl": 0.014608606776164379, "learning_rate": 4.954478577527761e-06, "loss": 7.302610902115703e-05, "num_tokens": 33939040.0, "reward": 2.172900438308716, "reward_std": 0.6916077136993408, "rewards/code_complexity_reward/mean": 0.7632812857627869, "rewards/code_complexity_reward/std": 0.23149599134922028, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.466796875, "rewards/code_syntax_reward/std": 0.12461719661951065, "rewards/reasoning_present_reward_func/mean": 0.09589843451976776, "rewards/reasoning_present_reward_func/std": 0.019852032884955406, "rewards/xmlcount_reward_func/mean": 0.464111328125, "rewards/xmlcount_reward_func/std": 0.08473130315542221, "step": 137, "step_time": 81.26297878660262 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 264.2265625, "completions/mean_terminated_length": 245.48741149902344, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2653208745177835, "epoch": 0.1573546180159635, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.02425193041563034, "kl": 0.013522978362743743, "learning_rate": 4.952568176836246e-06, "loss": 6.760028190910816e-05, "num_tokens": 34144456.0, "reward": 2.188037157058716, "reward_std": 0.700356125831604, "rewards/code_complexity_reward/mean": 0.7646484971046448, "rewards/code_complexity_reward/std": 0.23058733344078064, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.46875, "rewards/code_syntax_reward/std": 0.12114909291267395, "rewards/reasoning_present_reward_func/mean": 0.09648437798023224, "rewards/reasoning_present_reward_func/std": 0.01843547262251377, "rewards/xmlcount_reward_func/mean": 0.469482421875, "rewards/xmlcount_reward_func/std": 0.07134831696748734, "step": 138, "step_time": 72.39069515559822 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 275.048828125, "completions/mean_terminated_length": 264.4101867675781, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.2741988995112479, "epoch": 0.15849486887115166, "frac_reward_zero_std": 0.0234375, "grad_norm": 0.021697549149394035, "kl": 0.01608598781604087, "learning_rate": 4.9506188925637885e-06, "loss": 8.040119428187609e-05, "num_tokens": 34355377.0, "reward": 2.0613279342651367, "reward_std": 0.683155357837677, "rewards/code_complexity_reward/mean": 0.7291015386581421, "rewards/code_complexity_reward/std": 0.2376556396484375, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4619140625, "rewards/code_syntax_reward/std": 0.13276617228984833, "rewards/reasoning_present_reward_func/mean": 0.09687499701976776, "rewards/reasoning_present_reward_func/std": 0.017416279762983322, "rewards/xmlcount_reward_func/mean": 0.466796875, "rewards/xmlcount_reward_func/std": 0.08124576508998871, "step": 139, "step_time": 148.3157729106024 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.07421875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 276.259765625, "completions/mean_terminated_length": 257.3607482910156, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.28238724591210485, "epoch": 0.15963511972633979, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.020494777709245682, "kl": 0.01285980283137178, "learning_rate": 4.948630755614792e-06, "loss": 6.42605300527066e-05, "num_tokens": 34563094.0, "reward": 2.092529296875, "reward_std": 0.7102833986282349, "rewards/code_complexity_reward/mean": 0.738574206829071, "rewards/code_complexity_reward/std": 0.24694080650806427, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4599609375, "rewards/code_syntax_reward/std": 0.1358397752046585, "rewards/reasoning_present_reward_func/mean": 0.09687499701976776, "rewards/reasoning_present_reward_func/std": 0.01741628162562847, "rewards/xmlcount_reward_func/mean": 0.470947265625, "rewards/xmlcount_reward_func/std": 0.07322212308645248, "step": 140, "step_time": 68.38871859014034 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.056640625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 274.0, "completions/mean_terminated_length": 259.71014404296875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.27702692546881735, "epoch": 0.16077537058152794, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.01925739273428917, "kl": 0.013723489028052427, "learning_rate": 4.946603797509635e-06, "loss": 6.861129077151418e-05, "num_tokens": 34771526.0, "reward": 2.1199707984924316, "reward_std": 0.6962177157402039, "rewards/code_complexity_reward/mean": 0.7430664300918579, "rewards/code_complexity_reward/std": 0.23557353019714355, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4638671875, "rewards/code_syntax_reward/std": 0.1295902281999588, "rewards/reasoning_present_reward_func/mean": 0.09687499701976776, "rewards/reasoning_present_reward_func/std": 0.01741628162562847, "rewards/xmlcount_reward_func/mean": 0.470458984375, "rewards/xmlcount_reward_func/std": 0.07828040421009064, "step": 141, "step_time": 61.46626026183367 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.068359375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 277.04296875, "completions/mean_terminated_length": 259.80291748046875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2758798776194453, "epoch": 0.1619156214367161, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.020177066326141357, "kl": 0.013707508063816931, "learning_rate": 4.944538050384181e-06, "loss": 6.847757322248071e-05, "num_tokens": 34982760.0, "reward": 2.049560546875, "reward_std": 0.6659480333328247, "rewards/code_complexity_reward/mean": 0.7394531965255737, "rewards/code_complexity_reward/std": 0.23325756192207336, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.462890625, "rewards/code_syntax_reward/std": 0.1311914473772049, "rewards/reasoning_present_reward_func/mean": 0.09648437798023224, "rewards/reasoning_present_reward_func/std": 0.01843547262251377, "rewards/xmlcount_reward_func/mean": 0.469482421875, "rewards/xmlcount_reward_func/std": 0.07428772002458572, "step": 142, "step_time": 72.61229258589447 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.044921875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 262.744140625, "completions/mean_terminated_length": 251.02044677734375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2728367540985346, "epoch": 0.1630558722919042, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.02351108007133007, "kl": 0.016833383589982986, "learning_rate": 4.94243354698926e-06, "loss": 8.41415167087689e-05, "num_tokens": 35185349.0, "reward": 2.148388624191284, "reward_std": 0.6851129531860352, "rewards/code_complexity_reward/mean": 0.750292956829071, "rewards/code_complexity_reward/std": 0.2285727709531784, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.466796875, "rewards/code_syntax_reward/std": 0.12461719661951065, "rewards/reasoning_present_reward_func/mean": 0.09707030653953552, "rewards/reasoning_present_reward_func/std": 0.016880230978131294, "rewards/xmlcount_reward_func/mean": 0.474853515625, "rewards/xmlcount_reward_func/std": 0.06595844775438309, "step": 143, "step_time": 60.1153780631721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05859375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 262.90625, "completions/mean_terminated_length": 247.40249633789062, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.27336200210265815, "epoch": 0.16419612314709237, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.021022651344537735, "kl": 0.015011078750831075, "learning_rate": 4.9402903206901535e-06, "loss": 7.500709034502506e-05, "num_tokens": 35389393.0, "reward": 2.185302734375, "reward_std": 0.693938672542572, "rewards/code_complexity_reward/mean": 0.771289050579071, "rewards/code_complexity_reward/std": 0.22247913479804993, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.46875, "rewards/code_syntax_reward/std": 0.12114909291267395, "rewards/reasoning_present_reward_func/mean": 0.09687499701976776, "rewards/reasoning_present_reward_func/std": 0.017416279762983322, "rewards/xmlcount_reward_func/mean": 0.471435546875, "rewards/xmlcount_reward_func/std": 0.07786120474338531, "step": 144, "step_time": 73.6816528858617 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.048828125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 252.953125, "completions/mean_terminated_length": 239.65504455566406, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.27105632051825523, "epoch": 0.1653363740022805, "frac_reward_zero_std": 0.0625, "grad_norm": 0.02023329585790634, "kl": 0.017899181621032767, "learning_rate": 4.938108405466065e-06, "loss": 8.950225310400128e-05, "num_tokens": 35587361.0, "reward": 2.1859374046325684, "reward_std": 0.6667989492416382, "rewards/code_complexity_reward/mean": 0.7748047113418579, "rewards/code_complexity_reward/std": 0.21916484832763672, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.47265625, "rewards/code_syntax_reward/std": 0.11379580944776535, "rewards/reasoning_present_reward_func/mean": 0.0966796875, "rewards/reasoning_present_reward_func/std": 0.01793418452143669, "rewards/xmlcount_reward_func/mean": 0.474609375, "rewards/xmlcount_reward_func/std": 0.06421925872564316, "step": 145, "step_time": 65.82119208853692 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.041015625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 262.38671875, "completions/mean_terminated_length": 251.71080017089844, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.26399655058048666, "epoch": 0.16647662485746864, "frac_reward_zero_std": 0.03125, "grad_norm": 0.020080355927348137, "kl": 0.019243395829107612, "learning_rate": 4.935887835909581e-06, "loss": 9.634160960558802e-05, "num_tokens": 35791075.0, "reward": 2.174023389816284, "reward_std": 0.6573470830917358, "rewards/code_complexity_reward/mean": 0.7598632574081421, "rewards/code_complexity_reward/std": 0.20994216203689575, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4755859375, "rewards/code_syntax_reward/std": 0.10785966366529465, "rewards/reasoning_present_reward_func/mean": 0.09628906100988388, "rewards/reasoning_present_reward_func/std": 0.018921468406915665, "rewards/xmlcount_reward_func/mean": 0.47314453125, "rewards/xmlcount_reward_func/std": 0.07385111600160599, "step": 146, "step_time": 78.70800665672868 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04296875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 247.1640625, "completions/mean_terminated_length": 235.27345275878906, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.2568641162943095, "epoch": 0.1676168757126568, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.02062005177140236, "kl": 0.021548846605583094, "learning_rate": 4.933628647226123e-06, "loss": 0.00010779248259495944, "num_tokens": 35987327.0, "reward": 2.169921875, "reward_std": 0.6345897316932678, "rewards/code_complexity_reward/mean": 0.7666991949081421, "rewards/code_complexity_reward/std": 0.20576201379299164, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4755859375, "rewards/code_syntax_reward/std": 0.10785966366529465, "rewards/reasoning_present_reward_func/mean": 0.09804686903953552, "rewards/reasoning_present_reward_func/std": 0.013851807452738285, "rewards/xmlcount_reward_func/mean": 0.47607421875, "rewards/xmlcount_reward_func/std": 0.06382935494184494, "step": 147, "step_time": 86.13409539498389 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 267.486328125, "completions/mean_terminated_length": 255.46104431152344, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2779024199116975, "epoch": 0.16875712656784492, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.02130192704498768, "kl": 0.01583571576338727, "learning_rate": 4.93133087523339e-06, "loss": 7.932081643957645e-05, "num_tokens": 36191092.0, "reward": 2.2147459983825684, "reward_std": 0.6698404550552368, "rewards/code_complexity_reward/mean": 0.7583984136581421, "rewards/code_complexity_reward/std": 0.20734499394893646, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.4765625, "rewards/code_syntax_reward/std": 0.10578890144824982, "rewards/reasoning_present_reward_func/mean": 0.09843750298023224, "rewards/reasoning_present_reward_func/std": 0.012414088472723961, "rewards/xmlcount_reward_func/mean": 0.47900390625, "rewards/xmlcount_reward_func/std": 0.05681177228689194, "step": 148, "step_time": 71.80060283467174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.037109375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 252.7265625, "completions/mean_terminated_length": 242.73426818847656, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.26482702046632767, "epoch": 0.16989737742303307, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.02259456366300583, "kl": 0.021165423400816508, "learning_rate": 4.928994556360787e-06, "loss": 0.00010586907592369244, "num_tokens": 36387080.0, "reward": 2.142383098602295, "reward_std": 0.6679869294166565, "rewards/code_complexity_reward/mean": 0.7606445550918579, "rewards/code_complexity_reward/std": 0.22404171526432037, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.46875, "rewards/code_syntax_reward/std": 0.12114909291267395, "rewards/reasoning_present_reward_func/mean": 0.09804688394069672, "rewards/reasoning_present_reward_func/std": 0.013851807452738285, "rewards/xmlcount_reward_func/mean": 0.47900390625, "rewards/xmlcount_reward_func/std": 0.06146518141031265, "step": 149, "step_time": 63.88996078167111 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 244.99609375, "completions/mean_terminated_length": 236.383056640625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.26874064444564283, "epoch": 0.17103762827822122, "frac_reward_zero_std": 0.0546875, "grad_norm": 0.023173091933131218, "kl": 0.01932570982899051, "learning_rate": 4.926619727648852e-06, "loss": 9.659535135142505e-05, "num_tokens": 36580234.0, "reward": 2.192138671875, "reward_std": 0.6389685869216919, "rewards/code_complexity_reward/mean": 0.776562511920929, "rewards/code_complexity_reward/std": 0.20040930807590485, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.09843750298023224, "rewards/reasoning_present_reward_func/std": 0.012414088472723961, "rewards/xmlcount_reward_func/mean": 0.482177734375, "rewards/xmlcount_reward_func/std": 0.05709018558263779, "step": 150, "step_time": 63.06031093001366 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 268.673828125, "completions/mean_terminated_length": 254.5970916748047, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.26679482916370034, "epoch": 0.17217787913340935, "frac_reward_zero_std": 0.078125, "grad_norm": 0.021030332893133163, "kl": 0.017948284323210828, "learning_rate": 4.924206426748668e-06, "loss": 8.969189366325736e-05, "num_tokens": 36786999.0, "reward": 2.0974607467651367, "reward_std": 0.665111780166626, "rewards/code_complexity_reward/mean": 0.7383788824081421, "rewards/code_complexity_reward/std": 0.2285338044166565, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4677734375, "rewards/code_syntax_reward/std": 0.12289927154779434, "rewards/reasoning_present_reward_func/mean": 0.09687500447034836, "rewards/reasoning_present_reward_func/std": 0.017416279762983322, "rewards/xmlcount_reward_func/mean": 0.47607421875, "rewards/xmlcount_reward_func/std": 0.06334849447011948, "step": 151, "step_time": 61.00937512423843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 244.373046875, "completions/mean_terminated_length": 231.2110595703125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.26085855532437563, "epoch": 0.1733181299885975, "frac_reward_zero_std": 0.078125, "grad_norm": 0.021842850372195244, "kl": 0.019680753917782567, "learning_rate": 4.921754691921262e-06, "loss": 9.847053297562525e-05, "num_tokens": 36979090.0, "reward": 2.1809568405151367, "reward_std": 0.6109095811843872, "rewards/code_complexity_reward/mean": 0.7849609851837158, "rewards/code_complexity_reward/std": 0.1916515827178955, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09765625, "rewards/reasoning_present_reward_func/std": 0.015143636614084244, "rewards/xmlcount_reward_func/mean": 0.48388671875, "rewards/xmlcount_reward_func/std": 0.05231089144945145, "step": 152, "step_time": 60.78885213378817 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.037109375, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 237.021484375, "completions/mean_terminated_length": 226.42393493652344, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.26378359040245414, "epoch": 0.17445838084378562, "frac_reward_zero_std": 0.109375, "grad_norm": 0.024635186418890953, "kl": 0.017921304519404657, "learning_rate": 4.919264562037003e-06, "loss": 8.95819321158342e-05, "num_tokens": 37167041.0, "reward": 2.2046875953674316, "reward_std": 0.6340065002441406, "rewards/code_complexity_reward/mean": 0.7780272960662842, "rewards/code_complexity_reward/std": 0.19492235779762268, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09804687649011612, "rewards/reasoning_present_reward_func/std": 0.013851807452738285, "rewards/xmlcount_reward_func/mean": 0.48291015625, "rewards/xmlcount_reward_func/std": 0.055964481085538864, "step": 153, "step_time": 81.72879144176841 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025390625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 237.662109375, "completions/mean_terminated_length": 230.51502990722656, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.24909522687084973, "epoch": 0.17559863169897377, "frac_reward_zero_std": 0.0625, "grad_norm": 0.024518851190805435, "kl": 0.02422236251004506, "learning_rate": 4.9167360765749845e-06, "loss": 0.00012121675536036491, "num_tokens": 37357416.0, "reward": 2.2070798873901367, "reward_std": 0.6029646992683411, "rewards/code_complexity_reward/mean": 0.7764648199081421, "rewards/code_complexity_reward/std": 0.17345291376113892, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09785155951976776, "rewards/reasoning_present_reward_func/std": 0.014513419941067696, "rewards/xmlcount_reward_func/mean": 0.480224609375, "rewards/xmlcount_reward_func/std": 0.07256010919809341, "step": 154, "step_time": 78.11024989560246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.044921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 254.6328125, "completions/mean_terminated_length": 242.52760314941406, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.25466923718340695, "epoch": 0.17673888255416192, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.019911188632249832, "kl": 0.018248447915539145, "learning_rate": 4.914169275622397e-06, "loss": 9.135452273767442e-05, "num_tokens": 37556536.0, "reward": 2.1670899391174316, "reward_std": 0.6338876485824585, "rewards/code_complexity_reward/mean": 0.7607421875, "rewards/code_complexity_reward/std": 0.19610151648521423, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09824219346046448, "rewards/reasoning_present_reward_func/std": 0.013154060579836369, "rewards/xmlcount_reward_func/mean": 0.48583984375, "rewards/xmlcount_reward_func/std": 0.04802536964416504, "step": 155, "step_time": 59.76682709157467 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 245.2890625, "completions/mean_terminated_length": 235.57086181640625, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.268104714108631, "epoch": 0.17787913340935005, "frac_reward_zero_std": 0.046875, "grad_norm": 0.019911348819732666, "kl": 0.021182777563808486, "learning_rate": 4.911564199873894e-06, "loss": 0.00010594655759632587, "num_tokens": 37750568.0, "reward": 2.164306640625, "reward_std": 0.6242797374725342, "rewards/code_complexity_reward/mean": 0.7677733898162842, "rewards/code_complexity_reward/std": 0.19398115575313568, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.09843750298023224, "rewards/reasoning_present_reward_func/std": 0.012414088472723961, "rewards/xmlcount_reward_func/mean": 0.484619140625, "rewards/xmlcount_reward_func/std": 0.05726565793156624, "step": 156, "step_time": 74.12648445088416 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 237.5625, "completions/mean_terminated_length": 227.56275939941406, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2576041684951633, "epoch": 0.1790193842645382, "frac_reward_zero_std": 0.15625, "grad_norm": 0.022376835346221924, "kl": 0.025888340634992346, "learning_rate": 4.908920890630947e-06, "loss": 0.00012931902892887592, "num_tokens": 37938036.0, "reward": 2.233935594558716, "reward_std": 0.6366897225379944, "rewards/code_complexity_reward/mean": 0.7842773199081421, "rewards/code_complexity_reward/std": 0.19432221353054047, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.09785155951976776, "rewards/reasoning_present_reward_func/std": 0.014513419941067696, "rewards/xmlcount_reward_func/mean": 0.484619140625, "rewards/xmlcount_reward_func/std": 0.056729190051555634, "step": 157, "step_time": 68.47330464795232 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 244.751953125, "completions/mean_terminated_length": 237.23895263671875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2551751264836639, "epoch": 0.18015963511972635, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.023940252140164375, "kl": 0.03172882227227092, "learning_rate": 4.906239389801191e-06, "loss": 0.0001586630824021995, "num_tokens": 38132169.0, "reward": 2.1697754859924316, "reward_std": 0.6142755746841431, "rewards/code_complexity_reward/mean": 0.7752929329872131, "rewards/code_complexity_reward/std": 0.18765707314014435, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09882812947034836, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.484130859375, "rewards/xmlcount_reward_func/std": 0.055503182113170624, "step": 158, "step_time": 73.50936716981232 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 247.6015625, "completions/mean_terminated_length": 239.07257080078125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2575767361558974, "epoch": 0.18129988597491448, "frac_reward_zero_std": 0.09375, "grad_norm": 0.02242562733590603, "kl": 0.023258438828634098, "learning_rate": 4.903519739897755e-06, "loss": 0.00011627969797700644, "num_tokens": 38325669.0, "reward": 2.2418456077575684, "reward_std": 0.6485801339149475, "rewards/code_complexity_reward/mean": 0.7671874761581421, "rewards/code_complexity_reward/std": 0.1927184760570526, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.09843750298023224, "rewards/reasoning_present_reward_func/std": 0.012414088472723961, "rewards/xmlcount_reward_func/mean": 0.484619140625, "rewards/xmlcount_reward_func/std": 0.05832379311323166, "step": 159, "step_time": 73.50216684956104 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.044921875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 251.337890625, "completions/mean_terminated_length": 239.07769775390625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.25974873336963356, "epoch": 0.18244013683010263, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.023278098553419113, "kl": 0.027194289112230763, "learning_rate": 4.9007619840385975e-06, "loss": 0.00013604722335003316, "num_tokens": 38522866.0, "reward": 2.152587890625, "reward_std": 0.6321378350257874, "rewards/code_complexity_reward/mean": 0.76806640625, "rewards/code_complexity_reward/std": 0.20085620880126953, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4765625, "rewards/code_syntax_reward/std": 0.10578890144824982, "rewards/reasoning_present_reward_func/mean": 0.0986328125, "rewards/reasoning_present_reward_func/std": 0.011623830534517765, "rewards/xmlcount_reward_func/mean": 0.483154296875, "rewards/xmlcount_reward_func/std": 0.052952565252780914, "step": 160, "step_time": 81.75199691113085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.029296875, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 244.0, "completions/mean_terminated_length": 235.9114532470703, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2663580805528909, "epoch": 0.18358038768529075, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.020101329311728477, "kl": 0.021397848075139336, "learning_rate": 4.897966165945815e-06, "loss": 0.00010697123070713133, "num_tokens": 38715434.0, "reward": 2.1741700172424316, "reward_std": 0.6428482532501221, "rewards/code_complexity_reward/mean": 0.7694336175918579, "rewards/code_complexity_reward/std": 0.20654238760471344, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4736328125, "rewards/code_syntax_reward/std": 0.11186064779758453, "rewards/reasoning_present_reward_func/mean": 0.09785155951976776, "rewards/reasoning_present_reward_func/std": 0.014513419941067696, "rewards/xmlcount_reward_func/mean": 0.481689453125, "rewards/xmlcount_reward_func/std": 0.05955992639064789, "step": 161, "step_time": 79.03857815265656 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.033203125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 231.755859375, "completions/mean_terminated_length": 222.13131713867188, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.25718099693767726, "epoch": 0.1847206385404789, "frac_reward_zero_std": 0.0625, "grad_norm": 0.023063763976097107, "kl": 0.02652221792959608, "learning_rate": 4.8951323299449514e-06, "loss": 0.00013273909280542284, "num_tokens": 38901073.0, "reward": 2.1898927688598633, "reward_std": 0.6649136543273926, "rewards/code_complexity_reward/mean": 0.762499988079071, "rewards/code_complexity_reward/std": 0.21818052232265472, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4716796875, "rewards/code_syntax_reward/std": 0.11569035053253174, "rewards/reasoning_present_reward_func/mean": 0.09804687649011612, "rewards/reasoning_present_reward_func/std": 0.013851807452738285, "rewards/xmlcount_reward_func/mean": 0.484619140625, "rewards/xmlcount_reward_func/std": 0.060888733714818954, "step": 162, "step_time": 60.709455110132694 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 246.5625, "completions/mean_terminated_length": 239.1003875732422, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.259624685626477, "epoch": 0.18586088939566706, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.02234356850385666, "kl": 0.02286307113536168, "learning_rate": 4.892260520964295e-06, "loss": 0.00011425349657656625, "num_tokens": 39095793.0, "reward": 2.1360349655151367, "reward_std": 0.6139166355133057, "rewards/code_complexity_reward/mean": 0.7673828601837158, "rewards/code_complexity_reward/std": 0.1971186399459839, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.0986328125, "rewards/reasoning_present_reward_func/std": 0.011623830534517765, "rewards/xmlcount_reward_func/mean": 0.48876953125, "rewards/xmlcount_reward_func/std": 0.05183378607034683, "step": 163, "step_time": 58.073305807076395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 228.826171875, "completions/mean_terminated_length": 223.18527221679688, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.25624088244512677, "epoch": 0.18700114025085518, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.0208889190107584, "kl": 0.021797661102027632, "learning_rate": 4.889350784534168e-06, "loss": 0.0001088301942218095, "num_tokens": 39280596.0, "reward": 2.1885743141174316, "reward_std": 0.6017292737960815, "rewards/code_complexity_reward/mean": 0.7805664539337158, "rewards/code_complexity_reward/std": 0.18817350268363953, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.490234375, "rewards/xmlcount_reward_func/std": 0.04242921620607376, "step": 164, "step_time": 83.59937905147672 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.033203125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 239.525390625, "completions/mean_terminated_length": 230.16769409179688, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.25232275179587305, "epoch": 0.18814139110604333, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.021369775757193565, "kl": 0.02114885376067832, "learning_rate": 4.886403166786203e-06, "loss": 0.00010568702418822795, "num_tokens": 39472357.0, "reward": 2.213623046875, "reward_std": 0.6495924592018127, "rewards/code_complexity_reward/mean": 0.7673828601837158, "rewards/code_complexity_reward/std": 0.19617323577404022, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09785155951976776, "rewards/reasoning_present_reward_func/std": 0.014513419941067696, "rewards/xmlcount_reward_func/mean": 0.487060546875, "rewards/xmlcount_reward_func/std": 0.04992462322115898, "step": 165, "step_time": 60.1791763799265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 229.951171875, "completions/mean_terminated_length": 220.85281372070312, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.24167260387912393, "epoch": 0.18928164196123148, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.023746151477098465, "kl": 0.022565684645087458, "learning_rate": 4.883417714452607e-06, "loss": 0.00011278275633230805, "num_tokens": 39656564.0, "reward": 2.275195360183716, "reward_std": 0.6398002505302429, "rewards/code_complexity_reward/mean": 0.783398449420929, "rewards/code_complexity_reward/std": 0.1883021593093872, "rewards/code_execution_reward/mean": 0.41796875, "rewards/code_execution_reward/std": 0.4937073290348053, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.490234375, "rewards/xmlcount_reward_func/std": 0.04170232638716698, "step": 166, "step_time": 68.66317170485854 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 235.25, "completions/mean_terminated_length": 227.46986389160156, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.25473544490523636, "epoch": 0.1904218928164196, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.022675827145576477, "kl": 0.027893679158296436, "learning_rate": 4.880394474865433e-06, "loss": 0.00013952417066320777, "num_tokens": 39845272.0, "reward": 2.2098634243011475, "reward_std": 0.6045027375221252, "rewards/code_complexity_reward/mean": 0.7782226800918579, "rewards/code_complexity_reward/std": 0.18715964257717133, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.0986328125, "rewards/reasoning_present_reward_func/std": 0.011623830534517765, "rewards/xmlcount_reward_func/mean": 0.4873046875, "rewards/xmlcount_reward_func/std": 0.05208435282111168, "step": 167, "step_time": 79.18051070347428 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 243.708984375, "completions/mean_terminated_length": 236.16665649414062, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.259600767865777, "epoch": 0.19156214367160776, "frac_reward_zero_std": 0.09375, "grad_norm": 0.022515226155519485, "kl": 0.0270616806083126, "learning_rate": 4.8773334959558165e-06, "loss": 0.0001351018581772223, "num_tokens": 40039107.0, "reward": 2.190185546875, "reward_std": 0.6340104937553406, "rewards/code_complexity_reward/mean": 0.7678711414337158, "rewards/code_complexity_reward/std": 0.2010684758424759, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.09882812201976776, "rewards/reasoning_present_reward_func/std": 0.010772227309644222, "rewards/xmlcount_reward_func/mean": 0.486572265625, "rewards/xmlcount_reward_func/std": 0.0539226308465004, "step": 168, "step_time": 55.98768369946629 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 230.88671875, "completions/mean_terminated_length": 222.9839324951172, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.23944882955402136, "epoch": 0.19270239452679588, "frac_reward_zero_std": 0.1015625, "grad_norm": 0.02290533296763897, "kl": 0.027080842686700635, "learning_rate": 4.874234826253223e-06, "loss": 0.00013530784053727984, "num_tokens": 40227341.0, "reward": 2.2442383766174316, "reward_std": 0.6276808977127075, "rewards/code_complexity_reward/mean": 0.7754882574081421, "rewards/code_complexity_reward/std": 0.19106702506542206, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.0986328125, "rewards/reasoning_present_reward_func/std": 0.011623830534517765, "rewards/xmlcount_reward_func/mean": 0.490234375, "rewards/xmlcount_reward_func/std": 0.0431438647210598, "step": 169, "step_time": 64.51831744238734 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 230.919921875, "completions/mean_terminated_length": 223.01806640625, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.24180085049010813, "epoch": 0.19384264538198404, "frac_reward_zero_std": 0.140625, "grad_norm": 0.022209497168660164, "kl": 0.023022276785923168, "learning_rate": 4.871098514884675e-06, "loss": 0.00011502111738082021, "num_tokens": 40412316.0, "reward": 2.1861815452575684, "reward_std": 0.581021785736084, "rewards/code_complexity_reward/mean": 0.7725585699081421, "rewards/code_complexity_reward/std": 0.17394888401031494, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.489990234375, "rewards/xmlcount_reward_func/std": 0.04127552732825279, "step": 170, "step_time": 69.7029897402972 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 216.0859375, "completions/mean_terminated_length": 206.54031372070312, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.25427537481300533, "epoch": 0.1949828962371722, "frac_reward_zero_std": 0.109375, "grad_norm": 0.02193780243396759, "kl": 0.026694025975302793, "learning_rate": 4.867924611573977e-06, "loss": 0.0001334451953880489, "num_tokens": 40589312.0, "reward": 2.2439451217651367, "reward_std": 0.6348354816436768, "rewards/code_complexity_reward/mean": 0.7886718511581421, "rewards/code_complexity_reward/std": 0.18455159664154053, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09882812947034836, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.4912109375, "rewards/xmlcount_reward_func/std": 0.04608868435025215, "step": 171, "step_time": 61.84673099312931 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 222.896484375, "completions/mean_terminated_length": 218.30755615234375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.24554147780872881, "epoch": 0.1961231470923603, "frac_reward_zero_std": 0.09375, "grad_norm": 0.02146073803305626, "kl": 0.03090911945037078, "learning_rate": 4.864713166640921e-06, "loss": 0.00015454748063348234, "num_tokens": 40771375.0, "reward": 2.2533202171325684, "reward_std": 0.578116238117218, "rewards/code_complexity_reward/mean": 0.7977539300918579, "rewards/code_complexity_reward/std": 0.15733791887760162, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.48974609375, "rewards/xmlcount_reward_func/std": 0.05377012863755226, "step": 172, "step_time": 65.42159445118159 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 218.919921875, "completions/mean_terminated_length": 215.44467163085938, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.23598207952454686, "epoch": 0.19726339794754846, "frac_reward_zero_std": 0.078125, "grad_norm": 0.021027235314249992, "kl": 0.025800550793064758, "learning_rate": 4.8614642310004975e-06, "loss": 0.0001289524370804429, "num_tokens": 40952190.0, "reward": 2.3238282203674316, "reward_std": 0.6058433055877686, "rewards/code_complexity_reward/mean": 0.7883788347244263, "rewards/code_complexity_reward/std": 0.16925273835659027, "rewards/code_execution_reward/mean": 0.45703125, "rewards/code_execution_reward/std": 0.49863746762275696, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49169921875, "rewards/xmlcount_reward_func/std": 0.04414820298552513, "step": 173, "step_time": 71.92668642755598 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 213.466796875, "completions/mean_terminated_length": 208.125244140625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23444742639549077, "epoch": 0.19840364880273662, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.022063003852963448, "kl": 0.034713357366854325, "learning_rate": 4.8581778561620785e-06, "loss": 0.00017340423073619604, "num_tokens": 41129129.0, "reward": 2.3097167015075684, "reward_std": 0.637620747089386, "rewards/code_complexity_reward/mean": 0.7842773199081421, "rewards/code_complexity_reward/std": 0.18172866106033325, "rewards/code_execution_reward/mean": 0.453125, "rewards/code_execution_reward/std": 0.4982847273349762, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09843750298023224, "rewards/reasoning_present_reward_func/std": 0.012414087541401386, "rewards/xmlcount_reward_func/mean": 0.490478515625, "rewards/xmlcount_reward_func/std": 0.048845935612916946, "step": 174, "step_time": 61.965516679920256 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 220.25, "completions/mean_terminated_length": 212.04818725585938, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.24229028192348778, "epoch": 0.19954389965792474, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.02401071973145008, "kl": 0.02905059850309044, "learning_rate": 4.8548540942286095e-06, "loss": 0.00014515973452944309, "num_tokens": 41309121.0, "reward": 2.220703125, "reward_std": 0.5689762234687805, "rewards/code_complexity_reward/mean": 0.7890625, "rewards/code_complexity_reward/std": 0.16283178329467773, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4921875, "rewards/xmlcount_reward_func/std": 0.03750407695770264, "step": 175, "step_time": 59.70536960568279 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025390625, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 214.580078125, "completions/mean_terminated_length": 206.8316650390625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.24896568967960775, "epoch": 0.2006841505131129, "frac_reward_zero_std": 0.109375, "grad_norm": 0.02248302660882473, "kl": 0.028703694319119677, "learning_rate": 4.851492997895777e-06, "loss": 0.00014354989980347455, "num_tokens": 41487534.0, "reward": 2.2606444358825684, "reward_std": 0.6122996807098389, "rewards/code_complexity_reward/mean": 0.7964843511581421, "rewards/code_complexity_reward/std": 0.1782095730304718, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.49169921875, "rewards/xmlcount_reward_func/std": 0.04617929831147194, "step": 176, "step_time": 63.29000342451036 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.033203125, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 223.30859375, "completions/mean_terminated_length": 213.39395141601562, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.24505510181188583, "epoch": 0.20182440136830102, "frac_reward_zero_std": 0.125, "grad_norm": 0.022547319531440735, "kl": 0.03402118134545162, "learning_rate": 4.848094620451177e-06, "loss": 0.0001699854910839349, "num_tokens": 41670372.0, "reward": 2.2298340797424316, "reward_std": 0.6343843936920166, "rewards/code_complexity_reward/mean": 0.77880859375, "rewards/code_complexity_reward/std": 0.20111028850078583, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.489501953125, "rewards/xmlcount_reward_func/std": 0.051692452281713486, "step": 177, "step_time": 62.23196988273412 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 219.111328125, "completions/mean_terminated_length": 212.68063354492188, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2394577208906412, "epoch": 0.20296465222348917, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.023593472316861153, "kl": 0.031169467460131273, "learning_rate": 4.844659015773468e-06, "loss": 0.00015577883459627628, "num_tokens": 41848753.0, "reward": 2.2144532203674316, "reward_std": 0.625930666923523, "rewards/code_complexity_reward/mean": 0.78173828125, "rewards/code_complexity_reward/std": 0.1936453878879547, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.09921874850988388, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49365234375, "rewards/xmlcount_reward_func/std": 0.04013480618596077, "step": 178, "step_time": 72.41373723652214 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03515625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 227.041015625, "completions/mean_terminated_length": 216.65789794921875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.234333376865834, "epoch": 0.20410490307867732, "frac_reward_zero_std": 0.09375, "grad_norm": 0.025098547339439392, "kl": 0.02502976093092002, "learning_rate": 4.841186238331519e-06, "loss": 0.00012503174366429448, "num_tokens": 42034018.0, "reward": 2.17138671875, "reward_std": 0.6300192475318909, "rewards/code_complexity_reward/mean": 0.760937511920929, "rewards/code_complexity_reward/std": 0.20594753324985504, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49169921875, "rewards/xmlcount_reward_func/std": 0.03739883005619049, "step": 179, "step_time": 74.88143420685083 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 218.09375, "completions/mean_terminated_length": 211.04000854492188, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.23827219172380865, "epoch": 0.20524515393386544, "frac_reward_zero_std": 0.125, "grad_norm": 0.023962607607245445, "kl": 0.042177736293524504, "learning_rate": 4.8376763431835424e-06, "loss": 0.0002109348715748638, "num_tokens": 42215918.0, "reward": 2.19677734375, "reward_std": 0.6242074370384216, "rewards/code_complexity_reward/mean": 0.7787109613418579, "rewards/code_complexity_reward/std": 0.19207869470119476, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49072265625, "rewards/xmlcount_reward_func/std": 0.0504322350025177, "step": 180, "step_time": 86.40490295551717 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 209.072265625, "completions/mean_terminated_length": 206.68701171875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2379092932678759, "epoch": 0.2063854047890536, "frac_reward_zero_std": 0.109375, "grad_norm": 0.026419302448630333, "kl": 0.031743364088470116, "learning_rate": 4.834129385976227e-06, "loss": 0.00015865432214923203, "num_tokens": 42389775.0, "reward": 2.259960889816284, "reward_std": 0.5607064962387085, "rewards/code_complexity_reward/mean": 0.808789074420929, "rewards/code_complexity_reward/std": 0.1440974920988083, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.03281606733798981, "step": 181, "step_time": 64.71199979446828 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 204.03515625, "completions/mean_terminated_length": 199.76634216308594, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.23079298785887659, "epoch": 0.20752565564424175, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.02539563924074173, "kl": 0.036656051815953106, "learning_rate": 4.830545422943847e-06, "loss": 0.00018327421275898814, "num_tokens": 42559765.0, "reward": 2.2702150344848633, "reward_std": 0.5809635519981384, "rewards/code_complexity_reward/mean": 0.8013671636581421, "rewards/code_complexity_reward/std": 0.1606861650943756, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09921874850988388, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49365234375, "rewards/xmlcount_reward_func/std": 0.03696192800998688, "step": 182, "step_time": 66.84336153697222 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 229.650390625, "completions/mean_terminated_length": 222.87400817871094, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.24173549516126513, "epoch": 0.20866590649942987, "frac_reward_zero_std": 0.109375, "grad_norm": 0.0229007788002491, "kl": 0.03215309797087684, "learning_rate": 4.8269245109073795e-06, "loss": 0.00016067648539319634, "num_tokens": 42744778.0, "reward": 2.2540528774261475, "reward_std": 0.6048943400382996, "rewards/code_complexity_reward/mean": 0.7751953601837158, "rewards/code_complexity_reward/std": 0.17843623459339142, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.490966796875, "rewards/xmlcount_reward_func/std": 0.041500627994537354, "step": 183, "step_time": 81.58005122933537 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 218.70703125, "completions/mean_terminated_length": 210.4618377685547, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2498968429863453, "epoch": 0.20980615735461802, "frac_reward_zero_std": 0.15625, "grad_norm": 0.02175086922943592, "kl": 0.028954477747902274, "learning_rate": 4.823266707273596e-06, "loss": 0.00014476760406978428, "num_tokens": 42925616.0, "reward": 2.2271482944488525, "reward_std": 0.639391303062439, "rewards/code_complexity_reward/mean": 0.7843749523162842, "rewards/code_complexity_reward/std": 0.19657039642333984, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843363426625729, "rewards/xmlcount_reward_func/mean": 0.4921875, "rewards/xmlcount_reward_func/std": 0.03910070285201073, "step": 184, "step_time": 70.89134044572711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 217.982421875, "completions/mean_terminated_length": 210.92601013183594, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.23102311207912862, "epoch": 0.21094640820980615, "frac_reward_zero_std": 0.109375, "grad_norm": 0.024888966232538223, "kl": 0.048769825109047815, "learning_rate": 4.819572070034162e-06, "loss": 0.00024368715821765363, "num_tokens": 43105415.0, "reward": 2.2306151390075684, "reward_std": 0.6176615953445435, "rewards/code_complexity_reward/mean": 0.78466796875, "rewards/code_complexity_reward/std": 0.18766847252845764, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.494384765625, "rewards/xmlcount_reward_func/std": 0.03027050942182541, "step": 185, "step_time": 51.18960496690124 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 212.35546875, "completions/mean_terminated_length": 206.38645935058594, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23750296980142593, "epoch": 0.2120866590649943, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.02754652313888073, "kl": 0.028279644524445757, "learning_rate": 4.815840657764704e-06, "loss": 0.0001413475110894069, "num_tokens": 43282549.0, "reward": 2.2825193405151367, "reward_std": 0.5930283069610596, "rewards/code_complexity_reward/mean": 0.799609363079071, "rewards/code_complexity_reward/std": 0.16277872025966644, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49365234375, "rewards/xmlcount_reward_func/std": 0.033489786088466644, "step": 186, "step_time": 71.34858952369541 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 212.8046875, "completions/mean_terminated_length": 207.45127868652344, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.23947559273801744, "epoch": 0.21322690992018245, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.02478155307471752, "kl": 0.031648079137085006, "learning_rate": 4.812072529623894e-06, "loss": 0.00015828973846510053, "num_tokens": 43457625.0, "reward": 2.224658250808716, "reward_std": 0.6139651536941528, "rewards/code_complexity_reward/mean": 0.7855468988418579, "rewards/code_complexity_reward/std": 0.17869961261749268, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.490478515625, "rewards/xmlcount_reward_func/std": 0.04560868814587593, "step": 187, "step_time": 71.92652308288962 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 213.91796875, "completions/mean_terminated_length": 208.58448791503906, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2372625577263534, "epoch": 0.21436716077537057, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.023840755224227905, "kl": 0.03237377086770721, "learning_rate": 4.808267745352502e-06, "loss": 0.00016184907872229815, "num_tokens": 43634743.0, "reward": 2.23291015625, "reward_std": 0.6025149822235107, "rewards/code_complexity_reward/mean": 0.781542956829071, "rewards/code_complexity_reward/std": 0.17574599385261536, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09882812947034836, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.4912109375, "rewards/xmlcount_reward_func/std": 0.04335375875234604, "step": 188, "step_time": 61.34053249284625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 206.916015625, "completions/mean_terminated_length": 199.59400939941406, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23902193387039006, "epoch": 0.21550741163055873, "frac_reward_zero_std": 0.125, "grad_norm": 0.02604338526725769, "kl": 0.02976814963039942, "learning_rate": 4.804426365272455e-06, "loss": 0.00014882147661410272, "num_tokens": 43808972.0, "reward": 2.266357421875, "reward_std": 0.5706514120101929, "rewards/code_complexity_reward/mean": 0.8020508289337158, "rewards/code_complexity_reward/std": 0.14870837330818176, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.027255699038505554, "step": 189, "step_time": 77.59111227840185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 208.802734375, "completions/mean_terminated_length": 205.20751953125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.24840563372708857, "epoch": 0.21664766248574688, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.02365870773792267, "kl": 0.02878234614036046, "learning_rate": 4.800548450285878e-06, "loss": 0.00014388217823579907, "num_tokens": 43984823.0, "reward": 2.274707078933716, "reward_std": 0.5511484146118164, "rewards/code_complexity_reward/mean": 0.8096679449081421, "rewards/code_complexity_reward/std": 0.1335913985967636, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.02564004622399807, "step": 190, "step_time": 61.405951517634094 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 209.54296875, "completions/mean_terminated_length": 203.51792907714844, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23890477628447115, "epoch": 0.217787913340935, "frac_reward_zero_std": 0.109375, "grad_norm": 0.024077218025922775, "kl": 0.031408604263560846, "learning_rate": 4.79663406187413e-06, "loss": 0.00015704457473475486, "num_tokens": 44160121.0, "reward": 2.2732419967651367, "reward_std": 0.581254243850708, "rewards/code_complexity_reward/mean": 0.795117199420929, "rewards/code_complexity_reward/std": 0.16058719158172607, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.494140625, "rewards/xmlcount_reward_func/std": 0.027579164132475853, "step": 191, "step_time": 94.18101132381707 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025390625, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 212.04296875, "completions/mean_terminated_length": 204.22845458984375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23766269232146442, "epoch": 0.21892816419612315, "frac_reward_zero_std": 0.078125, "grad_norm": 0.02584214322268963, "kl": 0.031869797938270494, "learning_rate": 4.792683262096825e-06, "loss": 0.0001592974876984954, "num_tokens": 44336739.0, "reward": 2.1736817359924316, "reward_std": 0.5911877751350403, "rewards/code_complexity_reward/mean": 0.795117199420929, "rewards/code_complexity_reward/std": 0.1756329983472824, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09902344644069672, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.492431640625, "rewards/xmlcount_reward_func/std": 0.04530387744307518, "step": 192, "step_time": 82.07077015098184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 212.6328125, "completions/mean_terminated_length": 206.05987548828125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.23788071470335126, "epoch": 0.22006841505131128, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.023924728855490685, "kl": 0.033099083782872185, "learning_rate": 4.788696113590853e-06, "loss": 0.00016550016880501062, "num_tokens": 44515179.0, "reward": 2.2123045921325684, "reward_std": 0.623917818069458, "rewards/code_complexity_reward/mean": 0.7837890386581421, "rewards/code_complexity_reward/std": 0.1911211758852005, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.4921875, "rewards/xmlcount_reward_func/std": 0.0366797111928463, "step": 193, "step_time": 82.33433745242655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 205.05078125, "completions/mean_terminated_length": 199.5586395263672, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.2302745352499187, "epoch": 0.22120866590649943, "frac_reward_zero_std": 0.125, "grad_norm": 0.026492230594158173, "kl": 0.035950019882875495, "learning_rate": 4.7846726795693855e-06, "loss": 0.00017965026199817657, "num_tokens": 44689697.0, "reward": 2.2954587936401367, "reward_std": 0.5862206816673279, "rewards/code_complexity_reward/mean": 0.798535168170929, "rewards/code_complexity_reward/std": 0.15892496705055237, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09921874850988388, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.492431640625, "rewards/xmlcount_reward_func/std": 0.04105500876903534, "step": 194, "step_time": 61.909157894551754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 219.859375, "completions/mean_terminated_length": 212.84800720214844, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.23701980034820735, "epoch": 0.22234891676168758, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.023345282301306725, "kl": 0.05207747651729733, "learning_rate": 4.780613023820872e-06, "loss": 0.00026072573382407427, "num_tokens": 44870013.0, "reward": 2.1976561546325684, "reward_std": 0.5994080305099487, "rewards/code_complexity_reward/mean": 0.780078113079071, "rewards/code_complexity_reward/std": 0.18248553574085236, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09921874850988388, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4921875, "rewards/xmlcount_reward_func/std": 0.03910070285201073, "step": 195, "step_time": 56.774886024184525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 221.001953125, "completions/mean_terminated_length": 216.38294982910156, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2292621957603842, "epoch": 0.2234891676168757, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.02538500539958477, "kl": 0.03794606967130676, "learning_rate": 4.776517210708032e-06, "loss": 0.0001897631591418758, "num_tokens": 45052878.0, "reward": 2.2120604515075684, "reward_std": 0.6093428730964661, "rewards/code_complexity_reward/mean": 0.7769531011581421, "rewards/code_complexity_reward/std": 0.1818476915359497, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.492919921875, "rewards/xmlcount_reward_func/std": 0.04114219918847084, "step": 196, "step_time": 89.12613319698721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 205.466796875, "completions/mean_terminated_length": 203.66012573242188, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23098304006271064, "epoch": 0.22462941847206386, "frac_reward_zero_std": 0.109375, "grad_norm": 0.02773020789027214, "kl": 0.03456704708514735, "learning_rate": 4.772385305166828e-06, "loss": 0.0001729805808281526, "num_tokens": 45227053.0, "reward": 2.2867677211761475, "reward_std": 0.5820407271385193, "rewards/code_complexity_reward/mean": 0.7970702648162842, "rewards/code_complexity_reward/std": 0.15154524147510529, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03056892193853855, "step": 197, "step_time": 93.88157882075757 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 201.52734375, "completions/mean_terminated_length": 192.7991943359375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22548350133001804, "epoch": 0.22576966932725198, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.027167044579982758, "kl": 0.037734294222900644, "learning_rate": 4.768217372705442e-06, "loss": 0.00018867796461563557, "num_tokens": 45398391.0, "reward": 2.269335985183716, "reward_std": 0.6182935237884521, "rewards/code_complexity_reward/mean": 0.79345703125, "rewards/code_complexity_reward/std": 0.17628799378871918, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49365234375, "rewards/xmlcount_reward_func/std": 0.03778013959527016, "step": 198, "step_time": 71.74790160171688 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 208.767578125, "completions/mean_terminated_length": 203.95437622070312, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.23218028363771737, "epoch": 0.22690992018244013, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.036067452281713486, "kl": 0.04083838046062738, "learning_rate": 4.764013479403239e-06, "loss": 0.00020441453671082854, "num_tokens": 45573740.0, "reward": 2.21826171875, "reward_std": 0.5733873248100281, "rewards/code_complexity_reward/mean": 0.79296875, "rewards/code_complexity_reward/std": 0.16710782051086426, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.027850674465298653, "step": 199, "step_time": 63.66661747824401 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 197.396484375, "completions/mean_terminated_length": 196.16275024414062, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.23174899560399354, "epoch": 0.22805017103762829, "frac_reward_zero_std": 0.125, "grad_norm": 0.026559526100754738, "kl": 0.037563109188340604, "learning_rate": 4.759773691909708e-06, "loss": 0.00018778187222778797, "num_tokens": 45741683.0, "reward": 2.315966844558716, "reward_std": 0.580288290977478, "rewards/code_complexity_reward/mean": 0.8106445074081421, "rewards/code_complexity_reward/std": 0.14807629585266113, "rewards/code_execution_reward/mean": 0.41796875, "rewards/code_execution_reward/std": 0.4937073290348053, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.034219224005937576, "step": 200, "step_time": 79.03288272675127 }, { "epoch": 0.22805017103762829, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.03, "eval_completions/max_length": 321.4, "eval_completions/max_terminated_length": 307.92, "eval_completions/mean_length": 207.1925, "eval_completions/mean_terminated_length": 199.8173342895508, "eval_completions/min_length": 115.58, "eval_completions/min_terminated_length": 115.58, "eval_entropy": 0.23014824748039245, "eval_frac_reward_zero_std": 0.1, "eval_kl": 0.039177773837000135, "eval_loss": 0.00019613401673268527, "eval_num_tokens": 45741683.0, "eval_reward": 2.191625010967255, "eval_reward_std": 0.4421767998859286, "eval_rewards/code_complexity_reward/mean": 0.7800000035762786, "eval_rewards/code_complexity_reward/std": 0.11533802609890699, "eval_rewards/code_execution_reward/mean": 0.335, "eval_rewards/code_execution_reward/std": 0.34364947497844694, "eval_rewards/code_syntax_reward/mean": 0.48375, "eval_rewards/code_syntax_reward/std": 0.029291952848434447, "eval_rewards/reasoning_present_reward_func/mean": 0.09975000157952309, "eval_rewards/reasoning_present_reward_func/std": 0.000707106813788414, "eval_rewards/xmlcount_reward_func/mean": 0.493125, "eval_rewards/xmlcount_reward_func/std": 0.015888431146740913, "eval_runtime": 686.2562, "eval_samples_per_second": 0.146, "eval_steps_per_second": 0.019, "step": 200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 202.376953125, "completions/mean_terminated_length": 195.57884216308594, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2331515932455659, "epoch": 0.2291904218928164, "frac_reward_zero_std": 0.0859375, "grad_norm": 0.02573504112660885, "kl": 0.04687958097201772, "learning_rate": 4.755498077443419e-06, "loss": 0.0002342261141166091, "num_tokens": 45914064.0, "reward": 2.2050294876098633, "reward_std": 0.5831080675125122, "rewards/code_complexity_reward/mean": 0.7841796875, "rewards/code_complexity_reward/std": 0.1714269369840622, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.02241772972047329, "step": 201, "step_time": 64.21464846096933 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 203.888671875, "completions/mean_terminated_length": 200.85009765625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.219906453974545, "epoch": 0.23033067274800456, "frac_reward_zero_std": 0.125, "grad_norm": 0.024294868111610413, "kl": 0.033152098942082375, "learning_rate": 4.7511867037909484e-06, "loss": 0.0001656989479670301, "num_tokens": 46086831.0, "reward": 2.2671875953674316, "reward_std": 0.5999592542648315, "rewards/code_complexity_reward/mean": 0.78173828125, "rewards/code_complexity_reward/std": 0.17145489156246185, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09921874850988388, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.02570982649922371, "step": 202, "step_time": 74.08898787852377 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 208.453125, "completions/mean_terminated_length": 204.24554443359375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.22650932101532817, "epoch": 0.2314709236031927, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.02429436519742012, "kl": 0.03723881740006618, "learning_rate": 4.746839639305808e-06, "loss": 0.00018628017278388143, "num_tokens": 46262451.0, "reward": 2.248095750808716, "reward_std": 0.6188414692878723, "rewards/code_complexity_reward/mean": 0.7879883050918579, "rewards/code_complexity_reward/std": 0.18006645143032074, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843363426625729, "rewards/xmlcount_reward_func/mean": 0.492919921875, "rewards/xmlcount_reward_func/std": 0.04470406472682953, "step": 203, "step_time": 61.30083087552339 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 192.046875, "completions/mean_terminated_length": 186.96826171875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.21884006657637656, "epoch": 0.23261117445838084, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.024495530873537064, "kl": 0.04913065623259172, "learning_rate": 4.742456952907358e-06, "loss": 0.0002458140952512622, "num_tokens": 46428735.0, "reward": 2.3287596702575684, "reward_std": 0.6144773364067078, "rewards/code_complexity_reward/mean": 0.803906261920929, "rewards/code_complexity_reward/std": 0.16910137236118317, "rewards/code_execution_reward/mean": 0.443359375, "rewards/code_execution_reward/std": 0.49726733565330505, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.02955172397196293, "step": 204, "step_time": 61.82449077256024 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.017578125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 194.03515625, "completions/mean_terminated_length": 188.34591674804688, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23207042692229152, "epoch": 0.233751425313569, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.028409991413354874, "kl": 0.03799321461701766, "learning_rate": 4.73803871407972e-06, "loss": 0.000189926489838399, "num_tokens": 46594833.0, "reward": 2.233203172683716, "reward_std": 0.581638514995575, "rewards/code_complexity_reward/mean": 0.803417980670929, "rewards/code_complexity_reward/std": 0.16831453144550323, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49365234375, "rewards/xmlcount_reward_func/std": 0.04235878214240074, "step": 205, "step_time": 64.57151598110795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 205.734375, "completions/mean_terminated_length": 201.48912048339844, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.22872819053009152, "epoch": 0.2348916761687571, "frac_reward_zero_std": 0.171875, "grad_norm": 0.025460032746195793, "kl": 0.037666551244910806, "learning_rate": 4.733584992870669e-06, "loss": 0.0001883181103039533, "num_tokens": 46767221.0, "reward": 2.2455079555511475, "reward_std": 0.5678148865699768, "rewards/code_complexity_reward/mean": 0.7925781011581421, "rewards/code_complexity_reward/std": 0.1486521065235138, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.029890306293964386, "step": 206, "step_time": 61.994246003217995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021484375, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 204.328125, "completions/mean_terminated_length": 197.57286071777344, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.23057729029096663, "epoch": 0.23603192702394526, "frac_reward_zero_std": 0.140625, "grad_norm": 0.030107302591204643, "kl": 0.04271914379205555, "learning_rate": 4.729095859890529e-06, "loss": 0.0002134860260412097, "num_tokens": 46938789.0, "reward": 2.2652344703674316, "reward_std": 0.6201720237731934, "rewards/code_complexity_reward/mean": 0.79541015625, "rewards/code_complexity_reward/std": 0.1816251575946808, "rewards/code_execution_reward/mean": 0.390625, "rewards/code_execution_reward/std": 0.48836761713027954, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09921874850988388, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49462890625, "rewards/xmlcount_reward_func/std": 0.031791869550943375, "step": 207, "step_time": 76.1206715432927 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 196.8671875, "completions/mean_terminated_length": 195.00982666015625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21322114975191653, "epoch": 0.23717217787913342, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.025428064167499542, "kl": 0.03664780280087143, "learning_rate": 4.724571386311046e-06, "loss": 0.0001832997950259596, "num_tokens": 47106493.0, "reward": 2.3274900913238525, "reward_std": 0.5883219242095947, "rewards/code_complexity_reward/mean": 0.7964843511581421, "rewards/code_complexity_reward/std": 0.15091808140277863, "rewards/code_execution_reward/mean": 0.44140625, "rewards/code_execution_reward/std": 0.4970405399799347, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02960825525224209, "step": 208, "step_time": 78.753433492966 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 202.142578125, "completions/mean_terminated_length": 197.22421264648438, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.22303332900628448, "epoch": 0.23831242873432154, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.027854861691594124, "kl": 0.04250666408916004, "learning_rate": 4.720011643864268e-06, "loss": 0.00021253003797028214, "num_tokens": 47277310.0, "reward": 2.3069334030151367, "reward_std": 0.5646473169326782, "rewards/code_complexity_reward/mean": 0.8096679449081421, "rewards/code_complexity_reward/std": 0.13763193786144257, "rewards/code_execution_reward/mean": 0.40625, "rewards/code_execution_reward/std": 0.49161264300346375, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.026806091889739037, "step": 209, "step_time": 51.763648983091116 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 194.970703125, "completions/mean_terminated_length": 189.93850708007812, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.22585120610892773, "epoch": 0.2394526795895097, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.025951851159334183, "kl": 0.041373066924279556, "learning_rate": 4.715416704841404e-06, "loss": 0.00020688335644081235, "num_tokens": 47446083.0, "reward": 2.2499024868011475, "reward_std": 0.5858824849128723, "rewards/code_complexity_reward/mean": 0.806347668170929, "rewards/code_complexity_reward/std": 0.162970632314682, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.021923433989286423, "step": 210, "step_time": 65.47429473605007 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.02734375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 202.455078125, "completions/mean_terminated_length": 193.7530059814453, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.22247950732707977, "epoch": 0.24059293044469784, "frac_reward_zero_std": 0.140625, "grad_norm": 0.028105011209845543, "kl": 0.04095155556569807, "learning_rate": 4.710786642091673e-06, "loss": 0.00020466775458771735, "num_tokens": 47616768.0, "reward": 2.2005858421325684, "reward_std": 0.6052584052085876, "rewards/code_complexity_reward/mean": 0.7860351800918579, "rewards/code_complexity_reward/std": 0.18638937175273895, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49365234375, "rewards/xmlcount_reward_func/std": 0.0385809987783432, "step": 211, "step_time": 79.6990862712264 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 189.376953125, "completions/mean_terminated_length": 184.25596618652344, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22229598835110664, "epoch": 0.24173318129988597, "frac_reward_zero_std": 0.171875, "grad_norm": 0.028116891160607338, "kl": 0.04830027275602333, "learning_rate": 4.706121529021158e-06, "loss": 0.0002414334157947451, "num_tokens": 47783201.0, "reward": 2.24169921875, "reward_std": 0.5846574306488037, "rewards/code_complexity_reward/mean": 0.7986327409744263, "rewards/code_complexity_reward/std": 0.16968972980976105, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.024335084483027458, "step": 212, "step_time": 82.16768101416528 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 193.294921875, "completions/mean_terminated_length": 189.5158233642578, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.22160655446350574, "epoch": 0.24287343215507412, "frac_reward_zero_std": 0.125, "grad_norm": 0.026007214561104774, "kl": 0.04319517794647254, "learning_rate": 4.7014214395916355e-06, "loss": 0.00021596538135781884, "num_tokens": 47950764.0, "reward": 2.2558107376098633, "reward_std": 0.5697221159934998, "rewards/code_complexity_reward/mean": 0.79833984375, "rewards/code_complexity_reward/std": 0.1583370566368103, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.018141774460673332, "step": 213, "step_time": 78.25786825735122 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 199.873046875, "completions/mean_terminated_length": 193.65538024902344, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22207352379336953, "epoch": 0.24401368301026224, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.029315702617168427, "kl": 0.04293911808053963, "learning_rate": 4.696686448319408e-06, "loss": 0.00021457707043737173, "num_tokens": 48122491.0, "reward": 2.255078077316284, "reward_std": 0.586150586605072, "rewards/code_complexity_reward/mean": 0.794628918170929, "rewards/code_complexity_reward/std": 0.16414237022399902, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.02040531300008297, "step": 214, "step_time": 58.29769852757454 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 194.833984375, "completions/mean_terminated_length": 187.22201538085938, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.21504526771605015, "epoch": 0.2451539338654504, "frac_reward_zero_std": 0.140625, "grad_norm": 0.030673064291477203, "kl": 0.045243342814501375, "learning_rate": 4.691916630274117e-06, "loss": 0.00022623215045314282, "num_tokens": 48288046.0, "reward": 2.3095216751098633, "reward_std": 0.5830875635147095, "rewards/code_complexity_reward/mean": 0.799121081829071, "rewards/code_complexity_reward/std": 0.15509673953056335, "rewards/code_execution_reward/mean": 0.423828125, "rewards/code_execution_reward/std": 0.4946470856666565, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.02835538238286972, "step": 215, "step_time": 58.586875203065574 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 189.7109375, "completions/mean_terminated_length": 187.17323303222656, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2210942639503628, "epoch": 0.24629418472063855, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.02467297948896885, "kl": 0.05577639630064368, "learning_rate": 4.687112061077556e-06, "loss": 0.0002788651327136904, "num_tokens": 48452558.0, "reward": 2.3064942359924316, "reward_std": 0.5963650345802307, "rewards/code_complexity_reward/mean": 0.7984375357627869, "rewards/code_complexity_reward/std": 0.1616860181093216, "rewards/code_execution_reward/mean": 0.423828125, "rewards/code_execution_reward/std": 0.4946470856666565, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.03043578378856182, "step": 216, "step_time": 67.18287673778832 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 192.09375, "completions/mean_terminated_length": 188.30039978027344, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22090790909714997, "epoch": 0.24743443557582667, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.030347811058163643, "kl": 0.043168122036149725, "learning_rate": 4.6822728169024735e-06, "loss": 0.00021589023526757956, "num_tokens": 48621926.0, "reward": 2.2674803733825684, "reward_std": 0.5917957425117493, "rewards/code_complexity_reward/mean": 0.809863269329071, "rewards/code_complexity_reward/std": 0.1612250655889511, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4931640625, "rewards/xmlcount_reward_func/std": 0.03928355872631073, "step": 217, "step_time": 72.69167246483266 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 175.931640625, "completions/mean_terminated_length": 175.2739715576172, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.21041136467829347, "epoch": 0.24857468643101482, "frac_reward_zero_std": 0.15625, "grad_norm": 0.028349263593554497, "kl": 0.06678994427784346, "learning_rate": 4.67739897447136e-06, "loss": 0.0003337169182486832, "num_tokens": 48780199.0, "reward": 2.3234376907348633, "reward_std": 0.5652095675468445, "rewards/code_complexity_reward/mean": 0.82470703125, "rewards/code_complexity_reward/std": 0.13809548318386078, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.020545316860079765, "step": 218, "step_time": 56.89493023790419 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 191.8125, "completions/mean_terminated_length": 188.6548309326172, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21160261961631477, "epoch": 0.24971493728620298, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.028709225356578827, "kl": 0.044844219053629786, "learning_rate": 4.672490611055238e-06, "loss": 0.00022409536177292466, "num_tokens": 48948275.0, "reward": 2.268749952316284, "reward_std": 0.577702522277832, "rewards/code_complexity_reward/mean": 0.79931640625, "rewards/code_complexity_reward/std": 0.15654201805591583, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.029966136440634727, "step": 219, "step_time": 74.48024909663945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 191.2265625, "completions/mean_terminated_length": 188.06312561035156, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.21511587081477046, "epoch": 0.2508551881413911, "frac_reward_zero_std": 0.109375, "grad_norm": 0.03084740601480007, "kl": 0.04511119189555757, "learning_rate": 4.667547804472431e-06, "loss": 0.0002255164727102965, "num_tokens": 49114327.0, "reward": 2.270263671875, "reward_std": 0.5928828716278076, "rewards/code_complexity_reward/mean": 0.798144519329071, "rewards/code_complexity_reward/std": 0.16693821549415588, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.021246958523988724, "step": 220, "step_time": 71.76837267167866 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 185.25, "completions/mean_terminated_length": 183.96864318847656, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.21572508662939072, "epoch": 0.2519954389965792, "frac_reward_zero_std": 0.09375, "grad_norm": 0.031705696135759354, "kl": 0.05214087446802296, "learning_rate": 4.662570633087333e-06, "loss": 0.00026067436556331813, "num_tokens": 49275535.0, "reward": 2.2428221702575684, "reward_std": 0.550783097743988, "rewards/code_complexity_reward/mean": 0.80126953125, "rewards/code_complexity_reward/std": 0.14304949343204498, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.030623575672507286, "step": 221, "step_time": 68.62810588628054 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 183.830078125, "completions/mean_terminated_length": 181.24606323242188, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.21332691493444145, "epoch": 0.2531356898517674, "frac_reward_zero_std": 0.125, "grad_norm": 0.02752385288476944, "kl": 0.04880315571790561, "learning_rate": 4.657559175809168e-06, "loss": 0.0002438636147417128, "num_tokens": 49438104.0, "reward": 2.280322313308716, "reward_std": 0.5702183246612549, "rewards/code_complexity_reward/mean": 0.8038085699081421, "rewards/code_complexity_reward/std": 0.14808017015457153, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.03035719320178032, "step": 222, "step_time": 69.73946361243725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 193.31640625, "completions/mean_terminated_length": 188.89901733398438, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21391383768059313, "epoch": 0.2542759407069555, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.02931394800543785, "kl": 0.0766636977205053, "learning_rate": 4.6525135120907314e-06, "loss": 0.0003834320814348757, "num_tokens": 49606014.0, "reward": 2.252197265625, "reward_std": 0.5702690482139587, "rewards/code_complexity_reward/mean": 0.8076171875, "rewards/code_complexity_reward/std": 0.15442141890525818, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.03680243343114853, "step": 223, "step_time": 64.73808303009719 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 183.91015625, "completions/mean_terminated_length": 181.97642517089844, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21820041607134044, "epoch": 0.2554161915621437, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.030159788206219673, "kl": 0.04394841977045871, "learning_rate": 4.647433721927139e-06, "loss": 0.00021957021090202034, "num_tokens": 49767660.0, "reward": 2.3053221702575684, "reward_std": 0.5613093972206116, "rewards/code_complexity_reward/mean": 0.8182617425918579, "rewards/code_complexity_reward/std": 0.1400429904460907, "rewards/code_execution_reward/mean": 0.396484375, "rewards/code_execution_reward/std": 0.4896455705165863, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.021303100511431694, "step": 224, "step_time": 78.33010818995535 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 193.697265625, "completions/mean_terminated_length": 191.19094848632812, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.22022860473953187, "epoch": 0.25655644241733183, "frac_reward_zero_std": 0.078125, "grad_norm": 0.03277892619371414, "kl": 0.054383940587285906, "learning_rate": 4.642319885854557e-06, "loss": 0.00027203443460166454, "num_tokens": 49932789.0, "reward": 2.270703077316284, "reward_std": 0.5962033867835999, "rewards/code_complexity_reward/mean": 0.79736328125, "rewards/code_complexity_reward/std": 0.16710343956947327, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.020545316860079765, "step": 225, "step_time": 59.88065912947059 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 182.423828125, "completions/mean_terminated_length": 178.5158233642578, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.20196662726812065, "epoch": 0.2576966932725199, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.033912789076566696, "kl": 0.0434923974389676, "learning_rate": 4.637172084948917e-06, "loss": 0.000217362743569538, "num_tokens": 50092946.0, "reward": 2.300537109375, "reward_std": 0.5962299704551697, "rewards/code_complexity_reward/mean": 0.805957019329071, "rewards/code_complexity_reward/std": 0.16731055080890656, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.025073612108826637, "step": 226, "step_time": 53.80373262707144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 183.314453125, "completions/mean_terminated_length": 181.37721252441406, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2162446000147611, "epoch": 0.2588369441277081, "frac_reward_zero_std": 0.1875, "grad_norm": 0.028690280392766, "kl": 0.04822532896650955, "learning_rate": 4.631990400824643e-06, "loss": 0.0002409599255770445, "num_tokens": 50254939.0, "reward": 2.273242235183716, "reward_std": 0.5372288227081299, "rewards/code_complexity_reward/mean": 0.8145507574081421, "rewards/code_complexity_reward/std": 0.12828154861927032, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 227, "step_time": 60.629339788109064 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 194.609375, "completions/mean_terminated_length": 190.84585571289062, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2081727054901421, "epoch": 0.25997719498289623, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.028921842575073242, "kl": 0.057625998597359285, "learning_rate": 4.626774915633349e-06, "loss": 0.0002879187813960016, "num_tokens": 50423135.0, "reward": 2.2330565452575684, "reward_std": 0.5592290163040161, "rewards/code_complexity_reward/mean": 0.8052734732627869, "rewards/code_complexity_reward/std": 0.14208237826824188, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.027404291555285454, "step": 228, "step_time": 74.67147265188396 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 178.66796875, "completions/mean_terminated_length": 178.01565551757812, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21258361800573766, "epoch": 0.2611174458380844, "frac_reward_zero_std": 0.171875, "grad_norm": 0.02890676259994507, "kl": 0.054774006945081055, "learning_rate": 4.621525712062537e-06, "loss": 0.00027376849902793765, "num_tokens": 50583929.0, "reward": 2.2899413108825684, "reward_std": 0.6048272252082825, "rewards/code_complexity_reward/mean": 0.80615234375, "rewards/code_complexity_reward/std": 0.17223134636878967, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 229, "step_time": 78.1684575136751 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 176.017578125, "completions/mean_terminated_length": 171.3603973388672, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2036900429520756, "epoch": 0.26225769669327254, "frac_reward_zero_std": 0.203125, "grad_norm": 0.03220193088054657, "kl": 0.05164264692575671, "learning_rate": 4.616242873334292e-06, "loss": 0.0002581093867775053, "num_tokens": 50742902.0, "reward": 2.3429198265075684, "reward_std": 0.6204994320869446, "rewards/code_complexity_reward/mean": 0.8138672113418579, "rewards/code_complexity_reward/std": 0.17088204622268677, "rewards/code_execution_reward/mean": 0.44921875, "rewards/code_execution_reward/std": 0.497901052236557, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.03307618945837021, "step": 230, "step_time": 53.12719439715147 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 174.029296875, "completions/mean_terminated_length": 172.7039337158203, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.20634019281715155, "epoch": 0.2633979475484607, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.02729000523686409, "kl": 0.04909262349247001, "learning_rate": 4.610926483203954e-06, "loss": 0.00024543929612264037, "num_tokens": 50898433.0, "reward": 2.37548828125, "reward_std": 0.5754303336143494, "rewards/code_complexity_reward/mean": 0.8202148079872131, "rewards/code_complexity_reward/std": 0.13361799716949463, "rewards/code_execution_reward/mean": 0.46484375, "rewards/code_execution_reward/std": 0.49925029277801514, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.024554960429668427, "step": 231, "step_time": 63.52930781804025 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 183.255859375, "completions/mean_terminated_length": 178.6990203857422, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.20813252311199903, "epoch": 0.2645381984036488, "frac_reward_zero_std": 0.0703125, "grad_norm": 0.02842484787106514, "kl": 0.04780569835565984, "learning_rate": 4.6055766259588004e-06, "loss": 0.00023894933110568672, "num_tokens": 51059356.0, "reward": 2.219433546066284, "reward_std": 0.6262176632881165, "rewards/code_complexity_reward/mean": 0.784863293170929, "rewards/code_complexity_reward/std": 0.20471540093421936, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.494140625, "rewards/xmlcount_reward_func/std": 0.03704262897372246, "step": 232, "step_time": 77.74995607044548 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 183.259765625, "completions/mean_terminated_length": 180.6712646484375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2007271891925484, "epoch": 0.26567844925883694, "frac_reward_zero_std": 0.171875, "grad_norm": 0.03302643820643425, "kl": 0.054894226166652516, "learning_rate": 4.600193386416697e-06, "loss": 0.00027447211323305964, "num_tokens": 51223621.0, "reward": 2.3005857467651367, "reward_std": 0.6080949902534485, "rewards/code_complexity_reward/mean": 0.7992187738418579, "rewards/code_complexity_reward/std": 0.17890013754367828, "rewards/code_execution_reward/mean": 0.419921875, "rewards/code_execution_reward/std": 0.4940285086631775, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.026806091889739037, "step": 233, "step_time": 80.61187380552292 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 186.818359375, "completions/mean_terminated_length": 183.61143493652344, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.20824606786482036, "epoch": 0.2668187001140251, "frac_reward_zero_std": 0.1328125, "grad_norm": 0.028077203780412674, "kl": 0.04869444962241687, "learning_rate": 4.594776849924766e-06, "loss": 0.00024349242448806763, "num_tokens": 51390420.0, "reward": 2.2425782680511475, "reward_std": 0.5664620399475098, "rewards/code_complexity_reward/mean": 0.804882824420929, "rewards/code_complexity_reward/std": 0.15361140668392181, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.021923433989286423, "step": 234, "step_time": 65.31574263330549 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 179.41015625, "completions/mean_terminated_length": 174.8000030517578, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.20824247132986784, "epoch": 0.26795895096921324, "frac_reward_zero_std": 0.1875, "grad_norm": 0.03170226514339447, "kl": 0.052958467829739675, "learning_rate": 4.589327102358024e-06, "loss": 0.00026490347227081656, "num_tokens": 51549430.0, "reward": 2.2598631381988525, "reward_std": 0.5949274897575378, "rewards/code_complexity_reward/mean": 0.8024413585662842, "rewards/code_complexity_reward/std": 0.16940350830554962, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.028849191963672638, "step": 235, "step_time": 80.01825139857829 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 177.3984375, "completions/mean_terminated_length": 174.76377868652344, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.20695965108461678, "epoch": 0.2690992018244014, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.03241194412112236, "kl": 0.058142409136053175, "learning_rate": 4.5838442301180245e-06, "loss": 0.0002906534355133772, "num_tokens": 51708498.0, "reward": 2.310986280441284, "reward_std": 0.595204770565033, "rewards/code_complexity_reward/mean": 0.822460949420929, "rewards/code_complexity_reward/std": 0.15753543376922607, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.03331366926431656, "step": 236, "step_time": 66.92133775539696 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 176.56640625, "completions/mean_terminated_length": 172.5889434814453, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.21608795085921884, "epoch": 0.2702394526795895, "frac_reward_zero_std": 0.15625, "grad_norm": 0.03091750107705593, "kl": 0.056082894443534315, "learning_rate": 4.5783283201314876e-06, "loss": 0.0002803398238029331, "num_tokens": 51866096.0, "reward": 2.249755859375, "reward_std": 0.5589261054992676, "rewards/code_complexity_reward/mean": 0.8208984136581421, "rewards/code_complexity_reward/std": 0.15603569149971008, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.027334466576576233, "step": 237, "step_time": 65.16045489069074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 191.189453125, "completions/mean_terminated_length": 186.09722900390625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.19956202246248722, "epoch": 0.27137970353477764, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.027770236134529114, "kl": 0.05439307409687899, "learning_rate": 4.572779459848922e-06, "loss": 0.00027185212820768356, "num_tokens": 52032613.0, "reward": 2.1533203125, "reward_std": 0.5444715619087219, "rewards/code_complexity_reward/mean": 0.7916015386581421, "rewards/code_complexity_reward/std": 0.17302922904491425, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 238, "step_time": 64.3039772901684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 179.0546875, "completions/mean_terminated_length": 177.09234619140625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.20753977145068347, "epoch": 0.2725199543899658, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.027677100151777267, "kl": 0.0534482310176827, "learning_rate": 4.5671977372432355e-06, "loss": 0.00026712685939855874, "num_tokens": 52191893.0, "reward": 2.2757811546325684, "reward_std": 0.5663784742355347, "rewards/code_complexity_reward/mean": 0.81591796875, "rewards/code_complexity_reward/std": 0.1504901796579361, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.020545316860079765, "step": 239, "step_time": 74.3470072792843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 176.4296875, "completions/mean_terminated_length": 173.1203155517578, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.20518330205231905, "epoch": 0.27366020524515394, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.027612660080194473, "kl": 0.055845059774583206, "learning_rate": 4.561583240808344e-06, "loss": 0.00027924415189772844, "num_tokens": 52349041.0, "reward": 2.320996046066284, "reward_std": 0.5913834571838379, "rewards/code_complexity_reward/mean": 0.8185546398162842, "rewards/code_complexity_reward/std": 0.16241046786308289, "rewards/code_execution_reward/mean": 0.41796875, "rewards/code_execution_reward/std": 0.4937073290348053, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.02449164353311062, "step": 240, "step_time": 68.1505187600851 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 188.05859375, "completions/mean_terminated_length": 184.21739196777344, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.196523871505633, "epoch": 0.2748004561003421, "frac_reward_zero_std": 0.109375, "grad_norm": 0.029629768803715706, "kl": 0.05181047646328807, "learning_rate": 4.555936059557768e-06, "loss": 0.00025917121092788875, "num_tokens": 52514463.0, "reward": 2.2659177780151367, "reward_std": 0.5941623449325562, "rewards/code_complexity_reward/mean": 0.788867175579071, "rewards/code_complexity_reward/std": 0.17093929648399353, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.02449164353311062, "step": 241, "step_time": 84.25357607938349 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 178.24609375, "completions/mean_terminated_length": 174.9546356201172, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2059872499667108, "epoch": 0.2759407069555302, "frac_reward_zero_std": 0.109375, "grad_norm": 0.02915210835635662, "kl": 0.05363402588409372, "learning_rate": 4.5502562830232225e-06, "loss": 0.0002682644408196211, "num_tokens": 52675989.0, "reward": 2.281005859375, "reward_std": 0.5672912001609802, "rewards/code_complexity_reward/mean": 0.8091796636581421, "rewards/code_complexity_reward/std": 0.14950574934482574, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.018207494169473648, "step": 242, "step_time": 59.06751239486039 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 172.041015625, "completions/mean_terminated_length": 170.70785522460938, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2071245708502829, "epoch": 0.27708095781071834, "frac_reward_zero_std": 0.15625, "grad_norm": 0.030582277104258537, "kl": 0.05339636717690155, "learning_rate": 4.544544001253189e-06, "loss": 0.000266881485003978, "num_tokens": 52834074.0, "reward": 2.3287110328674316, "reward_std": 0.5701078176498413, "rewards/code_complexity_reward/mean": 0.8219726085662842, "rewards/code_complexity_reward/std": 0.14695411920547485, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.020545316860079765, "step": 243, "step_time": 66.10420730803162 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 173.619140625, "completions/mean_terminated_length": 171.624755859375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.2048875531181693, "epoch": 0.2782212086659065, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.0352105088531971, "kl": 0.05313079306506552, "learning_rate": 4.538799304811503e-06, "loss": 0.0002656897122506052, "num_tokens": 52991099.0, "reward": 2.253222703933716, "reward_std": 0.5546850562095642, "rewards/code_complexity_reward/mean": 0.8145507574081421, "rewards/code_complexity_reward/std": 0.14925841987133026, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 244, "step_time": 64.95857367571443 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 174.232421875, "completions/mean_terminated_length": 171.5728302001953, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2024712162092328, "epoch": 0.27936145952109465, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.033733565360307693, "kl": 0.06927126171649434, "learning_rate": 4.533022284775903e-06, "loss": 0.00034623686224222183, "num_tokens": 53149138.0, "reward": 2.2757325172424316, "reward_std": 0.5991009473800659, "rewards/code_complexity_reward/mean": 0.8042968511581421, "rewards/code_complexity_reward/std": 0.17399725317955017, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02960825525224209, "step": 245, "step_time": 61.57076280936599 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 159.81640625, "completions/mean_terminated_length": 158.435302734375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.19837709050625563, "epoch": 0.2805017103762828, "frac_reward_zero_std": 0.1875, "grad_norm": 0.03112269751727581, "kl": 0.06007756065810099, "learning_rate": 4.527213032736596e-06, "loss": 0.0003002701560035348, "num_tokens": 53298520.0, "reward": 2.3470702171325684, "reward_std": 0.5777119994163513, "rewards/code_complexity_reward/mean": 0.8263671398162842, "rewards/code_complexity_reward/std": 0.14251750707626343, "rewards/code_execution_reward/mean": 0.4296875, "rewards/code_execution_reward/std": 0.4955156147480011, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02059757336974144, "step": 246, "step_time": 76.83590437658131 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 179.140625, "completions/mean_terminated_length": 177.1787872314453, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.19586496357806027, "epoch": 0.28164196123147095, "frac_reward_zero_std": 0.09375, "grad_norm": 0.03198295459151268, "kl": 0.0658326160046272, "learning_rate": 4.521371640794802e-06, "loss": 0.00032915198244154453, "num_tokens": 53459180.0, "reward": 2.2698731422424316, "reward_std": 0.5453004240989685, "rewards/code_complexity_reward/mean": 0.808300793170929, "rewards/code_complexity_reward/std": 0.13934583961963654, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 247, "step_time": 68.56719286367297 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 175.421875, "completions/mean_terminated_length": 172.10256958007812, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.19202662515453994, "epoch": 0.28278221208665905, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.03230125457048416, "kl": 0.06379556772299111, "learning_rate": 4.5154982015612965e-06, "loss": 0.00031901057809591293, "num_tokens": 53616188.0, "reward": 2.269775390625, "reward_std": 0.5774089097976685, "rewards/code_complexity_reward/mean": 0.8079102039337158, "rewards/code_complexity_reward/std": 0.16269178688526154, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03337814658880234, "step": 248, "step_time": 60.20348109398037 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 169.326171875, "completions/mean_terminated_length": 169.326171875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2010133599396795, "epoch": 0.2839224629418472, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.03240222856402397, "kl": 0.08286728523671627, "learning_rate": 4.509592808154936e-06, "loss": 0.00041453063022345304, "num_tokens": 53769063.0, "reward": 2.3792967796325684, "reward_std": 0.5690235495567322, "rewards/code_complexity_reward/mean": 0.8257812857627869, "rewards/code_complexity_reward/std": 0.1268671751022339, "rewards/code_execution_reward/mean": 0.462890625, "rewards/code_execution_reward/std": 0.4991086423397064, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.031142795458436012, "step": 249, "step_time": 68.02197799086571 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 170.0, "completions/mean_terminated_length": 166.62722778320312, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.19818991725333035, "epoch": 0.28506271379703535, "frac_reward_zero_std": 0.15625, "grad_norm": 0.03170590475201607, "kl": 0.08953839287278242, "learning_rate": 4.50365555420119e-06, "loss": 0.0004476200556382537, "num_tokens": 53924327.0, "reward": 2.267333984375, "reward_std": 0.6032333970069885, "rewards/code_complexity_reward/mean": 0.810253918170929, "rewards/code_complexity_reward/std": 0.17437990009784698, "rewards/code_execution_reward/mean": 0.375, "rewards/code_execution_reward/std": 0.4845963716506958, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.026264816522598267, "step": 250, "step_time": 87.27630731742829 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 163.255859375, "completions/mean_terminated_length": 161.2003936767578, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.1935086459852755, "epoch": 0.2862029646522235, "frac_reward_zero_std": 0.1875, "grad_norm": 0.03381437808275223, "kl": 0.06031347921816632, "learning_rate": 4.497686533830648e-06, "loss": 0.0003015996189787984, "num_tokens": 54075958.0, "reward": 2.334912061691284, "reward_std": 0.5792465806007385, "rewards/code_complexity_reward/mean": 0.818164050579071, "rewards/code_complexity_reward/std": 0.1529301106929779, "rewards/code_execution_reward/mean": 0.427734375, "rewards/code_execution_reward/std": 0.4952339828014374, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02257700450718403, "step": 251, "step_time": 67.44797688443214 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 168.60546875, "completions/mean_terminated_length": 167.2588348388672, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.19315886427648365, "epoch": 0.28734321550741165, "frac_reward_zero_std": 0.15625, "grad_norm": 0.02976181171834469, "kl": 0.05416807485744357, "learning_rate": 4.491685841677538e-06, "loss": 0.0002707442035898566, "num_tokens": 54230232.0, "reward": 2.3012208938598633, "reward_std": 0.5679551959037781, "rewards/code_complexity_reward/mean": 0.808886706829071, "rewards/code_complexity_reward/std": 0.14729833602905273, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 252, "step_time": 50.22273126151413 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 178.001953125, "completions/mean_terminated_length": 173.37228393554688, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.20406437991186976, "epoch": 0.28848346636259975, "frac_reward_zero_std": 0.15625, "grad_norm": 0.03198167681694031, "kl": 0.0751058374880813, "learning_rate": 4.485653572878213e-06, "loss": 0.00037558120675385, "num_tokens": 54390665.0, "reward": 2.261523485183716, "reward_std": 0.5863887667655945, "rewards/code_complexity_reward/mean": 0.8091796636581421, "rewards/code_complexity_reward/std": 0.1689547449350357, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.03187067061662674, "step": 253, "step_time": 70.95272335782647 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 189.59375, "completions/mean_terminated_length": 184.4761962890625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.1963877882808447, "epoch": 0.2896237172177879, "frac_reward_zero_std": 0.1484375, "grad_norm": 0.03066008910536766, "kl": 0.07586333475774154, "learning_rate": 4.4795898230696535e-06, "loss": 0.00037918720045126975, "num_tokens": 54555881.0, "reward": 2.20556640625, "reward_std": 0.5596998929977417, "rewards/code_complexity_reward/mean": 0.7974609732627869, "rewards/code_complexity_reward/std": 0.16365863382816315, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.029966136440634727, "step": 254, "step_time": 57.20844522677362 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 159.8125, "completions/mean_terminated_length": 158.43138122558594, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.1910201805876568, "epoch": 0.29076396807297605, "frac_reward_zero_std": 0.203125, "grad_norm": 0.03338123857975006, "kl": 0.06078197021270171, "learning_rate": 4.473494688387945e-06, "loss": 0.0003039111034013331, "num_tokens": 54706713.0, "reward": 2.38623046875, "reward_std": 0.544835090637207, "rewards/code_complexity_reward/mean": 0.8418945074081421, "rewards/code_complexity_reward/std": 0.10924811661243439, "rewards/code_execution_reward/mean": 0.447265625, "rewards/code_execution_reward/std": 0.4976975917816162, "rewards/code_syntax_reward/mean": 0.498046875, "rewards/code_syntax_reward/std": 0.03121940791606903, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 255, "step_time": 61.734931310638785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 164.537109375, "completions/mean_terminated_length": 163.85714721679688, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.20203981082886457, "epoch": 0.2919042189281642, "frac_reward_zero_std": 0.1171875, "grad_norm": 0.03266831859946251, "kl": 0.0908281960291788, "learning_rate": 4.467368265466759e-06, "loss": 0.0004541774978861213, "num_tokens": 54858612.0, "reward": 2.3275880813598633, "reward_std": 0.5701494216918945, "rewards/code_complexity_reward/mean": 0.821484386920929, "rewards/code_complexity_reward/std": 0.1463761329650879, "rewards/code_execution_reward/mean": 0.41796875, "rewards/code_execution_reward/std": 0.4937073290348053, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03521692752838135, "step": 256, "step_time": 78.6811590148136 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 167.615234375, "completions/mean_terminated_length": 164.90354919433594, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2035896850284189, "epoch": 0.29304446978335236, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.02890358492732048, "kl": 0.07102720893453807, "learning_rate": 4.461210651435814e-06, "loss": 0.00035505741834640503, "num_tokens": 55011375.0, "reward": 2.2704100608825684, "reward_std": 0.57070392370224, "rewards/code_complexity_reward/mean": 0.8170897960662842, "rewards/code_complexity_reward/std": 0.15884220600128174, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.021923433989286423, "step": 257, "step_time": 59.484294259920716 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 166.306640625, "completions/mean_terminated_length": 163.5846405029297, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.18740237574093044, "epoch": 0.29418472063854045, "frac_reward_zero_std": 0.21875, "grad_norm": 0.031060634180903435, "kl": 0.06832174482406117, "learning_rate": 4.4550219439193435e-06, "loss": 0.0003415188111830503, "num_tokens": 55164240.0, "reward": 2.247802734375, "reward_std": 0.5498610138893127, "rewards/code_complexity_reward/mean": 0.8277343511581421, "rewards/code_complexity_reward/std": 0.14923924207687378, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.018207494169473648, "step": 258, "step_time": 88.30128087289631 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 499.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 159.31640625, "completions/mean_terminated_length": 159.31640625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.1882251997012645, "epoch": 0.2953249714937286, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.028710749000310898, "kl": 0.06381085584871471, "learning_rate": 4.448802241034541e-06, "loss": 0.0003188910777680576, "num_tokens": 55311374.0, "reward": 2.320605516433716, "reward_std": 0.5576633810997009, "rewards/code_complexity_reward/mean": 0.8299804925918579, "rewards/code_complexity_reward/std": 0.13818292319774628, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 259, "step_time": 76.41279315110296 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 161.396484375, "completions/mean_terminated_length": 160.02157592773438, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.19284012052230537, "epoch": 0.29646522234891676, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.03056541457772255, "kl": 0.0734664709016215, "learning_rate": 4.4425516413900085e-06, "loss": 0.0003673761966638267, "num_tokens": 55461889.0, "reward": 2.3185057640075684, "reward_std": 0.5618589520454407, "rewards/code_complexity_reward/mean": 0.8292968273162842, "rewards/code_complexity_reward/std": 0.13824130594730377, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02960825525224209, "step": 260, "step_time": 61.67384667135775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 160.875, "completions/mean_terminated_length": 157.4122314453125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.1889479912351817, "epoch": 0.2976054732041049, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.04103485867381096, "kl": 0.06564205235918052, "learning_rate": 4.4362702440841945e-06, "loss": 0.0003282622783444822, "num_tokens": 55611757.0, "reward": 2.2997560501098633, "reward_std": 0.5634463429450989, "rewards/code_complexity_reward/mean": 0.821582019329071, "rewards/code_complexity_reward/std": 0.14296290278434753, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 261, "step_time": 59.80026198551059 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 168.322265625, "completions/mean_terminated_length": 166.29666137695312, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.18828329001553357, "epoch": 0.29874572405929306, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.03165696933865547, "kl": 0.06225978783913888, "learning_rate": 4.429958148703818e-06, "loss": 0.0003112271660938859, "num_tokens": 55765258.0, "reward": 2.28515625, "reward_std": 0.5385681390762329, "rewards/code_complexity_reward/mean": 0.8210937976837158, "rewards/code_complexity_reward/std": 0.1253737062215805, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.025770151987671852, "step": 262, "step_time": 70.25336590409279 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 163.216796875, "completions/mean_terminated_length": 161.84902954101562, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.19345677294768393, "epoch": 0.2998859749144812, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.030327361077070236, "kl": 0.0660304365446791, "learning_rate": 4.423615455322293e-06, "loss": 0.00033021444687619805, "num_tokens": 55916405.0, "reward": 2.33251953125, "reward_std": 0.5788675546646118, "rewards/code_complexity_reward/mean": 0.8252929449081421, "rewards/code_complexity_reward/std": 0.1517346203327179, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 263, "step_time": 69.58441328257322 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 159.9765625, "completions/mean_terminated_length": 157.20472717285156, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.18905812571756542, "epoch": 0.3010262257696693, "frac_reward_zero_std": 0.125, "grad_norm": 0.037902913987636566, "kl": 0.09709456551354378, "learning_rate": 4.417242264498143e-06, "loss": 0.0004859779146499932, "num_tokens": 56066309.0, "reward": 2.3373045921325684, "reward_std": 0.5764244198799133, "rewards/code_complexity_reward/mean": 0.8253905773162842, "rewards/code_complexity_reward/std": 0.14446690678596497, "rewards/code_execution_reward/mean": 0.423828125, "rewards/code_execution_reward/std": 0.4946470856666565, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.024554960429668427, "step": 264, "step_time": 67.5835915254429 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 165.357421875, "completions/mean_terminated_length": 163.31434631347656, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.19376624235883355, "epoch": 0.30216647662485746, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.035874269902706146, "kl": 0.0750752062886022, "learning_rate": 4.410838677273403e-06, "loss": 0.00037549121771007776, "num_tokens": 56221420.0, "reward": 2.2840332984924316, "reward_std": 0.5760595798492432, "rewards/code_complexity_reward/mean": 0.8156249523162842, "rewards/code_complexity_reward/std": 0.15814481675624847, "rewards/code_execution_reward/mean": 0.380859375, "rewards/code_execution_reward/std": 0.48607301712036133, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.022640403360128403, "step": 265, "step_time": 61.18420105893165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 160.501953125, "completions/mean_terminated_length": 159.8140869140625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.19967085984535515, "epoch": 0.3033067274800456, "frac_reward_zero_std": 0.15625, "grad_norm": 0.03183835744857788, "kl": 0.0708322738064453, "learning_rate": 4.404404795172022e-06, "loss": 0.0003541673067957163, "num_tokens": 56372029.0, "reward": 2.299365282058716, "reward_std": 0.5935859084129333, "rewards/code_complexity_reward/mean": 0.8165038824081421, "rewards/code_complexity_reward/std": 0.16615863144397736, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 266, "step_time": 58.539037803187966 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 159.630859375, "completions/mean_terminated_length": 158.9412841796875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.18637713126372546, "epoch": 0.30444697833523376, "frac_reward_zero_std": 0.203125, "grad_norm": 0.030040379613637924, "kl": 0.06374651717487723, "learning_rate": 4.397940720198246e-06, "loss": 0.00031873100670054555, "num_tokens": 56522852.0, "reward": 2.3023924827575684, "reward_std": 0.5343031287193298, "rewards/code_complexity_reward/mean": 0.830761730670929, "rewards/code_complexity_reward/std": 0.1279677301645279, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 267, "step_time": 60.59860117919743 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 167.458984375, "completions/mean_terminated_length": 166.10784912109375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.20210714871063828, "epoch": 0.3055872291904219, "frac_reward_zero_std": 0.140625, "grad_norm": 0.03519962355494499, "kl": 0.07101402041735128, "learning_rate": 4.39144655483501e-06, "loss": 0.00035490503069013357, "num_tokens": 56676547.0, "reward": 2.265625, "reward_std": 0.5458233952522278, "rewards/code_complexity_reward/mean": 0.8272461295127869, "rewards/code_complexity_reward/std": 0.14074456691741943, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 268, "step_time": 71.68386499211192 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 169.65625, "completions/mean_terminated_length": 166.96063232421875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.1869495406281203, "epoch": 0.30672748004561, "frac_reward_zero_std": 0.1953125, "grad_norm": 4.296852111816406, "kl": 0.1581810435745865, "learning_rate": 4.38492240204231e-06, "loss": 0.0007905153906904161, "num_tokens": 56831223.0, "reward": 2.2833495140075684, "reward_std": 0.5864416360855103, "rewards/code_complexity_reward/mean": 0.8101562261581421, "rewards/code_complexity_reward/std": 0.16831842064857483, "rewards/code_execution_reward/mean": 0.390625, "rewards/code_execution_reward/std": 0.48836761713027954, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.030623575672507286, "step": 269, "step_time": 51.81777594424784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 164.55078125, "completions/mean_terminated_length": 161.81495666503906, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.18286923062987626, "epoch": 0.30786773090079816, "frac_reward_zero_std": 0.15625, "grad_norm": 0.03399346023797989, "kl": 0.07107645517680794, "learning_rate": 4.378368365255564e-06, "loss": 0.00035540180397219956, "num_tokens": 56983837.0, "reward": 2.329345703125, "reward_std": 0.5797710418701172, "rewards/code_complexity_reward/mean": 0.820019543170929, "rewards/code_complexity_reward/std": 0.15274590253829956, "rewards/code_execution_reward/mean": 0.421875, "rewards/code_execution_reward/std": 0.49434176087379456, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.01981583796441555, "step": 270, "step_time": 68.58899736590683 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 160.353515625, "completions/mean_terminated_length": 159.6653594970703, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.19422748452052474, "epoch": 0.3090079817559863, "frac_reward_zero_std": 0.203125, "grad_norm": 0.03827596828341484, "kl": 0.09404907241696492, "learning_rate": 4.371784548383985e-06, "loss": 0.00046999030746519566, "num_tokens": 57134566.0, "reward": 2.3488283157348633, "reward_std": 0.5504521727561951, "rewards/code_complexity_reward/mean": 0.838671863079071, "rewards/code_complexity_reward/std": 0.12069836258888245, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.022032126784324646, "step": 271, "step_time": 81.92601941991597 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 491.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 158.0625, "completions/mean_terminated_length": 158.0625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.1839744965545833, "epoch": 0.31014823261117447, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.03145177662372589, "kl": 0.0780118863331154, "learning_rate": 4.36517105580892e-06, "loss": 0.0003899744478985667, "num_tokens": 57283062.0, "reward": 2.3338379859924316, "reward_std": 0.5622273683547974, "rewards/code_complexity_reward/mean": 0.827343761920929, "rewards/code_complexity_reward/std": 0.1356811225414276, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 272, "step_time": 55.616742461919785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 158.46484375, "completions/mean_terminated_length": 156.3811492919922, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.19749901164323092, "epoch": 0.3112884834663626, "frac_reward_zero_std": 0.171875, "grad_norm": 0.034232817590236664, "kl": 0.07307144987862557, "learning_rate": 4.358527992382206e-06, "loss": 0.0003652115701697767, "num_tokens": 57432788.0, "reward": 2.3060545921325684, "reward_std": 0.5713991522789001, "rewards/code_complexity_reward/mean": 0.833984375, "rewards/code_complexity_reward/std": 0.14248062670230865, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.028043000027537346, "step": 273, "step_time": 63.83309366274625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01953125, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 182.814453125, "completions/mean_terminated_length": 176.25697326660156, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.1846045993734151, "epoch": 0.3124287343215507, "frac_reward_zero_std": 0.171875, "grad_norm": 0.031808506697416306, "kl": 0.07419132493669167, "learning_rate": 4.351855463424498e-06, "loss": 0.0003709621378220618, "num_tokens": 57597185.0, "reward": 2.265673875808716, "reward_std": 0.6211139559745789, "rewards/code_complexity_reward/mean": 0.7899414300918579, "rewards/code_complexity_reward/std": 0.19158264994621277, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.024997277185320854, "step": 274, "step_time": 58.78681189380586 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 164.107421875, "completions/mean_terminated_length": 163.42662048339844, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.18352670804597437, "epoch": 0.31356898517673887, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.03336268663406372, "kl": 0.09084491361863911, "learning_rate": 4.345153574723611e-06, "loss": 0.00045432333718053997, "num_tokens": 57747868.0, "reward": 2.307177782058716, "reward_std": 0.5580434203147888, "rewards/code_complexity_reward/mean": 0.824511706829071, "rewards/code_complexity_reward/std": 0.1339588314294815, "rewards/code_execution_reward/mean": 0.390625, "rewards/code_execution_reward/std": 0.48836761713027954, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.027517380192875862, "step": 275, "step_time": 77.17187189962715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 498.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 159.306640625, "completions/mean_terminated_length": 159.306640625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.1863459872547537, "epoch": 0.314709236031927, "frac_reward_zero_std": 0.15625, "grad_norm": 0.036653243005275726, "kl": 0.08811162493657321, "learning_rate": 4.338422432532829e-06, "loss": 0.00044042119407095015, "num_tokens": 57898961.0, "reward": 2.2759766578674316, "reward_std": 0.5441842675209045, "rewards/code_complexity_reward/mean": 0.827343761920929, "rewards/code_complexity_reward/std": 0.1382175236940384, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 276, "step_time": 62.02527721505612 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 459.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 156.630859375, "completions/mean_terminated_length": 156.630859375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.18741508550010622, "epoch": 0.31584948688711517, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.03449012339115143, "kl": 0.07979265681933612, "learning_rate": 4.331662143569235e-06, "loss": 0.0003988091484643519, "num_tokens": 58046276.0, "reward": 2.3021974563598633, "reward_std": 0.5190559029579163, "rewards/code_complexity_reward/mean": 0.8362305164337158, "rewards/code_complexity_reward/std": 0.11468005180358887, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 277, "step_time": 46.75839728675783 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 171.716796875, "completions/mean_terminated_length": 169.71119689941406, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.18775978009216487, "epoch": 0.3169897377423033, "frac_reward_zero_std": 0.171875, "grad_norm": 0.033961981534957886, "kl": 0.08191134920343757, "learning_rate": 4.324872815012005e-06, "loss": 0.00040943006752058864, "num_tokens": 58202647.0, "reward": 2.3077635765075684, "reward_std": 0.5680217742919922, "rewards/code_complexity_reward/mean": 0.8158203363418579, "rewards/code_complexity_reward/std": 0.14855371415615082, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 278, "step_time": 71.41777806542814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 167.34765625, "completions/mean_terminated_length": 162.57029724121094, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.18410431244410574, "epoch": 0.3181299885974915, "frac_reward_zero_std": 0.1875, "grad_norm": 0.03384959325194359, "kl": 0.08831545663997531, "learning_rate": 4.318054554500719e-06, "loss": 0.0004416859010234475, "num_tokens": 58360181.0, "reward": 2.232714891433716, "reward_std": 0.5840375423431396, "rewards/code_complexity_reward/mean": 0.804492175579071, "rewards/code_complexity_reward/std": 0.1712145358324051, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49462890625, "rewards/xmlcount_reward_func/std": 0.036283548921346664, "step": 279, "step_time": 69.03765210788697 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 161.51171875, "completions/mean_terminated_length": 158.75196838378906, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.181283496087417, "epoch": 0.31927023945267957, "frac_reward_zero_std": 0.21875, "grad_norm": 0.03358815237879753, "kl": 0.07213057501940057, "learning_rate": 4.3112074701336505e-06, "loss": 0.0003606574609875679, "num_tokens": 58510635.0, "reward": 2.318652391433716, "reward_std": 0.5652170777320862, "rewards/code_complexity_reward/mean": 0.8231445550918579, "rewards/code_complexity_reward/std": 0.14428606629371643, "rewards/code_execution_reward/mean": 0.40625, "rewards/code_execution_reward/std": 0.49161264300346375, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.021923433989286423, "step": 280, "step_time": 59.35955333895981 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 166.48828125, "completions/mean_terminated_length": 165.1333465576172, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.18855447391979396, "epoch": 0.3204104903078677, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.03607215732336044, "kl": 0.07023161672987044, "learning_rate": 4.304331670466052e-06, "loss": 0.00035120046231895685, "num_tokens": 58664853.0, "reward": 2.248779296875, "reward_std": 0.5599873661994934, "rewards/code_complexity_reward/mean": 0.8121093511581421, "rewards/code_complexity_reward/std": 0.16587501764297485, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 281, "step_time": 59.53413893748075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 159.716796875, "completions/mean_terminated_length": 157.64047241210938, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.1893180024344474, "epoch": 0.3215507411630559, "frac_reward_zero_std": 0.1875, "grad_norm": 0.03370405733585358, "kl": 0.08795136801199988, "learning_rate": 4.297427264508436e-06, "loss": 0.0004397503798827529, "num_tokens": 58813940.0, "reward": 2.2754883766174316, "reward_std": 0.5705700516700745, "rewards/code_complexity_reward/mean": 0.8221679925918579, "rewards/code_complexity_reward/std": 0.15369482338428497, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.028043000027537346, "step": 282, "step_time": 49.84702517092228 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 158.71484375, "completions/mean_terminated_length": 158.0234832763672, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.19146683439612389, "epoch": 0.322690992018244, "frac_reward_zero_std": 0.1640625, "grad_norm": 0.03678690642118454, "kl": 0.08500271220691502, "learning_rate": 4.290494361724844e-06, "loss": 0.00042493356158956885, "num_tokens": 58964730.0, "reward": 2.2718749046325684, "reward_std": 0.5648480653762817, "rewards/code_complexity_reward/mean": 0.8216796517372131, "rewards/code_complexity_reward/std": 0.1574898213148117, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 283, "step_time": 69.8419136768207 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 157.5546875, "completions/mean_terminated_length": 155.4656219482422, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.18590005091391504, "epoch": 0.3238312428734322, "frac_reward_zero_std": 0.171875, "grad_norm": 0.03646523132920265, "kl": 0.1007467043818906, "learning_rate": 4.283533072031116e-06, "loss": 0.0005034622736275196, "num_tokens": 59113190.0, "reward": 2.2183103561401367, "reward_std": 0.5362613797187805, "rewards/code_complexity_reward/mean": 0.8306640386581421, "rewards/code_complexity_reward/std": 0.14394888281822205, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03337814658880234, "step": 284, "step_time": 66.28020412754267 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 503.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 154.61328125, "completions/mean_terminated_length": 154.61328125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.18151395116001368, "epoch": 0.3249714937286203, "frac_reward_zero_std": 0.28125, "grad_norm": 0.03214148432016373, "kl": 0.0874317055568099, "learning_rate": 4.276543505793142e-06, "loss": 0.0004373154370114207, "num_tokens": 59259352.0, "reward": 2.3368163108825684, "reward_std": 0.5582393407821655, "rewards/code_complexity_reward/mean": 0.8291015625, "rewards/code_complexity_reward/std": 0.1459926962852478, "rewards/code_execution_reward/mean": 0.4140625, "rewards/code_execution_reward/std": 0.49304109811782837, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 285, "step_time": 65.95185692887753 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 156.26953125, "completions/mean_terminated_length": 154.87451171875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.17557296878658235, "epoch": 0.3261117445838084, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.03447257727384567, "kl": 0.08015204797266051, "learning_rate": 4.269525773825115e-06, "loss": 0.00040056032594293356, "num_tokens": 59406930.0, "reward": 2.3412108421325684, "reward_std": 0.5618396401405334, "rewards/code_complexity_reward/mean": 0.82861328125, "rewards/code_complexity_reward/std": 0.1365160197019577, "rewards/code_execution_reward/mean": 0.419921875, "rewards/code_execution_reward/std": 0.4940285086631775, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 286, "step_time": 52.499641325324774 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 167.427734375, "completions/mean_terminated_length": 166.07647705078125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.1952177545754239, "epoch": 0.3272519954389966, "frac_reward_zero_std": 0.1796875, "grad_norm": 0.03576726093888283, "kl": 0.09488019003765658, "learning_rate": 4.262479987387776e-06, "loss": 0.0004744016914628446, "num_tokens": 59561401.0, "reward": 2.2721190452575684, "reward_std": 0.5610854029655457, "rewards/code_complexity_reward/mean": 0.8236328363418579, "rewards/code_complexity_reward/std": 0.15156742930412292, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.494384765625, "rewards/xmlcount_reward_func/std": 0.04755738377571106, "step": 287, "step_time": 59.28349763993174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 151.029296875, "completions/mean_terminated_length": 150.32289123535156, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.17934878263622522, "epoch": 0.32839224629418473, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.0363546721637249, "kl": 0.08964689308777452, "learning_rate": 4.255406258186644e-06, "loss": 0.0004482632502913475, "num_tokens": 59707284.0, "reward": 2.334765672683716, "reward_std": 0.5859220623970032, "rewards/code_complexity_reward/mean": 0.83056640625, "rewards/code_complexity_reward/std": 0.15748845040798187, "rewards/code_execution_reward/mean": 0.419921875, "rewards/code_execution_reward/std": 0.4940285086631775, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.036414988338947296, "step": 288, "step_time": 57.34950440004468 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 158.697265625, "completions/mean_terminated_length": 158.00587463378906, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.18578319461084902, "epoch": 0.3295324971493729, "frac_reward_zero_std": 0.234375, "grad_norm": 0.035686612129211426, "kl": 0.09487741609336808, "learning_rate": 4.248304698370253e-06, "loss": 0.00047458152403123677, "num_tokens": 59857697.0, "reward": 2.27783203125, "reward_std": 0.5406103730201721, "rewards/code_complexity_reward/mean": 0.8240233659744263, "rewards/code_complexity_reward/std": 0.13493864238262177, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.028089813888072968, "step": 289, "step_time": 96.73858437500894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 164.734375, "completions/mean_terminated_length": 162.6876220703125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.18376589193940163, "epoch": 0.330672748004561, "frac_reward_zero_std": 0.171875, "grad_norm": 0.03233793005347252, "kl": 0.07583314180374146, "learning_rate": 4.241175420528369e-06, "loss": 0.0003791968629229814, "num_tokens": 60013609.0, "reward": 2.225292921066284, "reward_std": 0.5335628390312195, "rewards/code_complexity_reward/mean": 0.818164050579071, "rewards/code_complexity_reward/std": 0.1319066286087036, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 290, "step_time": 60.563510932028294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 137.80859375, "completions/mean_terminated_length": 137.07632446289062, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.18433524598367512, "epoch": 0.33181299885974913, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.034606464207172394, "kl": 0.09941666148370132, "learning_rate": 4.234018537690204e-06, "loss": 0.0004970295121893287, "num_tokens": 60151195.0, "reward": 2.3748536109924316, "reward_std": 0.5813862681388855, "rewards/code_complexity_reward/mean": 0.8436523675918579, "rewards/code_complexity_reward/std": 0.14733025431632996, "rewards/code_execution_reward/mean": 0.44140625, "rewards/code_execution_reward/std": 0.4970405399799347, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 291, "step_time": 78.56324558984488 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 159.744140625, "completions/mean_terminated_length": 158.3627471923828, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.17806038388516754, "epoch": 0.3329532497149373, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.03228051960468292, "kl": 0.09840102639282122, "learning_rate": 4.226834163322629e-06, "loss": 0.0004919616039842367, "num_tokens": 60302272.0, "reward": 2.325488567352295, "reward_std": 0.5522523522377014, "rewards/code_complexity_reward/mean": 0.8299804329872131, "rewards/code_complexity_reward/std": 0.13269226253032684, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 292, "step_time": 65.02683424111456 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 153.982421875, "completions/mean_terminated_length": 151.87229919433594, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.18281734362244606, "epoch": 0.33409350057012543, "frac_reward_zero_std": 0.1875, "grad_norm": 0.03700634092092514, "kl": 0.08161179645685479, "learning_rate": 4.21962241132837e-06, "loss": 0.0004078791243955493, "num_tokens": 60449531.0, "reward": 2.2916016578674316, "reward_std": 0.5407153367996216, "rewards/code_complexity_reward/mean": 0.8327147960662842, "rewards/code_complexity_reward/std": 0.12889356911182404, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.020545316860079765, "step": 293, "step_time": 70.40180592332035 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 149.61328125, "completions/mean_terminated_length": 148.90411376953125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.18027208745479584, "epoch": 0.3352337514253136, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.040027327835559845, "kl": 0.09362007153686136, "learning_rate": 4.212383396044204e-06, "loss": 0.00046791735803708434, "num_tokens": 60591589.0, "reward": 2.247119426727295, "reward_std": 0.5221654772758484, "rewards/code_complexity_reward/mean": 0.8306640386581421, "rewards/code_complexity_reward/std": 0.1379084289073944, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 294, "step_time": 56.86104054749012 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 154.861328125, "completions/mean_terminated_length": 153.46080017089844, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.17579694953747094, "epoch": 0.3363740022805017, "frac_reward_zero_std": 0.21875, "grad_norm": 0.03156908228993416, "kl": 0.08578503935132176, "learning_rate": 4.205117232239148e-06, "loss": 0.00042882608249783516, "num_tokens": 60739502.0, "reward": 2.3520021438598633, "reward_std": 0.5720471143722534, "rewards/code_complexity_reward/mean": 0.8279296159744263, "rewards/code_complexity_reward/std": 0.14884160459041595, "rewards/code_execution_reward/mean": 0.43359375, "rewards/code_execution_reward/std": 0.4960552453994751, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 295, "step_time": 77.22543529141694 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 162.0234375, "completions/mean_terminated_length": 161.3385467529297, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.18343262909911573, "epoch": 0.33751425313568983, "frac_reward_zero_std": 0.1875, "grad_norm": 0.037256523966789246, "kl": 0.10360482579562813, "learning_rate": 4.197824035112637e-06, "loss": 0.0005180733860470355, "num_tokens": 60891998.0, "reward": 2.263965129852295, "reward_std": 0.5606056451797485, "rewards/code_complexity_reward/mean": 0.8138672113418579, "rewards/code_complexity_reward/std": 0.15174485743045807, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 296, "step_time": 61.7051837425679 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 152.43359375, "completions/mean_terminated_length": 151.7299346923828, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.1801828306633979, "epoch": 0.338654503990878, "frac_reward_zero_std": 0.21875, "grad_norm": 0.03666107729077339, "kl": 0.08519767940742895, "learning_rate": 4.190503920292698e-06, "loss": 0.0004258530680090189, "num_tokens": 61039252.0, "reward": 2.25537109375, "reward_std": 0.5760492086410522, "rewards/code_complexity_reward/mean": 0.8208984136581421, "rewards/code_complexity_reward/std": 0.17188027501106262, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 297, "step_time": 88.79253713786602 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 150.939453125, "completions/mean_terminated_length": 147.37869262695312, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.17779929423704743, "epoch": 0.33979475484606614, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.03610488399863243, "kl": 0.09100483736256137, "learning_rate": 4.183157003834118e-06, "loss": 0.0004549172008410096, "num_tokens": 61184797.0, "reward": 2.339599609375, "reward_std": 0.5805456638336182, "rewards/code_complexity_reward/mean": 0.835253894329071, "rewards/code_complexity_reward/std": 0.14625906944274902, "rewards/code_execution_reward/mean": 0.416015625, "rewards/code_execution_reward/std": 0.493378221988678, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.021246958523988724, "step": 298, "step_time": 66.74903932213783 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 161.505859375, "completions/mean_terminated_length": 156.6475372314453, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.1915769560728222, "epoch": 0.3409350057012543, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.03509470447897911, "kl": 0.09503638878231868, "learning_rate": 4.175783402216604e-06, "loss": 0.00047512626042589545, "num_tokens": 61336192.0, "reward": 2.2711915969848633, "reward_std": 0.5895556807518005, "rewards/code_complexity_reward/mean": 0.8166992664337158, "rewards/code_complexity_reward/std": 0.17664363980293274, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.02048126794397831, "step": 299, "step_time": 63.81716666277498 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 470.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 149.876953125, "completions/mean_terminated_length": 149.876953125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.1806368539109826, "epoch": 0.34207525655644244, "frac_reward_zero_std": 0.25, "grad_norm": 0.03434237837791443, "kl": 0.09102017129771411, "learning_rate": 4.168383232342934e-06, "loss": 0.0004550249723251909, "num_tokens": 61480845.0, "reward": 2.3428711891174316, "reward_std": 0.5627202391624451, "rewards/code_complexity_reward/mean": 0.8397460579872131, "rewards/code_complexity_reward/std": 0.13766053318977356, "rewards/code_execution_reward/mean": 0.412109375, "rewards/code_execution_reward/std": 0.49269601702690125, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.030144967138767242, "step": 300, "step_time": 67.56325417198241 }, { "epoch": 0.34207525655644244, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0075, "eval_completions/max_length": 259.76, "eval_completions/max_terminated_length": 252.04, "eval_completions/mean_length": 154.28, "eval_completions/mean_terminated_length": 151.9546435546875, "eval_completions/min_length": 89.94, "eval_completions/min_terminated_length": 89.94, "eval_entropy": 0.18363240420818328, "eval_frac_reward_zero_std": 0.19, "eval_kl": 0.08952683206647634, "eval_loss": 0.00044860425987280905, "eval_num_tokens": 61480845.0, "eval_reward": 2.224937517642975, "eval_reward_std": 0.41487857021391394, "eval_rewards/code_complexity_reward/mean": 0.8258749961853027, "eval_rewards/code_complexity_reward/std": 0.10979426179081202, "eval_rewards/code_execution_reward/mean": 0.3125, "eval_rewards/code_execution_reward/std": 0.331284693479538, "eval_rewards/code_syntax_reward/mean": 0.48875, "eval_rewards/code_syntax_reward/std": 0.029377837479114533, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.4978125, "eval_rewards/xmlcount_reward_func/std": 0.006187184229493142, "eval_runtime": 558.6292, "eval_samples_per_second": 0.179, "eval_steps_per_second": 0.023, "step": 300 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 156.748046875, "completions/mean_terminated_length": 153.2445831298828, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.1843304841313511, "epoch": 0.34321550741163054, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.04004684463143349, "kl": 0.09416919114300981, "learning_rate": 4.160956611537106e-06, "loss": 0.000470909260911867, "num_tokens": 61630536.0, "reward": 2.2943358421325684, "reward_std": 0.5604719519615173, "rewards/code_complexity_reward/mean": 0.82666015625, "rewards/code_complexity_reward/std": 0.1482354700565338, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 301, "step_time": 56.99294387269765 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 434.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 147.95703125, "completions/mean_terminated_length": 147.95703125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.17992383427917957, "epoch": 0.3443557582668187, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.03754165396094322, "kl": 0.10916656104382128, "learning_rate": 4.153503657542479e-06, "loss": 0.0005455648060888052, "num_tokens": 61773326.0, "reward": 2.3327150344848633, "reward_std": 0.5333097577095032, "rewards/code_complexity_reward/mean": 0.8388671875, "rewards/code_complexity_reward/std": 0.11477487534284592, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 302, "step_time": 52.075482496991754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 147.607421875, "completions/mean_terminated_length": 146.89431762695312, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.17663776117842644, "epoch": 0.34549600912200684, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.038969025015830994, "kl": 0.10130784031935036, "learning_rate": 4.146024488519901e-06, "loss": 0.0005063934368081391, "num_tokens": 61916781.0, "reward": 2.3687500953674316, "reward_std": 0.5618606209754944, "rewards/code_complexity_reward/mean": 0.8483397960662842, "rewards/code_complexity_reward/std": 0.1316096931695938, "rewards/code_execution_reward/mean": 0.427734375, "rewards/code_execution_reward/std": 0.4952339828014374, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 303, "step_time": 77.65200370643288 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 151.328125, "completions/mean_terminated_length": 147.77120971679688, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.18194018979556859, "epoch": 0.346636259977195, "frac_reward_zero_std": 0.21875, "grad_norm": 0.03645700216293335, "kl": 0.11014609079575166, "learning_rate": 4.138519223045842e-06, "loss": 0.0005504468572326005, "num_tokens": 62062221.0, "reward": 2.2560548782348633, "reward_std": 0.5662969350814819, "rewards/code_complexity_reward/mean": 0.83447265625, "rewards/code_complexity_reward/std": 0.16089923679828644, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.032885149121284485, "step": 304, "step_time": 70.22358930390328 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 153.26171875, "completions/mean_terminated_length": 152.5596923828125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.18376672687008977, "epoch": 0.34777651083238315, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.0351661778986454, "kl": 0.0870819470146671, "learning_rate": 4.130987980110508e-06, "loss": 0.0004354008997324854, "num_tokens": 62209823.0, "reward": 2.301513671875, "reward_std": 0.5521961450576782, "rewards/code_complexity_reward/mean": 0.8386719226837158, "rewards/code_complexity_reward/std": 0.13504579663276672, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.027560751885175705, "step": 305, "step_time": 58.88717170339078 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 143.615234375, "completions/mean_terminated_length": 142.89431762695312, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.1720808253157884, "epoch": 0.34891676168757124, "frac_reward_zero_std": 0.203125, "grad_norm": 0.03737419471144676, "kl": 0.09997705760179088, "learning_rate": 4.123430879115963e-06, "loss": 0.000499906309414655, "num_tokens": 62351722.0, "reward": 2.270751953125, "reward_std": 0.5434117317199707, "rewards/code_complexity_reward/mean": 0.8352539539337158, "rewards/code_complexity_reward/std": 0.1380331963300705, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 306, "step_time": 81.65563834924251 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 145.2578125, "completions/mean_terminated_length": 144.5401153564453, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.175786197418347, "epoch": 0.3500570125427594, "frac_reward_zero_std": 0.234375, "grad_norm": 0.03551001846790314, "kl": 0.09300015354529023, "learning_rate": 4.115848039874225e-06, "loss": 0.00046490933164022863, "num_tokens": 62492518.0, "reward": 2.3814940452575684, "reward_std": 0.5599552989006042, "rewards/code_complexity_reward/mean": 0.8387695550918579, "rewards/code_complexity_reward/std": 0.13449130952358246, "rewards/code_execution_reward/mean": 0.44921875, "rewards/code_execution_reward/std": 0.497901052236557, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 307, "step_time": 68.75408702529967 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 153.28125, "completions/mean_terminated_length": 151.87451171875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.1754807336255908, "epoch": 0.35119726339794755, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.037293653935194016, "kl": 0.11616110673639923, "learning_rate": 4.108239582605374e-06, "loss": 0.0005808381829410791, "num_tokens": 62638642.0, "reward": 2.2987794876098633, "reward_std": 0.5514034032821655, "rewards/code_complexity_reward/mean": 0.8375976085662842, "rewards/code_complexity_reward/std": 0.1403360813856125, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03343535214662552, "step": 308, "step_time": 76.88416489306837 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 418.0, "completions/mean_length": 140.046875, "completions/mean_terminated_length": 139.31898498535156, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.1810116406995803, "epoch": 0.3523375142531357, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.03342399746179581, "kl": 0.09669352724449709, "learning_rate": 4.100605627935647e-06, "loss": 0.0004834646242670715, "num_tokens": 62777546.0, "reward": 2.36474609375, "reward_std": 0.5832355618476868, "rewards/code_complexity_reward/mean": 0.843066394329071, "rewards/code_complexity_reward/std": 0.1460697501897812, "rewards/code_execution_reward/mean": 0.435546875, "rewards/code_execution_reward/std": 0.49631330370903015, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.032946839928627014, "step": 309, "step_time": 51.2011554306373 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 486.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 146.177734375, "completions/mean_terminated_length": 146.177734375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.18265978328417987, "epoch": 0.35347776510832385, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.058114804327487946, "kl": 0.15553246054332703, "learning_rate": 4.0929462968955176e-06, "loss": 0.0007787410286255181, "num_tokens": 62921577.0, "reward": 2.3350586891174316, "reward_std": 0.5522892475128174, "rewards/code_complexity_reward/mean": 0.8421874642372131, "rewards/code_complexity_reward/std": 0.13248158991336823, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02333279326558113, "step": 310, "step_time": 64.02311707194895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 144.185546875, "completions/mean_terminated_length": 142.74314880371094, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.18185514875221997, "epoch": 0.35461801596351195, "frac_reward_zero_std": 0.25, "grad_norm": 0.03170058876276016, "kl": 0.10558754665544257, "learning_rate": 4.085261710917786e-06, "loss": 0.0005278007593005896, "num_tokens": 63063000.0, "reward": 2.2436037063598633, "reward_std": 0.5537537336349487, "rewards/code_complexity_reward/mean": 0.8392577767372131, "rewards/code_complexity_reward/std": 0.16305497288703918, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.027517380192875862, "step": 311, "step_time": 61.68309388682246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 145.087890625, "completions/mean_terminated_length": 144.36985778808594, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.18215069803409278, "epoch": 0.3557582668187001, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.038010258227586746, "kl": 0.096187005226966, "learning_rate": 4.0775519918356486e-06, "loss": 0.00048084347508847713, "num_tokens": 63207393.0, "reward": 2.2451171875, "reward_std": 0.5848696231842041, "rewards/code_complexity_reward/mean": 0.831835925579071, "rewards/code_complexity_reward/std": 0.18191777169704437, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.025862684473395348, "step": 312, "step_time": 76.95951562933624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 145.263671875, "completions/mean_terminated_length": 143.1021728515625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.17176020797342062, "epoch": 0.35689851767388825, "frac_reward_zero_std": 0.203125, "grad_norm": 0.03898897022008896, "kl": 0.10720018669962883, "learning_rate": 4.069817261880769e-06, "loss": 0.0005360671784728765, "num_tokens": 63350076.0, "reward": 2.274707078933716, "reward_std": 0.5386004447937012, "rewards/code_complexity_reward/mean": 0.8294921517372131, "rewards/code_complexity_reward/std": 0.13148754835128784, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.020545316860079765, "step": 313, "step_time": 59.28780959080905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 451.0, "completions/mean_length": 146.275390625, "completions/mean_terminated_length": 144.8411865234375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.1779502492863685, "epoch": 0.3580387685290764, "frac_reward_zero_std": 0.21875, "grad_norm": 0.039598070085048676, "kl": 0.09659276850288734, "learning_rate": 4.062057643681335e-06, "loss": 0.00048296665772795677, "num_tokens": 63495537.0, "reward": 2.2794432640075684, "reward_std": 0.5803733468055725, "rewards/code_complexity_reward/mean": 0.8280273675918579, "rewards/code_complexity_reward/std": 0.16127128899097443, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.03510142117738724, "step": 314, "step_time": 65.58251262176782 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 458.0, "completions/mean_length": 146.126953125, "completions/mean_terminated_length": 145.4109649658203, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.18157067720312625, "epoch": 0.35917901938426455, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.04209348186850548, "kl": 0.1051022203755565, "learning_rate": 4.054273260260125e-06, "loss": 0.0005254971329122782, "num_tokens": 63638270.0, "reward": 2.321044921875, "reward_std": 0.5636929273605347, "rewards/code_complexity_reward/mean": 0.83544921875, "rewards/code_complexity_reward/std": 0.1429634392261505, "rewards/code_execution_reward/mean": 0.396484375, "rewards/code_execution_reward/std": 0.4896455705165863, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.039319269359111786, "step": 315, "step_time": 60.10023773834109 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 486.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 162.138671875, "completions/mean_terminated_length": 162.138671875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.18618177226744592, "epoch": 0.3603192702394527, "frac_reward_zero_std": 0.234375, "grad_norm": 0.03666416183114052, "kl": 0.09965035045752302, "learning_rate": 4.046464235032546e-06, "loss": 0.0004982970422133803, "num_tokens": 63790365.0, "reward": 2.2631349563598633, "reward_std": 0.5517270565032959, "rewards/code_complexity_reward/mean": 0.831250011920929, "rewards/code_complexity_reward/std": 0.14608390629291534, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.025282343849539757, "step": 316, "step_time": 73.69783269334584 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 147.728515625, "completions/mean_terminated_length": 147.01565551757812, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.1777829211205244, "epoch": 0.3614595210946408, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.035193029791116714, "kl": 0.1094869555090554, "learning_rate": 4.0386306918046815e-06, "loss": 0.0005475811194628477, "num_tokens": 63932978.0, "reward": 2.31787109375, "reward_std": 0.5642916560173035, "rewards/code_complexity_reward/mean": 0.8423827886581421, "rewards/code_complexity_reward/std": 0.14988771080970764, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 317, "step_time": 57.40610258933157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 152.498046875, "completions/mean_terminated_length": 150.37918090820312, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.1758053310913965, "epoch": 0.36259977194982895, "frac_reward_zero_std": 0.15625, "grad_norm": 0.04635664448142052, "kl": 0.12215238471981138, "learning_rate": 4.0307727547713316e-06, "loss": 0.0006109464447945356, "num_tokens": 64079373.0, "reward": 2.2550292015075684, "reward_std": 0.5593781471252441, "rewards/code_complexity_reward/mean": 0.8333008289337158, "rewards/code_complexity_reward/std": 0.15842127799987793, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03250797092914581, "step": 318, "step_time": 80.88088291231543 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 151.412109375, "completions/mean_terminated_length": 150.70645141601562, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.17583925905637443, "epoch": 0.3637400228050171, "frac_reward_zero_std": 0.21875, "grad_norm": 0.03663669154047966, "kl": 0.11598581465659663, "learning_rate": 4.0228905485140415e-06, "loss": 0.0005798873608000576, "num_tokens": 64224816.0, "reward": 2.3151369094848633, "reward_std": 0.5595902800559998, "rewards/code_complexity_reward/mean": 0.831347644329071, "rewards/code_complexity_reward/std": 0.15387600660324097, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 319, "step_time": 59.124232738278806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 149.771484375, "completions/mean_terminated_length": 148.35098266601562, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.1892436717171222, "epoch": 0.36488027366020526, "frac_reward_zero_std": 0.21875, "grad_norm": 0.040138911455869675, "kl": 0.1051160802016966, "learning_rate": 4.014984197999125e-06, "loss": 0.0005255830474197865, "num_tokens": 64370647.0, "reward": 2.3375487327575684, "reward_std": 0.5565105676651001, "rewards/code_complexity_reward/mean": 0.8470703363418579, "rewards/code_complexity_reward/std": 0.1332317292690277, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03521692752838135, "step": 320, "step_time": 70.76105080917478 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 416.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 141.8125, "completions/mean_terminated_length": 141.8125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.18044898193329573, "epoch": 0.3660205245153934, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.036013659089803696, "kl": 0.10314117732923478, "learning_rate": 4.007053828575684e-06, "loss": 0.0005156917613931, "num_tokens": 64510799.0, "reward": 2.344482421875, "reward_std": 0.5904905796051025, "rewards/code_complexity_reward/mean": 0.84033203125, "rewards/code_complexity_reward/std": 0.1644812971353531, "rewards/code_execution_reward/mean": 0.419921875, "rewards/code_execution_reward/std": 0.4940285086631775, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 321, "step_time": 54.354924915358424 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 413.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 134.3515625, "completions/mean_terminated_length": 134.3515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.1833689750637859, "epoch": 0.3671607753705815, "frac_reward_zero_std": 0.234375, "grad_norm": 0.03994036093354225, "kl": 0.12095197133021429, "learning_rate": 3.999099565973623e-06, "loss": 0.0006047550705261528, "num_tokens": 64647955.0, "reward": 2.3040528297424316, "reward_std": 0.555306613445282, "rewards/code_complexity_reward/mean": 0.8478515148162842, "rewards/code_complexity_reward/std": 0.14154332876205444, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 322, "step_time": 52.13081456255168 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 144.525390625, "completions/mean_terminated_length": 143.08432006835938, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.1813739335630089, "epoch": 0.36830102622576966, "frac_reward_zero_std": 0.2109375, "grad_norm": 0.0365995354950428, "kl": 0.10282180306967348, "learning_rate": 3.991121536301653e-06, "loss": 0.0005141175934113562, "num_tokens": 64790680.0, "reward": 2.2750487327575684, "reward_std": 0.5545764565467834, "rewards/code_complexity_reward/mean": 0.839062511920929, "rewards/code_complexity_reward/std": 0.14915777742862701, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 323, "step_time": 62.39390226267278 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 156.619140625, "completions/mean_terminated_length": 155.22549438476562, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.1687436691718176, "epoch": 0.3694412770809578, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.03495791181921959, "kl": 0.09819076088024303, "learning_rate": 3.983119866045297e-06, "loss": 0.0004909251583740115, "num_tokens": 64938621.0, "reward": 2.306201219558716, "reward_std": 0.5694260001182556, "rewards/code_complexity_reward/mean": 0.8243163824081421, "rewards/code_complexity_reward/std": 0.14926277101039886, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.021246958523988724, "step": 324, "step_time": 60.34685722179711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 135.7734375, "completions/mean_terminated_length": 135.0371856689453, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.18197002005763352, "epoch": 0.37058152793614596, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.04186364635825157, "kl": 0.11120127356844023, "learning_rate": 3.975094682064875e-06, "loss": 0.0005560624413192272, "num_tokens": 65075105.0, "reward": 2.2804198265075684, "reward_std": 0.5659112930297852, "rewards/code_complexity_reward/mean": 0.85009765625, "rewards/code_complexity_reward/std": 0.15982191264629364, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 325, "step_time": 63.833049275912344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 150.720703125, "completions/mean_terminated_length": 150.01370239257812, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.17747458606027067, "epoch": 0.3717217787913341, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.03792896866798401, "kl": 0.0994430921273306, "learning_rate": 3.967046111593505e-06, "loss": 0.0004970278823748231, "num_tokens": 65220058.0, "reward": 2.2464356422424316, "reward_std": 0.5365808606147766, "rewards/code_complexity_reward/mean": 0.8385741710662842, "rewards/code_complexity_reward/std": 0.15678183734416962, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 326, "step_time": 56.563934955745935 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 135.572265625, "completions/mean_terminated_length": 132.60826110839844, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.1798736936179921, "epoch": 0.3728620296465222, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.0395401269197464, "kl": 0.12140081560937688, "learning_rate": 3.958974282235079e-06, "loss": 0.0006068919319659472, "num_tokens": 65357603.0, "reward": 2.302539110183716, "reward_std": 0.5748955011367798, "rewards/code_complexity_reward/mean": 0.8487304449081421, "rewards/code_complexity_reward/std": 0.15818610787391663, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.025821086019277573, "step": 327, "step_time": 65.55770174320787 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 138.404296875, "completions/mean_terminated_length": 138.404296875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.1849681786261499, "epoch": 0.37400228050171036, "frac_reward_zero_std": 0.25, "grad_norm": 0.041027847677469254, "kl": 0.10284460010007024, "learning_rate": 3.9508793219622375e-06, "loss": 0.0005142554873600602, "num_tokens": 65496534.0, "reward": 2.315722703933716, "reward_std": 0.5725469589233398, "rewards/code_complexity_reward/mean": 0.8428710699081421, "rewards/code_complexity_reward/std": 0.15796825289726257, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 328, "step_time": 55.269560519605875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 145.103515625, "completions/mean_terminated_length": 144.38551330566406, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.17713128379546106, "epoch": 0.3751425313568985, "frac_reward_zero_std": 0.21875, "grad_norm": 0.04187674820423126, "kl": 0.19872769777430221, "learning_rate": 3.942761359114345e-06, "loss": 0.0009930685628205538, "num_tokens": 65639747.0, "reward": 2.2716307640075684, "reward_std": 0.5337985754013062, "rewards/code_complexity_reward/mean": 0.8402343988418579, "rewards/code_complexity_reward/std": 0.13071510195732117, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.03438635915517807, "step": 329, "step_time": 56.99475202802569 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 507.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 145.603515625, "completions/mean_terminated_length": 145.603515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.1839770465157926, "epoch": 0.37628278221208666, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.03934166952967644, "kl": 0.12822935543954372, "learning_rate": 3.934620522395458e-06, "loss": 0.0006414442323148251, "num_tokens": 65782864.0, "reward": 2.2361817359924316, "reward_std": 0.5833369493484497, "rewards/code_complexity_reward/mean": 0.8207030892372131, "rewards/code_complexity_reward/std": 0.18466421961784363, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 330, "step_time": 98.81758610438555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 143.140625, "completions/mean_terminated_length": 140.23622131347656, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.1884652206208557, "epoch": 0.3774230330672748, "frac_reward_zero_std": 0.234375, "grad_norm": 0.03968772292137146, "kl": 0.12318839965155348, "learning_rate": 3.926456940872274e-06, "loss": 0.0006158493924885988, "num_tokens": 65925136.0, "reward": 2.3095216751098633, "reward_std": 0.5585756897926331, "rewards/code_complexity_reward/mean": 0.849609375, "rewards/code_complexity_reward/std": 0.1607828289270401, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.0376812107861042, "step": 331, "step_time": 50.14839053992182 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 138.265625, "completions/mean_terminated_length": 137.53424072265625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.18426200421527028, "epoch": 0.37856328392246297, "frac_reward_zero_std": 0.1875, "grad_norm": 0.037930045276880264, "kl": 0.11442378477659076, "learning_rate": 3.918270743972097e-06, "loss": 0.0005720682675018907, "num_tokens": 66065452.0, "reward": 2.284472703933716, "reward_std": 0.542267382144928, "rewards/code_complexity_reward/mean": 0.8462890386581421, "rewards/code_complexity_reward/std": 0.14980709552764893, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 332, "step_time": 60.20683410298079 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 419.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 141.19140625, "completions/mean_terminated_length": 141.19140625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.1809764530044049, "epoch": 0.37970353477765106, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.038628753274679184, "kl": 0.11604047752916813, "learning_rate": 3.910062061480778e-06, "loss": 0.0005800322396680713, "num_tokens": 66204118.0, "reward": 2.262939453125, "reward_std": 0.5213024616241455, "rewards/code_complexity_reward/mean": 0.850878894329071, "rewards/code_complexity_reward/std": 0.12879924476146698, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 333, "step_time": 52.442690894939005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 142.041015625, "completions/mean_terminated_length": 141.31703186035156, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.17685531126335263, "epoch": 0.3808437856328392, "frac_reward_zero_std": 0.2265625, "grad_norm": 0.04264665022492409, "kl": 0.11056002415716648, "learning_rate": 3.901831023540662e-06, "loss": 0.0005529189947992563, "num_tokens": 66346251.0, "reward": 2.322558641433716, "reward_std": 0.592394232749939, "rewards/code_complexity_reward/mean": 0.834765613079071, "rewards/code_complexity_reward/std": 0.17394451797008514, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03567269444465637, "step": 334, "step_time": 61.2326064389199 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 139.119140625, "completions/mean_terminated_length": 138.38943481445312, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.16944938385859132, "epoch": 0.38198403648802737, "frac_reward_zero_std": 0.265625, "grad_norm": 0.04338826239109039, "kl": 0.11495114071294665, "learning_rate": 3.89357776064852e-06, "loss": 0.000574595935177058, "num_tokens": 66486288.0, "reward": 2.330371141433716, "reward_std": 0.5714039206504822, "rewards/code_complexity_reward/mean": 0.8374999761581421, "rewards/code_complexity_reward/std": 0.1469520777463913, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02697930857539177, "step": 335, "step_time": 59.7849525520578 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 143.66796875, "completions/mean_terminated_length": 142.94715881347656, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.18211938021704555, "epoch": 0.3831242873432155, "frac_reward_zero_std": 0.28125, "grad_norm": 0.037040576338768005, "kl": 0.10986422607675195, "learning_rate": 3.885302403653483e-06, "loss": 0.0005492692580446601, "num_tokens": 66630426.0, "reward": 2.305469036102295, "reward_std": 0.5436044335365295, "rewards/code_complexity_reward/mean": 0.8470703363418579, "rewards/code_complexity_reward/std": 0.14521463215351105, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 336, "step_time": 62.85876890551299 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 139.13671875, "completions/mean_terminated_length": 136.93910217285156, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.1831680426839739, "epoch": 0.38426453819840367, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.03929983079433441, "kl": 0.13653136312495917, "learning_rate": 3.8770050837549675e-06, "loss": 0.0006826686440035701, "num_tokens": 66770912.0, "reward": 2.2859864234924316, "reward_std": 0.5748863220214844, "rewards/code_complexity_reward/mean": 0.8405272960662842, "rewards/code_complexity_reward/std": 0.16182397305965424, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 337, "step_time": 71.72713617701083 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 145.767578125, "completions/mean_terminated_length": 143.60903930664062, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.18312817765399814, "epoch": 0.38540478905359177, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.04341476783156395, "kl": 0.14692587457830086, "learning_rate": 3.868685932500596e-06, "loss": 0.0007349318475462496, "num_tokens": 66913125.0, "reward": 2.335156202316284, "reward_std": 0.5751579999923706, "rewards/code_complexity_reward/mean": 0.8390624523162842, "rewards/code_complexity_reward/std": 0.1524987518787384, "rewards/code_execution_reward/mean": 0.408203125, "rewards/code_execution_reward/std": 0.49198177456855774, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03647070750594139, "step": 338, "step_time": 85.99832464754581 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 135.171875, "completions/mean_terminated_length": 133.69412231445312, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.1760847669793293, "epoch": 0.3865450399087799, "frac_reward_zero_std": 0.265625, "grad_norm": 0.03998004272580147, "kl": 0.13560570415575057, "learning_rate": 3.860345081784107e-06, "loss": 0.0006782247219234705, "num_tokens": 67051205.0, "reward": 2.2916016578674316, "reward_std": 0.5578761696815491, "rewards/code_complexity_reward/mean": 0.8397460579872131, "rewards/code_complexity_reward/std": 0.14886173605918884, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.024608410894870758, "step": 339, "step_time": 77.15905260667205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 134.59765625, "completions/mean_terminated_length": 133.11766052246094, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.17590574943460524, "epoch": 0.38768529076396807, "frac_reward_zero_std": 0.34375, "grad_norm": 0.04096691310405731, "kl": 0.11780810495838523, "learning_rate": 3.851982663843272e-06, "loss": 0.0005889898748137057, "num_tokens": 67186715.0, "reward": 2.3781251907348633, "reward_std": 0.5707374215126038, "rewards/code_complexity_reward/mean": 0.86474609375, "rewards/code_complexity_reward/std": 0.1507720947265625, "rewards/code_execution_reward/mean": 0.423828125, "rewards/code_execution_reward/std": 0.4946470856666565, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 340, "step_time": 85.14327089395374 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 137.876953125, "completions/mean_terminated_length": 136.40980529785156, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.1728229825384915, "epoch": 0.3888255416191562, "frac_reward_zero_std": 0.265625, "grad_norm": 0.04643494635820389, "kl": 0.12576501641888171, "learning_rate": 3.84359881125779e-06, "loss": 0.0006290701567195356, "num_tokens": 67327056.0, "reward": 2.3130860328674316, "reward_std": 0.5598824620246887, "rewards/code_complexity_reward/mean": 0.8407226204872131, "rewards/code_complexity_reward/std": 0.15281720459461212, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02697930857539177, "step": 341, "step_time": 75.99976966436952 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 140.09765625, "completions/mean_terminated_length": 139.36985778808594, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.17908683535642922, "epoch": 0.3899657924743444, "frac_reward_zero_std": 0.28125, "grad_norm": 0.0392393134534359, "kl": 0.1354942535981536, "learning_rate": 3.835193656947192e-06, "loss": 0.0006772298365831375, "num_tokens": 67465726.0, "reward": 2.335156202316284, "reward_std": 0.563262403011322, "rewards/code_complexity_reward/mean": 0.849414050579071, "rewards/code_complexity_reward/std": 0.14604923129081726, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843363426625729, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.0347534641623497, "step": 342, "step_time": 60.192975403741 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 434.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 134.435546875, "completions/mean_terminated_length": 134.435546875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.18361955741420388, "epoch": 0.39110604332953247, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.03940524905920029, "kl": 0.13225679838797078, "learning_rate": 3.826767334168731e-06, "loss": 0.0006614226149395108, "num_tokens": 67601737.0, "reward": 2.3833985328674316, "reward_std": 0.5675700902938843, "rewards/code_complexity_reward/mean": 0.8548828363418579, "rewards/code_complexity_reward/std": 0.14157895743846893, "rewards/code_execution_reward/mean": 0.4375, "rewards/code_execution_reward/std": 0.49656352400779724, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.024554960429668427, "step": 343, "step_time": 55.114684231579304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 508.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 137.208984375, "completions/mean_terminated_length": 137.208984375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.17973351012915373, "epoch": 0.3922462941847206, "frac_reward_zero_std": 0.296875, "grad_norm": 0.05079783871769905, "kl": 0.225426280987449, "learning_rate": 3.8183199765152704e-06, "loss": 0.0011271695839241147, "num_tokens": 67741104.0, "reward": 2.32373046875, "reward_std": 0.5510871410369873, "rewards/code_complexity_reward/mean": 0.850878894329071, "rewards/code_complexity_reward/std": 0.15016040205955505, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 344, "step_time": 50.16357114445418 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 489.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 144.14453125, "completions/mean_terminated_length": 144.14453125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.18334122258238494, "epoch": 0.3933865450399088, "frac_reward_zero_std": 0.265625, "grad_norm": 0.04426350072026253, "kl": 0.12135374057106674, "learning_rate": 3.809851717913164e-06, "loss": 0.0006066829664632678, "num_tokens": 67882830.0, "reward": 2.2919435501098633, "reward_std": 0.5272095203399658, "rewards/code_complexity_reward/mean": 0.8547852039337158, "rewards/code_complexity_reward/std": 0.1382783204317093, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 345, "step_time": 64.62868870142847 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 140.115234375, "completions/mean_terminated_length": 139.38748168945312, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.185214206809178, "epoch": 0.3945267958950969, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.03761621192097664, "kl": 0.1276192672085017, "learning_rate": 3.8013626926201343e-06, "loss": 0.0006381875136867166, "num_tokens": 68022693.0, "reward": 2.33056640625, "reward_std": 0.5652296543121338, "rewards/code_complexity_reward/mean": 0.841796875, "rewards/code_complexity_reward/std": 0.15480273962020874, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.031092895194888115, "step": 346, "step_time": 57.368430708535016 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 504.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 134.5546875, "completions/mean_terminated_length": 134.5546875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.18906808248721063, "epoch": 0.3956670467502851, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.04315173998475075, "kl": 0.12340735550969839, "learning_rate": 3.792853035223144e-06, "loss": 0.0006170420674607158, "num_tokens": 68159029.0, "reward": 2.2496581077575684, "reward_std": 0.5411457419395447, "rewards/code_complexity_reward/mean": 0.854199230670929, "rewards/code_complexity_reward/std": 0.1590607911348343, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 347, "step_time": 87.29698855616152 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 137.322265625, "completions/mean_terminated_length": 136.5890350341797, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.18080420792102814, "epoch": 0.39680729760547323, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.04275553673505783, "kl": 0.11636855511460453, "learning_rate": 3.7843228806362635e-06, "loss": 0.0005819030338898301, "num_tokens": 68296274.0, "reward": 2.3270020484924316, "reward_std": 0.5315760374069214, "rewards/code_complexity_reward/mean": 0.858105480670929, "rewards/code_complexity_reward/std": 0.1132499948143959, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 348, "step_time": 76.90898245573044 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 133.978515625, "completions/mean_terminated_length": 133.23873901367188, "completions/min_length": 8.0, "completions/min_terminated_length": 8.0, "entropy": 0.17845030734315515, "epoch": 0.3979475484606613, "frac_reward_zero_std": 0.234375, "grad_norm": 0.046260908246040344, "kl": 0.12399074528366327, "learning_rate": 3.775772364098529e-06, "loss": 0.0006199270719662309, "num_tokens": 68431687.0, "reward": 2.301074266433716, "reward_std": 0.5623188018798828, "rewards/code_complexity_reward/mean": 0.8578125238418579, "rewards/code_complexity_reward/std": 0.1442895382642746, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843363426625729, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.0396316759288311, "step": 349, "step_time": 69.04581289086491 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 134.341796875, "completions/mean_terminated_length": 133.6027374267578, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.17135611234698445, "epoch": 0.3990877993158495, "frac_reward_zero_std": 0.3125, "grad_norm": 0.04020734503865242, "kl": 0.14945140737108886, "learning_rate": 3.7672016211717977e-06, "loss": 0.0007475618040189147, "num_tokens": 68568602.0, "reward": 2.26708984375, "reward_std": 0.5519165992736816, "rewards/code_complexity_reward/mean": 0.844042956829071, "rewards/code_complexity_reward/std": 0.155897319316864, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.025862684473395348, "step": 350, "step_time": 56.698882705532014 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 129.060546875, "completions/mean_terminated_length": 129.060546875, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.1854087132960558, "epoch": 0.40022805017103763, "frac_reward_zero_std": 0.296875, "grad_norm": 0.039427585899829865, "kl": 0.14195412560366094, "learning_rate": 3.758610787738604e-06, "loss": 0.0007095603505149484, "num_tokens": 68703461.0, "reward": 2.2968263626098633, "reward_std": 0.537082314491272, "rewards/code_complexity_reward/mean": 0.8656249642372131, "rewards/code_complexity_reward/std": 0.13770554959774017, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.0376812107861042, "step": 351, "step_time": 47.735771836712956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 439.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 134.544921875, "completions/mean_terminated_length": 134.544921875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.17535269656218588, "epoch": 0.4013683010262258, "frac_reward_zero_std": 0.296875, "grad_norm": 0.038325924426317215, "kl": 0.12722041946835816, "learning_rate": 3.7500000000000005e-06, "loss": 0.0006359751569107175, "num_tokens": 68839684.0, "reward": 2.260205030441284, "reward_std": 0.5325575470924377, "rewards/code_complexity_reward/mean": 0.857617199420929, "rewards/code_complexity_reward/std": 0.1443677842617035, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.03260336071252823, "step": 352, "step_time": 52.40127022378147 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 453.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 128.5, "completions/mean_terminated_length": 128.5, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.17710350966081023, "epoch": 0.40250855188141393, "frac_reward_zero_std": 0.328125, "grad_norm": 0.04353522136807442, "kl": 0.14330574846826494, "learning_rate": 3.7413693944734e-06, "loss": 0.0007163699483498931, "num_tokens": 68971996.0, "reward": 2.3550782203674316, "reward_std": 0.5609378814697266, "rewards/code_complexity_reward/mean": 0.8622069954872131, "rewards/code_complexity_reward/std": 0.14884555339813232, "rewards/code_execution_reward/mean": 0.40234375, "rewards/code_execution_reward/std": 0.4908501207828522, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.019099153578281403, "step": 353, "step_time": 56.703766698017716 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 128.11328125, "completions/mean_terminated_length": 127.3620376586914, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.184057813603431, "epoch": 0.40364880273660203, "frac_reward_zero_std": 0.28125, "grad_norm": 0.039592355489730835, "kl": 0.1344603926409036, "learning_rate": 3.7327191079904096e-06, "loss": 0.0006721469108015299, "num_tokens": 69105102.0, "reward": 2.2960450649261475, "reward_std": 0.5551729202270508, "rewards/code_complexity_reward/mean": 0.85888671875, "rewards/code_complexity_reward/std": 0.1588674634695053, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 354, "step_time": 65.74647424276918 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 139.42578125, "completions/mean_terminated_length": 137.96470642089844, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.19121414702385664, "epoch": 0.4047890535917902, "frac_reward_zero_std": 0.1953125, "grad_norm": 0.048448920249938965, "kl": 0.13491533463820815, "learning_rate": 3.7240492776946663e-06, "loss": 0.0006744756828993559, "num_tokens": 69243080.0, "reward": 2.2601075172424316, "reward_std": 0.5570897459983826, "rewards/code_complexity_reward/mean": 0.8373047113418579, "rewards/code_complexity_reward/std": 0.170400008559227, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.022693097591400146, "step": 355, "step_time": 69.97066642809659 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 136.501953125, "completions/mean_terminated_length": 135.0294189453125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.17900485917925835, "epoch": 0.40592930444697833, "frac_reward_zero_std": 0.25, "grad_norm": 0.042443305253982544, "kl": 0.14570557058323175, "learning_rate": 3.7153600410396558e-06, "loss": 0.0007283228915184736, "num_tokens": 69380177.0, "reward": 2.342041015625, "reward_std": 0.5791563391685486, "rewards/code_complexity_reward/mean": 0.849902331829071, "rewards/code_complexity_reward/std": 0.15719841420650482, "rewards/code_execution_reward/mean": 0.404296875, "rewards/code_execution_reward/std": 0.4912354052066803, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.01981583796441555, "step": 356, "step_time": 66.451556394808 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 133.21484375, "completions/mean_terminated_length": 132.4735870361328, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.17909494508057833, "epoch": 0.4070695553021665, "frac_reward_zero_std": 0.28125, "grad_norm": 0.039044689387083054, "kl": 0.13175025896634907, "learning_rate": 3.7066515357865384e-06, "loss": 0.0006586962263099849, "num_tokens": 69519775.0, "reward": 2.271923780441284, "reward_std": 0.5908604264259338, "rewards/code_complexity_reward/mean": 0.8427734375, "rewards/code_complexity_reward/std": 0.1840948760509491, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 357, "step_time": 66.7346915518865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 136.78515625, "completions/mean_terminated_length": 134.5736846923828, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.18364982958883047, "epoch": 0.40820980615735464, "frac_reward_zero_std": 0.359375, "grad_norm": 0.03923090174794197, "kl": 0.13072354451287538, "learning_rate": 3.6979239000019622e-06, "loss": 0.0006534787244163454, "num_tokens": 69657749.0, "reward": 2.262695550918579, "reward_std": 0.5410904288291931, "rewards/code_complexity_reward/mean": 0.8609374761581421, "rewards/code_complexity_reward/std": 0.15717509388923645, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02465205453336239, "step": 358, "step_time": 58.991739535704255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 136.763671875, "completions/mean_terminated_length": 135.2921600341797, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.18472468294203281, "epoch": 0.40935005701254273, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.04106444492936134, "kl": 0.13142212992534041, "learning_rate": 3.689177272055877e-06, "loss": 0.0006568831740878522, "num_tokens": 69798280.0, "reward": 2.2532715797424316, "reward_std": 0.5523613095283508, "rewards/code_complexity_reward/mean": 0.8509765863418579, "rewards/code_complexity_reward/std": 0.16111470758914948, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 359, "step_time": 61.61036565434188 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 133.98046875, "completions/mean_terminated_length": 132.498046875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.17830763151869178, "epoch": 0.4104903078677309, "frac_reward_zero_std": 0.2421875, "grad_norm": 0.04209123179316521, "kl": 0.1306033075088635, "learning_rate": 3.6804117906193367e-06, "loss": 0.0006529727252200246, "num_tokens": 69935314.0, "reward": 2.325976610183716, "reward_std": 0.5785627365112305, "rewards/code_complexity_reward/mean": 0.851855456829071, "rewards/code_complexity_reward/std": 0.16969949007034302, "rewards/code_execution_reward/mean": 0.390625, "rewards/code_execution_reward/std": 0.48836761713027954, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02697930857539177, "step": 360, "step_time": 54.50802757963538 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 387.0, "completions/mean_length": 135.859375, "completions/mean_terminated_length": 131.3992156982422, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.1828588997013867, "epoch": 0.41163055872291904, "frac_reward_zero_std": 0.265625, "grad_norm": 0.04134120047092438, "kl": 0.15249922138173133, "learning_rate": 3.671627594662303e-06, "loss": 0.0007622989942319691, "num_tokens": 70073058.0, "reward": 2.2568359375, "reward_std": 0.574781596660614, "rewards/code_complexity_reward/mean": 0.841503918170929, "rewards/code_complexity_reward/std": 0.17620450258255005, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.036414988338947296, "step": 361, "step_time": 61.65806740988046 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 394.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 127.390625, "completions/mean_terminated_length": 127.390625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.18455844197887927, "epoch": 0.4127708095781072, "frac_reward_zero_std": 0.2578125, "grad_norm": 0.047279905527830124, "kl": 0.1543070049956441, "learning_rate": 3.6628248234514434e-06, "loss": 0.0007712772348895669, "num_tokens": 70207042.0, "reward": 2.3034181594848633, "reward_std": 0.5322482585906982, "rewards/code_complexity_reward/mean": 0.8684570789337158, "rewards/code_complexity_reward/std": 0.13173596560955048, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.025862684473395348, "step": 362, "step_time": 50.32021019048989 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 130.8359375, "completions/mean_terminated_length": 129.3411865234375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.18328645825386047, "epoch": 0.41391106043329534, "frac_reward_zero_std": 0.3125, "grad_norm": 0.04586375132203102, "kl": 0.14451999391894788, "learning_rate": 3.6540036165479203e-06, "loss": 0.000722619122825563, "num_tokens": 70343202.0, "reward": 2.2494630813598633, "reward_std": 0.5148584842681885, "rewards/code_complexity_reward/mean": 0.858203113079071, "rewards/code_complexity_reward/std": 0.1332312971353531, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 363, "step_time": 52.54512592963874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 130.85546875, "completions/mean_terminated_length": 130.10958862304688, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.1786373956128955, "epoch": 0.4150513112884835, "frac_reward_zero_std": 0.296875, "grad_norm": 0.04396289214491844, "kl": 0.1250297516817227, "learning_rate": 3.6451641138051806e-06, "loss": 0.0006253418978303671, "num_tokens": 70479532.0, "reward": 2.3041017055511475, "reward_std": 0.5656272768974304, "rewards/code_complexity_reward/mean": 0.85546875, "rewards/code_complexity_reward/std": 0.16280736029148102, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.024554960429668427, "step": 364, "step_time": 90.74921915214509 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 132.73046875, "completions/mean_terminated_length": 131.98825073242188, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.18072471674531698, "epoch": 0.4161915621436716, "frac_reward_zero_std": 0.34375, "grad_norm": 0.03672073408961296, "kl": 0.14770546602085233, "learning_rate": 3.6363064553667378e-06, "loss": 0.0007386663346551359, "num_tokens": 70614838.0, "reward": 2.325732469558716, "reward_std": 0.5531162619590759, "rewards/code_complexity_reward/mean": 0.8592773079872131, "rewards/code_complexity_reward/std": 0.14684171974658966, "rewards/code_execution_reward/mean": 0.376953125, "rewards/code_execution_reward/std": 0.4850969910621643, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 365, "step_time": 59.07554939202964 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 132.30078125, "completions/mean_terminated_length": 131.55772399902344, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.18060655216686428, "epoch": 0.41733181299885974, "frac_reward_zero_std": 0.2734375, "grad_norm": 0.03944363817572594, "kl": 0.1394879954168573, "learning_rate": 3.627430781663948e-06, "loss": 0.0006974421557970345, "num_tokens": 70751376.0, "reward": 2.2862305641174316, "reward_std": 0.5247223377227783, "rewards/code_complexity_reward/mean": 0.8600585460662842, "rewards/code_complexity_reward/std": 0.13067413866519928, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.019055325537919998, "step": 366, "step_time": 58.01570801716298 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.013671875, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 135.751953125, "completions/mean_terminated_length": 130.53663635253906, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.18377566221170127, "epoch": 0.4184720638540479, "frac_reward_zero_std": 0.25, "grad_norm": 0.041616279631853104, "kl": 0.1459845006465912, "learning_rate": 3.618537233413789e-06, "loss": 0.0007297039846889675, "num_tokens": 70889105.0, "reward": 2.3224120140075684, "reward_std": 0.5961569547653198, "rewards/code_complexity_reward/mean": 0.8472656011581421, "rewards/code_complexity_reward/std": 0.17087990045547485, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03149271756410599, "step": 367, "step_time": 69.71899533923715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 132.62890625, "completions/mean_terminated_length": 131.88648986816406, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.18523633712902665, "epoch": 0.41961231470923605, "frac_reward_zero_std": 0.3125, "grad_norm": 0.04267369955778122, "kl": 0.1487955718766898, "learning_rate": 3.6096259516166226e-06, "loss": 0.0007439666660502553, "num_tokens": 71025567.0, "reward": 2.2428712844848633, "reward_std": 0.5410885810852051, "rewards/code_complexity_reward/mean": 0.8541991710662842, "rewards/code_complexity_reward/std": 0.15291346609592438, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 368, "step_time": 58.23848411720246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 130.587890625, "completions/mean_terminated_length": 129.84149169921875, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.18904644972644746, "epoch": 0.4207525655644242, "frac_reward_zero_std": 0.265625, "grad_norm": 0.044098492711782455, "kl": 0.13476943760178983, "learning_rate": 3.600697077553964e-06, "loss": 0.0006739358650520444, "num_tokens": 71160732.0, "reward": 2.2166991233825684, "reward_std": 0.5209828615188599, "rewards/code_complexity_reward/mean": 0.858593761920929, "rewards/code_complexity_reward/std": 0.15236033499240875, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.024608410894870758, "step": 369, "step_time": 50.0125925578177 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 503.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 121.66796875, "completions/mean_terminated_length": 121.66796875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.18805115832947195, "epoch": 0.4218928164196123, "frac_reward_zero_std": 0.28125, "grad_norm": 0.05152072384953499, "kl": 0.15572794433683157, "learning_rate": 3.5917507527862394e-06, "loss": 0.0007785988273099065, "num_tokens": 71290542.0, "reward": 2.2990236282348633, "reward_std": 0.5088853240013123, "rewards/code_complexity_reward/mean": 0.8779296875, "rewards/code_complexity_reward/std": 0.12050614506006241, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.025862684473395348, "step": 370, "step_time": 66.624246744439 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 138.611328125, "completions/mean_terminated_length": 137.14706420898438, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.18871800135821104, "epoch": 0.42303306727480045, "frac_reward_zero_std": 0.28125, "grad_norm": 0.04262511804699898, "kl": 0.14619566418696195, "learning_rate": 3.5827871191505425e-06, "loss": 0.0007310936343856156, "num_tokens": 71429115.0, "reward": 2.2305665016174316, "reward_std": 0.554293692111969, "rewards/code_complexity_reward/mean": 0.8501952886581421, "rewards/code_complexity_reward/std": 0.16966630518436432, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.032150521874427795, "step": 371, "step_time": 69.04733058065176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 131.57421875, "completions/mean_terminated_length": 130.08236694335938, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.1840219055302441, "epoch": 0.4241733181299886, "frac_reward_zero_std": 0.359375, "grad_norm": 0.046318419277668, "kl": 0.1476511696819216, "learning_rate": 3.573806318758388e-06, "loss": 0.0007382620242424309, "num_tokens": 71564953.0, "reward": 2.2190918922424316, "reward_std": 0.5448774099349976, "rewards/code_complexity_reward/mean": 0.8514648675918579, "rewards/code_complexity_reward/std": 0.17007768154144287, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03343535214662552, "step": 372, "step_time": 71.95006292127073 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 494.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 123.958984375, "completions/mean_terminated_length": 123.958984375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.18941315077245235, "epoch": 0.42531356898517675, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.04808053746819496, "kl": 0.20332771248649806, "learning_rate": 3.5648084939934523e-06, "loss": 0.0010163006372749805, "num_tokens": 71696580.0, "reward": 2.3343262672424316, "reward_std": 0.5494568943977356, "rewards/code_complexity_reward/mean": 0.874804675579071, "rewards/code_complexity_reward/std": 0.13936519622802734, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.041540902107954025, "step": 373, "step_time": 57.80078539159149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 464.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 130.603515625, "completions/mean_terminated_length": 130.603515625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.17553611495532095, "epoch": 0.4264538198403649, "frac_reward_zero_std": 0.3359375, "grad_norm": 0.041353754699230194, "kl": 0.13030734925996512, "learning_rate": 3.5557937875093242e-06, "loss": 0.00065154570620507, "num_tokens": 71829437.0, "reward": 2.305712938308716, "reward_std": 0.5586168169975281, "rewards/code_complexity_reward/mean": 0.8594726324081421, "rewards/code_complexity_reward/std": 0.15477752685546875, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 374, "step_time": 55.58223391324282 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 127.66015625, "completions/mean_terminated_length": 126.90802001953125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.19052471267059445, "epoch": 0.427594070695553, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.03813108429312706, "kl": 0.16739734273869544, "learning_rate": 3.5467623422272353e-06, "loss": 0.0008367928676307201, "num_tokens": 71962763.0, "reward": 2.274462938308716, "reward_std": 0.5342876315116882, "rewards/code_complexity_reward/mean": 0.864550769329071, "rewards/code_complexity_reward/std": 0.14667946100234985, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.0385771282017231, "step": 375, "step_time": 58.75610865931958 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 413.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 126.017578125, "completions/mean_terminated_length": 126.017578125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.18222700618207455, "epoch": 0.42873432155074115, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.046756017953157425, "kl": 0.13921050017233938, "learning_rate": 3.537714301333801e-06, "loss": 0.0006960374303162098, "num_tokens": 72095364.0, "reward": 2.2861328125, "reward_std": 0.5670787692070007, "rewards/code_complexity_reward/mean": 0.8599609136581421, "rewards/code_complexity_reward/std": 0.17331349849700928, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 376, "step_time": 62.89314012043178 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 132.693359375, "completions/mean_terminated_length": 130.457763671875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.18516724731307477, "epoch": 0.4298745724059293, "frac_reward_zero_std": 0.328125, "grad_norm": 0.040884289890527725, "kl": 0.15134802681859583, "learning_rate": 3.528649808278747e-06, "loss": 0.0007568774744868279, "num_tokens": 72231975.0, "reward": 2.2443361282348633, "reward_std": 0.5591654777526855, "rewards/code_complexity_reward/mean": 0.8551758527755737, "rewards/code_complexity_reward/std": 0.16867610812187195, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03391507267951965, "step": 377, "step_time": 99.01133030653 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 121.689453125, "completions/mean_terminated_length": 120.9256362915039, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.18634209339506924, "epoch": 0.43101482326111745, "frac_reward_zero_std": 0.390625, "grad_norm": 0.03902257978916168, "kl": 0.14981501130387187, "learning_rate": 3.519569006772633e-06, "loss": 0.0007491194410249591, "num_tokens": 72362900.0, "reward": 2.3131837844848633, "reward_std": 0.5633928179740906, "rewards/code_complexity_reward/mean": 0.867968738079071, "rewards/code_complexity_reward/std": 0.16349706053733826, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 378, "step_time": 49.43754382338375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 121.501953125, "completions/mean_terminated_length": 120.7377700805664, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.19270320027135313, "epoch": 0.4321550741163056, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.04610256850719452, "kl": 0.14775455079507083, "learning_rate": 3.5104720407845794e-06, "loss": 0.0007385470671579242, "num_tokens": 72492077.0, "reward": 2.2808594703674316, "reward_std": 0.5359961986541748, "rewards/code_complexity_reward/mean": 0.8698241710662842, "rewards/code_complexity_reward/std": 0.15161415934562683, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 379, "step_time": 50.07234184164554 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 119.849609375, "completions/mean_terminated_length": 119.08219146728516, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.1872581783682108, "epoch": 0.43329532497149376, "frac_reward_zero_std": 0.390625, "grad_norm": 0.04405009374022484, "kl": 0.16829235898330808, "learning_rate": 3.5013590545399818e-06, "loss": 0.0008414685726165771, "num_tokens": 72621616.0, "reward": 2.26904296875, "reward_std": 0.5309534668922424, "rewards/code_complexity_reward/mean": 0.8692382574081421, "rewards/code_complexity_reward/std": 0.15023140609264374, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 380, "step_time": 58.172806962393224 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 474.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 121.64453125, "completions/mean_terminated_length": 121.64453125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.1848908041138202, "epoch": 0.43443557582668185, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.04077712818980217, "kl": 0.16700799076352268, "learning_rate": 3.492230192518221e-06, "loss": 0.0008349134004674852, "num_tokens": 72751322.0, "reward": 2.36474609375, "reward_std": 0.5398048162460327, "rewards/code_complexity_reward/mean": 0.880175769329071, "rewards/code_complexity_reward/std": 0.12180382758378983, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.025862684473395348, "step": 381, "step_time": 48.546042942442 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 491.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 123.24609375, "completions/mean_terminated_length": 123.24609375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.18980969884432852, "epoch": 0.43557582668187, "frac_reward_zero_std": 0.328125, "grad_norm": 0.04104543477296829, "kl": 0.16635115444660187, "learning_rate": 3.483085599450381e-06, "loss": 0.0008316098246723413, "num_tokens": 72882312.0, "reward": 2.298877239227295, "reward_std": 0.5091553330421448, "rewards/code_complexity_reward/mean": 0.879199206829071, "rewards/code_complexity_reward/std": 0.1198577955365181, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 382, "step_time": 66.0089364927262 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 122.861328125, "completions/mean_terminated_length": 122.09980010986328, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.18688723188824952, "epoch": 0.43671607753705816, "frac_reward_zero_std": 0.375, "grad_norm": 0.04039694741368294, "kl": 0.17993255506735295, "learning_rate": 3.473925420316946e-06, "loss": 0.0008999048732221127, "num_tokens": 73012669.0, "reward": 2.368701219558716, "reward_std": 0.5445497035980225, "rewards/code_complexity_reward/mean": 0.8794921636581421, "rewards/code_complexity_reward/std": 0.13501226902008057, "rewards/code_execution_reward/mean": 0.3984375, "rewards/code_execution_reward/std": 0.4900552034378052, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.025244520977139473, "step": 383, "step_time": 58.040107912383974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 117.599609375, "completions/mean_terminated_length": 116.82778930664062, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.18656076351180673, "epoch": 0.4378563283922463, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.043250780552625656, "kl": 0.1697742516407743, "learning_rate": 3.464749800345507e-06, "loss": 0.0008488300372846425, "num_tokens": 73142392.0, "reward": 2.258056640625, "reward_std": 0.5241289734840393, "rewards/code_complexity_reward/mean": 0.8775390386581421, "rewards/code_complexity_reward/std": 0.1452893614768982, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 384, "step_time": 49.10049186553806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 471.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 131.2421875, "completions/mean_terminated_length": 131.2421875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.18278058990836143, "epoch": 0.43899657924743446, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.04140820726752281, "kl": 0.15074389590881765, "learning_rate": 3.4555588850084575e-06, "loss": 0.000753608881495893, "num_tokens": 73278308.0, "reward": 2.268847942352295, "reward_std": 0.5239389538764954, "rewards/code_complexity_reward/mean": 0.8628906011581421, "rewards/code_complexity_reward/std": 0.14224866032600403, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.019099153578281403, "step": 385, "step_time": 46.34156478382647 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 500.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 120.39453125, "completions/mean_terminated_length": 120.39453125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.18758688680827618, "epoch": 0.44013683010262256, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.0400676354765892, "kl": 0.16425067675299942, "learning_rate": 3.4463528200206868e-06, "loss": 0.0008212362299673259, "num_tokens": 73408562.0, "reward": 2.3549318313598633, "reward_std": 0.5650613903999329, "rewards/code_complexity_reward/mean": 0.8727538585662842, "rewards/code_complexity_reward/std": 0.14756613969802856, "rewards/code_execution_reward/mean": 0.39453125, "rewards/code_execution_reward/std": 0.4892277717590332, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.030623575672507286, "step": 386, "step_time": 56.59926247037947 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 129.966796875, "completions/mean_terminated_length": 129.21917724609375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.1823364517185837, "epoch": 0.4412770809578107, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.044509854167699814, "kl": 0.14605205587577075, "learning_rate": 3.4371317513372692e-06, "loss": 0.0007302571902982891, "num_tokens": 73544241.0, "reward": 2.2676756381988525, "reward_std": 0.5483028888702393, "rewards/code_complexity_reward/mean": 0.8555663824081421, "rewards/code_complexity_reward/std": 0.16245809197425842, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02465205453336239, "step": 387, "step_time": 52.26677996944636 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 125.888671875, "completions/mean_terminated_length": 125.13307189941406, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.18011628068052232, "epoch": 0.44241733181299886, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.042429592460393906, "kl": 0.14929365122225136, "learning_rate": 3.427895825151153e-06, "loss": 0.000746397185139358, "num_tokens": 73675264.0, "reward": 2.282421827316284, "reward_std": 0.559136152267456, "rewards/code_complexity_reward/mean": 0.86083984375, "rewards/code_complexity_reward/std": 0.16213074326515198, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.028089813888072968, "step": 388, "step_time": 79.80228346027434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 464.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 118.37109375, "completions/mean_terminated_length": 118.37109375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.18635266576893628, "epoch": 0.443557582668187, "frac_reward_zero_std": 0.296875, "grad_norm": 0.0425318107008934, "kl": 0.17530523077584803, "learning_rate": 3.4186451878908393e-06, "loss": 0.0008767056278884411, "num_tokens": 73804882.0, "reward": 2.3683106899261475, "reward_std": 0.550362765789032, "rewards/code_complexity_reward/mean": 0.8776366710662842, "rewards/code_complexity_reward/std": 0.12791754305362701, "rewards/code_execution_reward/mean": 0.400390625, "rewards/code_execution_reward/std": 0.4904567301273346, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03521692752838135, "step": 389, "step_time": 47.18408346362412 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 125.4140625, "completions/mean_terminated_length": 123.8980484008789, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.19021382206119597, "epoch": 0.44469783352337516, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.042318765074014664, "kl": 0.14867751114070415, "learning_rate": 3.4093799862180627e-06, "loss": 0.0007431576959788799, "num_tokens": 73937582.0, "reward": 2.282275438308716, "reward_std": 0.5584977865219116, "rewards/code_complexity_reward/mean": 0.8611328601837158, "rewards/code_complexity_reward/std": 0.1589566171169281, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.030670344829559326, "step": 390, "step_time": 89.59999470971525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 486.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 125.94140625, "completions/mean_terminated_length": 125.94140625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.19446942675858736, "epoch": 0.44583808437856326, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.04609624296426773, "kl": 0.17673834192100912, "learning_rate": 3.4001003670254656e-06, "loss": 0.0008837352506816387, "num_tokens": 74072588.0, "reward": 2.262255907058716, "reward_std": 0.5253946781158447, "rewards/code_complexity_reward/mean": 0.868945300579071, "rewards/code_complexity_reward/std": 0.15028499066829681, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.03992818295955658, "step": 391, "step_time": 58.787011036649346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 120.123046875, "completions/mean_terminated_length": 120.123046875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.18672989262267947, "epoch": 0.4469783352337514, "frac_reward_zero_std": 0.3125, "grad_norm": 0.04528536647558212, "kl": 0.14738887175917625, "learning_rate": 3.390806477434269e-06, "loss": 0.0007367623038589954, "num_tokens": 74201339.0, "reward": 2.2570314407348633, "reward_std": 0.5298864841461182, "rewards/code_complexity_reward/mean": 0.86669921875, "rewards/code_complexity_reward/std": 0.1558055877685547, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 392, "step_time": 51.933684720657766 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 125.150390625, "completions/mean_terminated_length": 124.39334869384766, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.1830222075805068, "epoch": 0.44811858608893956, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.04444868117570877, "kl": 0.1604934453498572, "learning_rate": 3.381498464791939e-06, "loss": 0.0008024530252441764, "num_tokens": 74334308.0, "reward": 2.2916994094848633, "reward_std": 0.5824189782142639, "rewards/code_complexity_reward/mean": 0.858691394329071, "rewards/code_complexity_reward/std": 0.17669405043125153, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 393, "step_time": 58.97537276893854 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 122.978515625, "completions/mean_terminated_length": 122.21721649169922, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.1921136686578393, "epoch": 0.4492588369441277, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.042878683656454086, "kl": 0.16597853903658688, "learning_rate": 3.372176476669853e-06, "loss": 0.0008297109743580222, "num_tokens": 74465513.0, "reward": 2.263427734375, "reward_std": 0.5309746861457825, "rewards/code_complexity_reward/mean": 0.8779296875, "rewards/code_complexity_reward/std": 0.14166748523712158, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03337814658880234, "step": 394, "step_time": 56.530573510564864 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 132.337890625, "completions/mean_terminated_length": 130.84902954101562, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.1840333880390972, "epoch": 0.45039908779931587, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.0377763956785202, "kl": 0.1529093428980559, "learning_rate": 3.362840660860958e-06, "loss": 0.0007647082093171775, "num_tokens": 74601686.0, "reward": 2.313525438308716, "reward_std": 0.540594220161438, "rewards/code_complexity_reward/mean": 0.8625977039337158, "rewards/code_complexity_reward/std": 0.15241849422454834, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03428199514746666, "step": 395, "step_time": 71.81780820153654 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 321.0, "completions/mean_length": 123.890625, "completions/mean_terminated_length": 123.13111114501953, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.19608492241241038, "epoch": 0.45153933865450396, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.042173754423856735, "kl": 0.15032148628961295, "learning_rate": 3.353491165377429e-06, "loss": 0.0007515776087529957, "num_tokens": 74732090.0, "reward": 2.288378953933716, "reward_std": 0.5520635843276978, "rewards/code_complexity_reward/mean": 0.8721679449081421, "rewards/code_complexity_reward/std": 0.15276890993118286, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.030144967138767242, "step": 396, "step_time": 58.19775234721601 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 126.548828125, "completions/mean_terminated_length": 125.03726196289062, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.19224893697537482, "epoch": 0.4526795895096921, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.043557677417993546, "kl": 0.16475624777376652, "learning_rate": 3.3441281384483215e-06, "loss": 0.0008236112771555781, "num_tokens": 74864875.0, "reward": 2.2925782203674316, "reward_std": 0.5886866450309753, "rewards/code_complexity_reward/mean": 0.8543944954872131, "rewards/code_complexity_reward/std": 0.18257030844688416, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.025821086019277573, "step": 397, "step_time": 59.85923853609711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 472.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 125.890625, "completions/mean_terminated_length": 125.890625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.17609607218764722, "epoch": 0.45381984036488027, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.04550276696681976, "kl": 0.16528159379959106, "learning_rate": 3.3347517285172225e-06, "loss": 0.0008266210788860917, "num_tokens": 74998487.0, "reward": 2.316162109375, "reward_std": 0.5651240944862366, "rewards/code_complexity_reward/mean": 0.8555663824081421, "rewards/code_complexity_reward/std": 0.15477801859378815, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 398, "step_time": 56.6802940601483 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 131.85546875, "completions/mean_terminated_length": 131.11154174804688, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.19465996627695858, "epoch": 0.4549600912200684, "frac_reward_zero_std": 0.3125, "grad_norm": 0.03967063128948212, "kl": 0.1522540762089193, "learning_rate": 3.325362084239894e-06, "loss": 0.0007612542249262333, "num_tokens": 75133745.0, "reward": 2.2638182640075684, "reward_std": 0.5450235605239868, "rewards/code_complexity_reward/mean": 0.8678710460662842, "rewards/code_complexity_reward/std": 0.16453681886196136, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 399, "step_time": 70.76058609131724 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 442.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 124.93359375, "completions/mean_terminated_length": 124.93359375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.1891685025766492, "epoch": 0.45610034207525657, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.04436409845948219, "kl": 0.16885516932234168, "learning_rate": 3.31595935448192e-06, "loss": 0.0008443153346888721, "num_tokens": 75266603.0, "reward": 2.3233399391174316, "reward_std": 0.5551702380180359, "rewards/code_complexity_reward/mean": 0.8773437738418579, "rewards/code_complexity_reward/std": 0.15168453752994537, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.028089813888072968, "step": 400, "step_time": 62.4180976934731 }, { "epoch": 0.45610034207525657, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.005, "eval_completions/max_length": 215.6, "eval_completions/max_terminated_length": 203.84, "eval_completions/mean_length": 127.3275, "eval_completions/mean_terminated_length": 125.51785736083984, "eval_completions/min_length": 78.2, "eval_completions/min_terminated_length": 78.2, "eval_entropy": 0.18608157515525817, "eval_frac_reward_zero_std": 0.28, "eval_kl": 0.15036571115255357, "eval_loss": 0.0007513108430430293, "eval_num_tokens": 75266603.0, "eval_reward": 2.257812600135803, "eval_reward_std": 0.42881025157868863, "eval_rewards/code_complexity_reward/mean": 0.8627499949932098, "eval_rewards/code_complexity_reward/std": 0.09983778864145279, "eval_rewards/code_execution_reward/mean": 0.3075, "eval_rewards/code_execution_reward/std": 0.3626027238368988, "eval_rewards/code_syntax_reward/mean": 0.49125, "eval_rewards/code_syntax_reward/std": 0.022306769788265228, "eval_rewards/reasoning_present_reward_func/mean": 0.09975000157952309, "eval_rewards/reasoning_present_reward_func/std": 0.000707106813788414, "eval_rewards/xmlcount_reward_func/mean": 0.4965625, "eval_rewards/xmlcount_reward_func/std": 0.009722718000411988, "eval_runtime": 453.6369, "eval_samples_per_second": 0.22, "eval_steps_per_second": 0.029, "step": 400 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 121.86328125, "completions/mean_terminated_length": 118.79133605957031, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.18452235776931047, "epoch": 0.4572405929304447, "frac_reward_zero_std": 0.3125, "grad_norm": 0.041354820132255554, "kl": 0.16844560171011835, "learning_rate": 3.3065436883163453e-06, "loss": 0.0008424263214692473, "num_tokens": 75398697.0, "reward": 2.2672364711761475, "reward_std": 0.5752920508384705, "rewards/code_complexity_reward/mean": 0.8624023199081421, "rewards/code_complexity_reward/std": 0.17763254046440125, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03516262024641037, "step": 401, "step_time": 57.57185255829245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 121.349609375, "completions/mean_terminated_length": 120.58512878417969, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.1911913213552907, "epoch": 0.4583808437856328, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.044758785516023636, "kl": 0.16513978457078338, "learning_rate": 3.2971152350213106e-06, "loss": 0.0008258229354396462, "num_tokens": 75530064.0, "reward": 2.3092775344848633, "reward_std": 0.5663158893585205, "rewards/code_complexity_reward/mean": 0.8645507097244263, "rewards/code_complexity_reward/std": 0.1732282191514969, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 402, "step_time": 60.83214425295591 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 132.220703125, "completions/mean_terminated_length": 131.4774932861328, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.1915932446718216, "epoch": 0.45952109464082097, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.03970784693956375, "kl": 0.14799431688152254, "learning_rate": 3.2876741440776853e-06, "loss": 0.0007399916066788137, "num_tokens": 75667021.0, "reward": 2.239306688308716, "reward_std": 0.5683034658432007, "rewards/code_complexity_reward/mean": 0.8534179925918579, "rewards/code_complexity_reward/std": 0.18323318660259247, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 403, "step_time": 61.2376615870744 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 487.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 123.97265625, "completions/mean_terminated_length": 123.97265625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.180886683287099, "epoch": 0.4606613454960091, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.04014888405799866, "kl": 0.16640125203412026, "learning_rate": 3.2782205651667013e-06, "loss": 0.0008321847999468446, "num_tokens": 75797831.0, "reward": 2.246875047683716, "reward_std": 0.588494598865509, "rewards/code_complexity_reward/mean": 0.8431640863418579, "rewards/code_complexity_reward/std": 0.19288252294063568, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 404, "step_time": 47.87241576705128 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 443.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 126.896484375, "completions/mean_terminated_length": 126.896484375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.18644631933420897, "epoch": 0.4618015963511973, "frac_reward_zero_std": 0.34375, "grad_norm": 0.04636787995696068, "kl": 0.15261980472132564, "learning_rate": 3.2687546481675776e-06, "loss": 0.0007630261825397611, "num_tokens": 75931454.0, "reward": 2.2607421875, "reward_std": 0.5587997436523438, "rewards/code_complexity_reward/mean": 0.8557617664337158, "rewards/code_complexity_reward/std": 0.17096194624900818, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02697930857539177, "step": 405, "step_time": 45.3986659552902 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 127.369140625, "completions/mean_terminated_length": 127.369140625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.19007356534712017, "epoch": 0.4629418472063854, "frac_reward_zero_std": 0.28125, "grad_norm": 0.04196585342288017, "kl": 0.14166517974808812, "learning_rate": 3.259276543155142e-06, "loss": 0.0007082646479830146, "num_tokens": 76064939.0, "reward": 2.2542967796325684, "reward_std": 0.5012690424919128, "rewards/code_complexity_reward/mean": 0.8724609017372131, "rewards/code_complexity_reward/std": 0.12132422626018524, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.023378821089863777, "step": 406, "step_time": 46.02709480561316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 125.06640625, "completions/mean_terminated_length": 123.54902648925781, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.17638201382942498, "epoch": 0.4640820980615735, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.04621737450361252, "kl": 0.19588419795036316, "learning_rate": 3.2497864003974554e-06, "loss": 0.000979577424004674, "num_tokens": 76197637.0, "reward": 2.301513671875, "reward_std": 0.5349698066711426, "rewards/code_complexity_reward/mean": 0.87158203125, "rewards/code_complexity_reward/std": 0.13851334154605865, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03343535214662552, "step": 407, "step_time": 68.55234510917217 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 120.68359375, "completions/mean_terminated_length": 120.68359375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.1918132845312357, "epoch": 0.4652223489167617, "frac_reward_zero_std": 0.3125, "grad_norm": 0.05499803274869919, "kl": 0.16732276743277907, "learning_rate": 3.2402843703534283e-06, "loss": 0.0008366013644263148, "num_tokens": 76326439.0, "reward": 2.2735352516174316, "reward_std": 0.5657873153686523, "rewards/code_complexity_reward/mean": 0.8643554449081421, "rewards/code_complexity_reward/std": 0.17282043397426605, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.026930565014481544, "step": 408, "step_time": 42.537293306551874 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 129.474609375, "completions/mean_terminated_length": 127.97451782226562, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.19182226341217756, "epoch": 0.4663625997719498, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.04105975106358528, "kl": 0.1511634956113994, "learning_rate": 3.2307706036704328e-06, "loss": 0.0007557313656434417, "num_tokens": 76460794.0, "reward": 2.2815918922424316, "reward_std": 0.5388450622558594, "rewards/code_complexity_reward/mean": 0.8656249642372131, "rewards/code_complexity_reward/std": 0.15311507880687714, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 409, "step_time": 52.174283905886114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 121.23046875, "completions/mean_terminated_length": 121.23046875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.18904555635526776, "epoch": 0.467502850627138, "frac_reward_zero_std": 0.2890625, "grad_norm": 0.04235319048166275, "kl": 0.15319054713472724, "learning_rate": 3.221245251181919e-06, "loss": 0.0007661065901629627, "num_tokens": 76591588.0, "reward": 2.2435545921325684, "reward_std": 0.5339500904083252, "rewards/code_complexity_reward/mean": 0.8703124523162842, "rewards/code_complexity_reward/std": 0.16065393388271332, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.03210962936282158, "step": 410, "step_time": 58.31908944621682 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 131.64453125, "completions/mean_terminated_length": 130.1529541015625, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.1944365524686873, "epoch": 0.46864310148232613, "frac_reward_zero_std": 0.40625, "grad_norm": 0.03772205486893654, "kl": 0.1768490846734494, "learning_rate": 3.2117084639050204e-06, "loss": 0.00088401889661327, "num_tokens": 76729482.0, "reward": 2.3226561546325684, "reward_std": 0.5868728160858154, "rewards/code_complexity_reward/mean": 0.8597656488418579, "rewards/code_complexity_reward/std": 0.18137963116168976, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.04044608399271965, "step": 411, "step_time": 51.270871356129646 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 457.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 129.595703125, "completions/mean_terminated_length": 129.595703125, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.19186081504449248, "epoch": 0.4697833523375142, "frac_reward_zero_std": 0.390625, "grad_norm": 0.0407959520816803, "kl": 0.15698469197377563, "learning_rate": 3.2021603930381582e-06, "loss": 0.0007850199472159147, "num_tokens": 76863671.0, "reward": 2.2946290969848633, "reward_std": 0.5294326543807983, "rewards/code_complexity_reward/mean": 0.86572265625, "rewards/code_complexity_reward/std": 0.14146240055561066, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 412, "step_time": 47.31912287604064 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 367.0, "completions/max_terminated_length": 367.0, "completions/mean_length": 115.19140625, "completions/mean_terminated_length": 115.19140625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.19664128578733653, "epoch": 0.4709236031927024, "frac_reward_zero_std": 0.390625, "grad_norm": 0.04570984095335007, "kl": 0.1974201927660033, "learning_rate": 3.1926011899586483e-06, "loss": 0.0009879186982288957, "num_tokens": 76991741.0, "reward": 2.3006837368011475, "reward_std": 0.5344740152359009, "rewards/code_complexity_reward/mean": 0.88525390625, "rewards/code_complexity_reward/std": 0.1422916054725647, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 413, "step_time": 56.20667791552842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 466.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 121.947265625, "completions/mean_terminated_length": 121.947265625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.18173651304095984, "epoch": 0.47206385404789053, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04942749813199043, "kl": 0.22455714631360024, "learning_rate": 3.1830310062202996e-06, "loss": 0.0011232119286432862, "num_tokens": 77121754.0, "reward": 2.3118653297424316, "reward_std": 0.5295935273170471, "rewards/code_complexity_reward/mean": 0.873730480670929, "rewards/code_complexity_reward/std": 0.14807423949241638, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 414, "step_time": 46.83975786436349 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 126.794921875, "completions/mean_terminated_length": 125.2843246459961, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.19618339091539383, "epoch": 0.4732041049030787, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.044677723199129105, "kl": 0.16002909478265792, "learning_rate": 3.1734499935510093e-06, "loss": 0.0008001453243196011, "num_tokens": 77253061.0, "reward": 2.2372560501098633, "reward_std": 0.49872878193855286, "rewards/code_complexity_reward/mean": 0.87646484375, "rewards/code_complexity_reward/std": 0.1352963000535965, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.024042511358857155, "step": 415, "step_time": 50.472095406614244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 506.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 116.77734375, "completions/mean_terminated_length": 116.77734375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.18411309993825853, "epoch": 0.47434435575826683, "frac_reward_zero_std": 0.40625, "grad_norm": 0.038311637938022614, "kl": 0.1620446521556005, "learning_rate": 3.1638583038503596e-06, "loss": 0.0008101383573375642, "num_tokens": 77380535.0, "reward": 2.350830078125, "reward_std": 0.5450727343559265, "rewards/code_complexity_reward/mean": 0.880175769329071, "rewards/code_complexity_reward/std": 0.12714910507202148, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.03168932721018791, "step": 416, "step_time": 58.20013665687293 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 119.125, "completions/mean_terminated_length": 118.35616302490234, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.18997555365785956, "epoch": 0.475484606613455, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.04496529698371887, "kl": 0.1588612758787349, "learning_rate": 3.15425608918721e-06, "loss": 0.0007941796211525798, "num_tokens": 77509691.0, "reward": 2.3268556594848633, "reward_std": 0.5624994039535522, "rewards/code_complexity_reward/mean": 0.877246081829071, "rewards/code_complexity_reward/std": 0.16264338791370392, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 417, "step_time": 60.44075878709555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 123.521484375, "completions/mean_terminated_length": 123.521484375, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.18629115400835872, "epoch": 0.4766248574686431, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.09592591971158981, "kl": 0.22596826881635934, "learning_rate": 3.144643501797282e-06, "loss": 0.0011299514444544911, "num_tokens": 77641162.0, "reward": 2.2953615188598633, "reward_std": 0.5762179493904114, "rewards/code_complexity_reward/mean": 0.863574206829071, "rewards/code_complexity_reward/std": 0.1777394413948059, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 418, "step_time": 55.10252110846341 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 122.365234375, "completions/mean_terminated_length": 122.365234375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.1854600014630705, "epoch": 0.47776510832383123, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.043681006878614426, "kl": 0.1720415613381192, "learning_rate": 3.1350206940807523e-06, "loss": 0.0008601195877417922, "num_tokens": 77773049.0, "reward": 2.2475099563598633, "reward_std": 0.5551396012306213, "rewards/code_complexity_reward/mean": 0.860644519329071, "rewards/code_complexity_reward/std": 0.17667004466056824, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 419, "step_time": 35.88049926608801 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 124.841796875, "completions/mean_terminated_length": 124.08414459228516, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.19370042206719518, "epoch": 0.4789053591790194, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.04351407662034035, "kl": 0.16623323736712337, "learning_rate": 3.125387818599831e-06, "loss": 0.0008310168632306159, "num_tokens": 77904012.0, "reward": 2.3331055641174316, "reward_std": 0.5167831182479858, "rewards/code_complexity_reward/mean": 0.882031261920929, "rewards/code_complexity_reward/std": 0.11330031603574753, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4970703125, "rewards/code_syntax_reward/std": 0.038198307156562805, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 420, "step_time": 66.56381694134325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 466.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 119.1953125, "completions/mean_terminated_length": 119.1953125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.17892097355797887, "epoch": 0.48004561003420754, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.04813588038086891, "kl": 0.1732405312359333, "learning_rate": 3.1157450280763464e-06, "loss": 0.0008662950131110847, "num_tokens": 78033132.0, "reward": 2.3072755336761475, "reward_std": 0.5434042811393738, "rewards/code_complexity_reward/mean": 0.8798828125, "rewards/code_complexity_reward/std": 0.14363734424114227, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.023952921852469444, "step": 421, "step_time": 54.4853121554479 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 122.384765625, "completions/mean_terminated_length": 120.08840942382812, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.17611867596860975, "epoch": 0.4811858608893957, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.04268954321742058, "kl": 0.18196698313113302, "learning_rate": 3.10609247538932e-06, "loss": 0.0009098662994801998, "num_tokens": 78164641.0, "reward": 2.3396973609924316, "reward_std": 0.5811685919761658, "rewards/code_complexity_reward/mean": 0.8692382574081421, "rewards/code_complexity_reward/std": 0.16791574656963348, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02960825525224209, "step": 422, "step_time": 61.226425561122596 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 121.015625, "completions/mean_terminated_length": 120.25048828125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.1837502378039062, "epoch": 0.4823261117445838, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.04275209829211235, "kl": 0.18663235811982304, "learning_rate": 3.096430313572547e-06, "loss": 0.0009330929606221616, "num_tokens": 78293753.0, "reward": 2.253955125808716, "reward_std": 0.5483253598213196, "rewards/code_complexity_reward/mean": 0.8714843988418579, "rewards/code_complexity_reward/std": 0.15943068265914917, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.019755469635128975, "step": 423, "step_time": 71.30556803755462 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 477.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 132.693359375, "completions/mean_terminated_length": 132.693359375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.19592675054445863, "epoch": 0.48346636259977194, "frac_reward_zero_std": 0.296875, "grad_norm": 0.07278963178396225, "kl": 0.1626153108663857, "learning_rate": 3.0867586958121653e-06, "loss": 0.0008131182985380292, "num_tokens": 78432360.0, "reward": 2.241503953933716, "reward_std": 0.5670921802520752, "rewards/code_complexity_reward/mean": 0.8543945550918579, "rewards/code_complexity_reward/std": 0.17962567508220673, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.03879402577877045, "step": 424, "step_time": 55.979094293899834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 126.4140625, "completions/mean_terminated_length": 126.4140625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.19044809485785663, "epoch": 0.4846066134549601, "frac_reward_zero_std": 0.359375, "grad_norm": 0.04103491082787514, "kl": 0.15629818162415177, "learning_rate": 3.0770777754442333e-06, "loss": 0.0007814873824827373, "num_tokens": 78564048.0, "reward": 2.285205364227295, "reward_std": 0.5412234663963318, "rewards/code_complexity_reward/mean": 0.869433581829071, "rewards/code_complexity_reward/std": 0.14754438400268555, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.035264380276203156, "step": 425, "step_time": 46.89876246638596 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 121.876953125, "completions/mean_terminated_length": 120.3470687866211, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.1951297577470541, "epoch": 0.48574686431014824, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.046992674469947815, "kl": 0.17825728200841695, "learning_rate": 3.0673877059522906e-06, "loss": 0.0008913032943382859, "num_tokens": 78694605.0, "reward": 2.337451219558716, "reward_std": 0.5572441220283508, "rewards/code_complexity_reward/mean": 0.877734363079071, "rewards/code_complexity_reward/std": 0.1507721245288849, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.030623575672507286, "step": 426, "step_time": 61.01261132955551 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 126.044921875, "completions/mean_terminated_length": 123.7701416015625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.1890582744963467, "epoch": 0.4868871151653364, "frac_reward_zero_std": 0.3125, "grad_norm": 0.04474252834916115, "kl": 0.16880175180267543, "learning_rate": 3.057688640964934e-06, "loss": 0.0008438217919319868, "num_tokens": 78824348.0, "reward": 2.2790040969848633, "reward_std": 0.5551824569702148, "rewards/code_complexity_reward/mean": 0.8728514909744263, "rewards/code_complexity_reward/std": 0.1655663102865219, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02697930857539177, "step": 427, "step_time": 85.88813353702426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 127.568359375, "completions/mean_terminated_length": 126.060791015625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.19193098437972367, "epoch": 0.4880273660205245, "frac_reward_zero_std": 0.34375, "grad_norm": 0.042351577430963516, "kl": 0.15414535044692457, "learning_rate": 3.047980734253372e-06, "loss": 0.0007706115720793605, "num_tokens": 78958587.0, "reward": 2.3043947219848633, "reward_std": 0.5661848783493042, "rewards/code_complexity_reward/mean": 0.8668944835662842, "rewards/code_complexity_reward/std": 0.160882830619812, "rewards/code_execution_reward/mean": 0.3515625, "rewards/code_execution_reward/std": 0.4779251217842102, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.033960822969675064, "step": 428, "step_time": 58.78330996260047 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 477.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 120.859375, "completions/mean_terminated_length": 120.859375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.1953797668684274, "epoch": 0.48916761687571264, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.04210756719112396, "kl": 0.17014644446317106, "learning_rate": 3.038264139728997e-06, "loss": 0.0008507130551151931, "num_tokens": 79088171.0, "reward": 2.225390672683716, "reward_std": 0.547822892665863, "rewards/code_complexity_reward/mean": 0.8607422113418579, "rewards/code_complexity_reward/std": 0.17343607544898987, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03890470787882805, "step": 429, "step_time": 48.334947736002505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 327.0, "completions/mean_length": 115.61328125, "completions/mean_terminated_length": 114.83757019042969, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.1969982199370861, "epoch": 0.4903078677309008, "frac_reward_zero_std": 0.390625, "grad_norm": 0.039572574198246, "kl": 0.1787292886292562, "learning_rate": 3.0285390114409353e-06, "loss": 0.0008934312500059605, "num_tokens": 79217377.0, "reward": 2.292041063308716, "reward_std": 0.554071307182312, "rewards/code_complexity_reward/mean": 0.876953125, "rewards/code_complexity_reward/std": 0.15723296999931335, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03607473522424698, "step": 430, "step_time": 57.02608647197485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 461.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 119.337890625, "completions/mean_terminated_length": 119.337890625, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.1885069808922708, "epoch": 0.49144811858608894, "frac_reward_zero_std": 0.375, "grad_norm": 0.03890189155936241, "kl": 0.18978303333278745, "learning_rate": 3.018805503573612e-06, "loss": 0.0009485873160883784, "num_tokens": 79346042.0, "reward": 2.284717082977295, "reward_std": 0.5508351922035217, "rewards/code_complexity_reward/mean": 0.8765624761581421, "rewards/code_complexity_reward/std": 0.1598908007144928, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.030623575672507286, "step": 431, "step_time": 47.09009541384876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 389.0, "completions/max_terminated_length": 389.0, "completions/mean_length": 112.48828125, "completions/mean_terminated_length": 112.48828125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.1885553360916674, "epoch": 0.4925883694412771, "frac_reward_zero_std": 0.484375, "grad_norm": 0.043302081525325775, "kl": 0.18712277058511972, "learning_rate": 3.0090637704443033e-06, "loss": 0.0009355904767289758, "num_tokens": 79470676.0, "reward": 2.3509278297424316, "reward_std": 0.5371614694595337, "rewards/code_complexity_reward/mean": 0.8888671398162842, "rewards/code_complexity_reward/std": 0.14046011865139008, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 432, "step_time": 50.973479443229735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 113.251953125, "completions/mean_terminated_length": 113.251953125, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.1959410652052611, "epoch": 0.49372862029646525, "frac_reward_zero_std": 0.375, "grad_norm": 0.040632523596286774, "kl": 0.17330206464976072, "learning_rate": 2.9993139665006904e-06, "loss": 0.0008663589833304286, "num_tokens": 79595285.0, "reward": 2.2375001907348633, "reward_std": 0.5389864444732666, "rewards/code_complexity_reward/mean": 0.87353515625, "rewards/code_complexity_reward/std": 0.16343249380588531, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 433, "step_time": 38.26567635126412 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 478.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 118.30078125, "completions/mean_terminated_length": 118.30078125, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.19373841653577983, "epoch": 0.49486887115165334, "frac_reward_zero_std": 0.40625, "grad_norm": 0.04177283123135567, "kl": 0.1670774833764881, "learning_rate": 2.989556246318412e-06, "loss": 0.0008353932062163949, "num_tokens": 79721847.0, "reward": 2.3326661586761475, "reward_std": 0.5207942128181458, "rewards/code_complexity_reward/mean": 0.8901366591453552, "rewards/code_complexity_reward/std": 0.12758244574069977, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 434, "step_time": 47.61543778050691 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 495.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 118.41796875, "completions/mean_terminated_length": 118.41796875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.18574211047962308, "epoch": 0.4960091220068415, "frac_reward_zero_std": 0.359375, "grad_norm": 0.04759420454502106, "kl": 0.1609982494264841, "learning_rate": 2.9797907645986124e-06, "loss": 0.0008049196330830455, "num_tokens": 79849645.0, "reward": 2.324951171875, "reward_std": 0.5340749621391296, "rewards/code_complexity_reward/mean": 0.883593738079071, "rewards/code_complexity_reward/std": 0.13169218599796295, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 435, "step_time": 49.05147493071854 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 454.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 119.81640625, "completions/mean_terminated_length": 119.81640625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.19358524749986827, "epoch": 0.49714937286202965, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.05796132981777191, "kl": 0.26007712143473327, "learning_rate": 2.9700176761654875e-06, "loss": 0.0013009295798838139, "num_tokens": 79980023.0, "reward": 2.2843751907348633, "reward_std": 0.5330491662025452, "rewards/code_complexity_reward/mean": 0.880859375, "rewards/code_complexity_reward/std": 0.1394876092672348, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.030144967138767242, "step": 436, "step_time": 54.735026109963655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 407.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 116.876953125, "completions/mean_terminated_length": 116.876953125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.1895879351068288, "epoch": 0.4982896237172178, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04403749853372574, "kl": 0.17041321902070194, "learning_rate": 2.960237135963834e-06, "loss": 0.0008523057913407683, "num_tokens": 80109264.0, "reward": 2.2538087368011475, "reward_std": 0.5607712864875793, "rewards/code_complexity_reward/mean": 0.8680664300918579, "rewards/code_complexity_reward/std": 0.17286798357963562, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 437, "step_time": 58.5730384606868 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 466.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 114.087890625, "completions/mean_terminated_length": 114.087890625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.19061742816120386, "epoch": 0.49942987457240595, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04390174150466919, "kl": 0.22747431066818535, "learning_rate": 2.9504492990565885e-06, "loss": 0.0011363752419129014, "num_tokens": 80235129.0, "reward": 2.3809571266174316, "reward_std": 0.5500878095626831, "rewards/code_complexity_reward/mean": 0.8922851085662842, "rewards/code_complexity_reward/std": 0.13636335730552673, "rewards/code_execution_reward/mean": 0.396484375, "rewards/code_execution_reward/std": 0.4896455705165863, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 438, "step_time": 64.20098885893822 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 122.3828125, "completions/mean_terminated_length": 122.3828125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.19141742098145187, "epoch": 0.500570125427594, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.045638103038072586, "kl": 0.17788598546758294, "learning_rate": 2.9406543206223735e-06, "loss": 0.0008894908241927624, "num_tokens": 80365165.0, "reward": 2.3016600608825684, "reward_std": 0.5360416769981384, "rewards/code_complexity_reward/mean": 0.8805663585662842, "rewards/code_complexity_reward/std": 0.14106899499893188, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.022032126784324646, "step": 439, "step_time": 68.94489244464785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 406.0, "completions/max_terminated_length": 406.0, "completions/mean_length": 119.185546875, "completions/mean_terminated_length": 119.185546875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.1929114512167871, "epoch": 0.5017103762827823, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.052232496440410614, "kl": 0.17807021748740226, "learning_rate": 2.930852355953034e-06, "loss": 0.0008902998524717987, "num_tokens": 80492576.0, "reward": 2.2985353469848633, "reward_std": 0.5303004384040833, "rewards/code_complexity_reward/mean": 0.8798828125, "rewards/code_complexity_reward/std": 0.14039914309978485, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 440, "step_time": 64.23015901539475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 510.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 117.091796875, "completions/mean_terminated_length": 117.091796875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.18431372032500803, "epoch": 0.5028506271379704, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.04617723822593689, "kl": 0.17941793415229768, "learning_rate": 2.9210435604511756e-06, "loss": 0.0008970896014943719, "num_tokens": 80619907.0, "reward": 2.343066692352295, "reward_std": 0.5235695242881775, "rewards/code_complexity_reward/mean": 0.8936523199081421, "rewards/code_complexity_reward/std": 0.11755724251270294, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 441, "step_time": 70.34586771950126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 511.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 126.513671875, "completions/mean_terminated_length": 126.513671875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.19931833143346012, "epoch": 0.5039908779931584, "frac_reward_zero_std": 0.375, "grad_norm": 0.04290877282619476, "kl": 0.18566951248794794, "learning_rate": 2.9112280896277017e-06, "loss": 0.0009282664395868778, "num_tokens": 80753658.0, "reward": 2.260058641433716, "reward_std": 0.5435094833374023, "rewards/code_complexity_reward/mean": 0.87109375, "rewards/code_complexity_reward/std": 0.16107021272182465, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.026872845366597176, "step": 442, "step_time": 48.61235083732754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 120.904296875, "completions/mean_terminated_length": 120.1389389038086, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.20454977359622717, "epoch": 0.5051311288483467, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.04251473769545555, "kl": 0.18741561693605036, "learning_rate": 2.90140609909935e-06, "loss": 0.0009369627805426717, "num_tokens": 80884525.0, "reward": 2.2898926734924316, "reward_std": 0.5232011079788208, "rewards/code_complexity_reward/mean": 0.8812499642372131, "rewards/code_complexity_reward/std": 0.1358095109462738, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 443, "step_time": 57.264054141007364 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 482.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 115.041015625, "completions/mean_terminated_length": 115.041015625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.18546200403943658, "epoch": 0.5062713797035348, "frac_reward_zero_std": 0.40625, "grad_norm": 0.045562244951725006, "kl": 0.17474447470158339, "learning_rate": 2.8915777445862185e-06, "loss": 0.0008737553143873811, "num_tokens": 81011102.0, "reward": 2.3299317359924316, "reward_std": 0.5287399291992188, "rewards/code_complexity_reward/mean": 0.880566418170929, "rewards/code_complexity_reward/std": 0.13769939541816711, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 444, "step_time": 54.54093983396888 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 123.708984375, "completions/mean_terminated_length": 122.9491195678711, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.19535031565465033, "epoch": 0.507411630558723, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.04043477773666382, "kl": 0.18224362982437015, "learning_rate": 2.8817431819093065e-06, "loss": 0.000910943082999438, "num_tokens": 81141505.0, "reward": 2.2530760765075684, "reward_std": 0.5550220608711243, "rewards/code_complexity_reward/mean": 0.8701171875, "rewards/code_complexity_reward/std": 0.17588353157043457, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 445, "step_time": 49.43013122212142 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 479.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 122.79296875, "completions/mean_terminated_length": 122.79296875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.20172198954969645, "epoch": 0.508551881413911, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.03921181708574295, "kl": 0.1731492601102218, "learning_rate": 2.8719025669880357e-06, "loss": 0.0008657922153361142, "num_tokens": 81272923.0, "reward": 2.2535643577575684, "reward_std": 0.5607020258903503, "rewards/code_complexity_reward/mean": 0.8697265386581421, "rewards/code_complexity_reward/std": 0.17673246562480927, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.03842822089791298, "step": 446, "step_time": 54.68148603197187 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 486.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 123.509765625, "completions/mean_terminated_length": 123.509765625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.19529078830964863, "epoch": 0.5096921322690992, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.05185670778155327, "kl": 0.1769433809677139, "learning_rate": 2.8620560558377825e-06, "loss": 0.0008846692508086562, "num_tokens": 81404628.0, "reward": 2.284716844558716, "reward_std": 0.5424965023994446, "rewards/code_complexity_reward/mean": 0.8829101324081421, "rewards/code_complexity_reward/std": 0.15207605063915253, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03521692752838135, "step": 447, "step_time": 48.83464278001338 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 378.0, "completions/mean_length": 117.65234375, "completions/mean_terminated_length": 116.88062286376953, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.19370974763296545, "epoch": 0.5108323831242874, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.047509223222732544, "kl": 0.17718097381293774, "learning_rate": 2.8522038045674026e-06, "loss": 0.0008858527289703488, "num_tokens": 81532290.0, "reward": 2.3216795921325684, "reward_std": 0.5429734587669373, "rewards/code_complexity_reward/mean": 0.88720703125, "rewards/code_complexity_reward/std": 0.1436435729265213, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.02798757515847683, "step": 448, "step_time": 67.16657931543887 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 489.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 121.5, "completions/mean_terminated_length": 121.5, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.19805829762481153, "epoch": 0.5119726339794755, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.04329917952418327, "kl": 0.1854558492777869, "learning_rate": 2.8423459693767586e-06, "loss": 0.0009269392467103899, "num_tokens": 81662426.0, "reward": 2.2569823265075684, "reward_std": 0.5478711128234863, "rewards/code_complexity_reward/mean": 0.8785156011581421, "rewards/code_complexity_reward/std": 0.16310149431228638, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.041431497782468796, "step": 449, "step_time": 50.421297007240355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 119.787109375, "completions/mean_terminated_length": 118.2490234375, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.19558180775493383, "epoch": 0.5131128848346637, "frac_reward_zero_std": 0.40625, "grad_norm": 0.05021924898028374, "kl": 0.17417367140296847, "learning_rate": 2.8324827065542405e-06, "loss": 0.0008707644883543253, "num_tokens": 81793825.0, "reward": 2.303027391433716, "reward_std": 0.538993775844574, "rewards/code_complexity_reward/mean": 0.8780273199081421, "rewards/code_complexity_reward/std": 0.1491679847240448, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02465205453336239, "step": 450, "step_time": 60.90639554243535 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 123.173828125, "completions/mean_terminated_length": 121.6490249633789, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.1876258656848222, "epoch": 0.5142531356898518, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.046596914529800415, "kl": 0.18345391482580453, "learning_rate": 2.822614172474289e-06, "loss": 0.0009174746228381991, "num_tokens": 81923786.0, "reward": 2.292724609375, "reward_std": 0.5411419868469238, "rewards/code_complexity_reward/mean": 0.8707031011581421, "rewards/code_complexity_reward/std": 0.15037816762924194, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.01981583796441555, "step": 451, "step_time": 57.50988831464201 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 120.12109375, "completions/mean_terminated_length": 119.35420989990234, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.19642137247137725, "epoch": 0.5153933865450399, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.04164701700210571, "kl": 0.17714834155049175, "learning_rate": 2.8127405235949173e-06, "loss": 0.0008858089568093419, "num_tokens": 82051736.0, "reward": 2.2994141578674316, "reward_std": 0.5442814826965332, "rewards/code_complexity_reward/mean": 0.8771483898162842, "rewards/code_complexity_reward/std": 0.1504242867231369, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 452, "step_time": 58.61843925062567 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 421.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 120.009765625, "completions/mean_terminated_length": 120.009765625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.19446955691091716, "epoch": 0.5165336374002281, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04197918623685837, "kl": 0.1771772694773972, "learning_rate": 2.80286191645523e-06, "loss": 0.0008856073254719377, "num_tokens": 82184257.0, "reward": 2.3058595657348633, "reward_std": 0.5228275656700134, "rewards/code_complexity_reward/mean": 0.8859374523162842, "rewards/code_complexity_reward/std": 0.13262921571731567, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03729969263076782, "step": 453, "step_time": 52.89277493208647 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 120.951171875, "completions/mean_terminated_length": 119.41765594482422, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.1865008152090013, "epoch": 0.5176738882554162, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.042556282132864, "kl": 0.17424987605772913, "learning_rate": 2.792978507672941e-06, "loss": 0.0008712565177120268, "num_tokens": 82315564.0, "reward": 2.2901368141174316, "reward_std": 0.5713781714439392, "rewards/code_complexity_reward/mean": 0.8719726800918579, "rewards/code_complexity_reward/std": 0.17570270597934723, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.019055325537919998, "step": 454, "step_time": 50.49403819255531 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 113.380859375, "completions/mean_terminated_length": 112.60078430175781, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.20128001854754984, "epoch": 0.5188141391106044, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04715017229318619, "kl": 0.19285939657129347, "learning_rate": 2.7830904539418884e-06, "loss": 0.0009644448873586953, "num_tokens": 82441795.0, "reward": 2.292675733566284, "reward_std": 0.5506401658058167, "rewards/code_complexity_reward/mean": 0.8828124403953552, "rewards/code_complexity_reward/std": 0.15358564257621765, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02697930857539177, "step": 455, "step_time": 50.4806450298056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 377.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 118.625, "completions/mean_terminated_length": 118.625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.19294820493087173, "epoch": 0.5199543899657925, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.04705251380801201, "kl": 0.1819582408061251, "learning_rate": 2.7731979120295564e-06, "loss": 0.0009096665307879448, "num_tokens": 82568591.0, "reward": 2.248584270477295, "reward_std": 0.4927971661090851, "rewards/code_complexity_reward/mean": 0.8819335699081421, "rewards/code_complexity_reward/std": 0.12988317012786865, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 456, "step_time": 46.16228280775249 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 464.0, "completions/max_terminated_length": 464.0, "completions/mean_length": 129.31640625, "completions/mean_terminated_length": 129.31640625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2006624578498304, "epoch": 0.5210946408209807, "frac_reward_zero_std": 0.3046875, "grad_norm": 0.04432983323931694, "kl": 0.17634391167666763, "learning_rate": 2.763301038774583e-06, "loss": 0.0008815351757220924, "num_tokens": 82703081.0, "reward": 2.254443407058716, "reward_std": 0.5445972681045532, "rewards/code_complexity_reward/mean": 0.8643554449081421, "rewards/code_complexity_reward/std": 0.16802632808685303, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.494384765625, "rewards/xmlcount_reward_func/std": 0.04691002145409584, "step": 457, "step_time": 54.911855028010905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 120.236328125, "completions/mean_terminated_length": 118.70000457763672, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.19554939563386142, "epoch": 0.5222348916761688, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.05369971692562103, "kl": 0.18464726326055825, "learning_rate": 2.7533999910842766e-06, "loss": 0.0009233374148607254, "num_tokens": 82834674.0, "reward": 2.2425780296325684, "reward_std": 0.572428822517395, "rewards/code_complexity_reward/mean": 0.8712890148162842, "rewards/code_complexity_reward/std": 0.1865483522415161, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.026930565014481544, "step": 458, "step_time": 51.15643201675266 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 353.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 113.40234375, "completions/mean_terminated_length": 113.40234375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.1892514326609671, "epoch": 0.5233751425313569, "frac_reward_zero_std": 0.484375, "grad_norm": 0.039775095880031586, "kl": 0.18127721373457462, "learning_rate": 2.743494925932129e-06, "loss": 0.0009065882186405361, "num_tokens": 82959736.0, "reward": 2.372997999191284, "reward_std": 0.551605224609375, "rewards/code_complexity_reward/mean": 0.891406238079071, "rewards/code_complexity_reward/std": 0.13079753518104553, "rewards/code_execution_reward/mean": 0.392578125, "rewards/code_execution_reward/std": 0.4888018071651459, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.026328405365347862, "step": 459, "step_time": 38.99826338980347 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 113.015625, "completions/mean_terminated_length": 112.23483276367188, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.1982444585300982, "epoch": 0.5245153933865451, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04487890005111694, "kl": 0.1954982888419181, "learning_rate": 2.7335860003553257e-06, "loss": 0.00097763747908175, "num_tokens": 83086804.0, "reward": 2.309326171875, "reward_std": 0.5384344458580017, "rewards/code_complexity_reward/mean": 0.888964831829071, "rewards/code_complexity_reward/std": 0.1452115774154663, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 460, "step_time": 50.09399913623929 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 122.171875, "completions/mean_terminated_length": 121.40900421142578, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.18731628987006843, "epoch": 0.5256556442417332, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.06169587001204491, "kl": 0.2628934889798984, "learning_rate": 2.723673371452254e-06, "loss": 0.0013157392386347055, "num_tokens": 83215556.0, "reward": 2.3082518577575684, "reward_std": 0.5487679839134216, "rewards/code_complexity_reward/mean": 0.8792968988418579, "rewards/code_complexity_reward/std": 0.15073560178279877, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.033528104424476624, "step": 461, "step_time": 49.60817673616111 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 117.609375, "completions/mean_terminated_length": 116.83757019042969, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.1980795650742948, "epoch": 0.5267958950969214, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.042059145867824554, "kl": 0.20113247982226312, "learning_rate": 2.713757196380017e-06, "loss": 0.0010058978805318475, "num_tokens": 83343984.0, "reward": 2.2382326126098633, "reward_std": 0.5342794060707092, "rewards/code_complexity_reward/mean": 0.8798828125, "rewards/code_complexity_reward/std": 0.1634557843208313, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.03168932721018791, "step": 462, "step_time": 57.300350029952824 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 411.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 114.205078125, "completions/mean_terminated_length": 114.205078125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.19841273385100067, "epoch": 0.5279361459521095, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04513176903128624, "kl": 0.18404238403309137, "learning_rate": 2.70383763235194e-06, "loss": 0.000920384656637907, "num_tokens": 83470109.0, "reward": 2.252734661102295, "reward_std": 0.4899277687072754, "rewards/code_complexity_reward/mean": 0.894824206829071, "rewards/code_complexity_reward/std": 0.12662860751152039, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03567269444465637, "step": 463, "step_time": 44.25766897108406 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 120.8515625, "completions/mean_terminated_length": 119.31765747070312, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.199290428776294, "epoch": 0.5290763968072976, "frac_reward_zero_std": 0.359375, "grad_norm": 0.06373570114374161, "kl": 0.22266581025905907, "learning_rate": 2.693914836635076e-06, "loss": 0.00111292558722198, "num_tokens": 83598073.0, "reward": 2.3009767532348633, "reward_std": 0.547853410243988, "rewards/code_complexity_reward/mean": 0.8805663585662842, "rewards/code_complexity_reward/std": 0.15894180536270142, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03567269444465637, "step": 464, "step_time": 50.182875756174326 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 121.677734375, "completions/mean_terminated_length": 121.677734375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2077512047253549, "epoch": 0.5302166476624858, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.049214303493499756, "kl": 0.18381741154007614, "learning_rate": 2.6839889665477144e-06, "loss": 0.0009189534466713667, "num_tokens": 83729840.0, "reward": 2.296142578125, "reward_std": 0.515546977519989, "rewards/code_complexity_reward/mean": 0.891894519329071, "rewards/code_complexity_reward/std": 0.13259315490722656, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.022693097591400146, "step": 465, "step_time": 46.526943049393594 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 116.96484375, "completions/mean_terminated_length": 116.19178009033203, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20548563078045845, "epoch": 0.5313568985176739, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04978284612298012, "kl": 0.21764241065829992, "learning_rate": 2.6740601794568866e-06, "loss": 0.0010882659116759896, "num_tokens": 83857582.0, "reward": 2.3310060501098633, "reward_std": 0.5347437858581543, "rewards/code_complexity_reward/mean": 0.88671875, "rewards/code_complexity_reward/std": 0.13995866477489471, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 466, "step_time": 56.71448160149157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 391.0, "completions/max_terminated_length": 391.0, "completions/mean_length": 116.572265625, "completions/mean_terminated_length": 116.572265625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.1965284429024905, "epoch": 0.5324971493728621, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04321002960205078, "kl": 0.19837211887352169, "learning_rate": 2.664128632775871e-06, "loss": 0.0009921484161168337, "num_tokens": 83984403.0, "reward": 2.350634813308716, "reward_std": 0.5469903945922852, "rewards/code_complexity_reward/mean": 0.8900390863418579, "rewards/code_complexity_reward/std": 0.1392543613910675, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 467, "step_time": 47.186658379621804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 120.861328125, "completions/mean_terminated_length": 119.32746124267578, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.19065061351284385, "epoch": 0.5336374002280502, "frac_reward_zero_std": 0.3515625, "grad_norm": 0.04466906189918518, "kl": 0.1889311196282506, "learning_rate": 2.6541944839616957e-06, "loss": 0.0009446152253076434, "num_tokens": 84112440.0, "reward": 2.3092775344848633, "reward_std": 0.5282891392707825, "rewards/code_complexity_reward/mean": 0.889941394329071, "rewards/code_complexity_reward/std": 0.13993076980113983, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02059757336974144, "step": 468, "step_time": 49.73513897322118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 459.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 115.044921875, "completions/mean_terminated_length": 115.044921875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.19910043384879827, "epoch": 0.5347776510832383, "frac_reward_zero_std": 0.421875, "grad_norm": 0.04045619070529938, "kl": 0.19616307457908988, "learning_rate": 2.644257890512646e-06, "loss": 0.0009808861650526524, "num_tokens": 84238459.0, "reward": 2.3629395961761475, "reward_std": 0.5614554286003113, "rewards/code_complexity_reward/mean": 0.8895508050918579, "rewards/code_complexity_reward/std": 0.15109901130199432, "rewards/code_execution_reward/mean": 0.388671875, "rewards/code_execution_reward/std": 0.4879252314567566, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.0376812107861042, "step": 469, "step_time": 46.08504185266793 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 374.0, "completions/max_terminated_length": 374.0, "completions/mean_length": 117.5625, "completions/mean_terminated_length": 117.5625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.20022979727946222, "epoch": 0.5359179019384265, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.05083252117037773, "kl": 0.23903277411591262, "learning_rate": 2.634319009965762e-06, "loss": 0.0011949893087148666, "num_tokens": 84367919.0, "reward": 2.232959270477295, "reward_std": 0.4935433268547058, "rewards/code_complexity_reward/mean": 0.892578125, "rewards/code_complexity_reward/std": 0.13803766667842865, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03516262024641037, "step": 470, "step_time": 41.931415332481265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 121.822265625, "completions/mean_terminated_length": 121.05870819091797, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.19177483464591205, "epoch": 0.5370581527936146, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.043954022228717804, "kl": 0.1756314563099295, "learning_rate": 2.6243779998943496e-06, "loss": 0.0008779728086665273, "num_tokens": 84499868.0, "reward": 2.21875, "reward_std": 0.5610415935516357, "rewards/code_complexity_reward/mean": 0.86865234375, "rewards/code_complexity_reward/std": 0.18814872205257416, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.04039584472775459, "step": 471, "step_time": 60.449785232543945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 130.4921875, "completions/mean_terminated_length": 129.74559020996094, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.2011350600514561, "epoch": 0.5381984036488028, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.0398978516459465, "kl": 0.16857284377329051, "learning_rate": 2.614435017905469e-06, "loss": 0.0008428164292126894, "num_tokens": 84635180.0, "reward": 2.1912598609924316, "reward_std": 0.5203049778938293, "rewards/code_complexity_reward/mean": 0.8708007335662842, "rewards/code_complexity_reward/std": 0.16708606481552124, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.04275382310152054, "step": 472, "step_time": 58.364726674743 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 122.658203125, "completions/mean_terminated_length": 121.13137817382812, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.20347803342156112, "epoch": 0.5393386545039909, "frac_reward_zero_std": 0.40625, "grad_norm": 0.04072870686650276, "kl": 0.19219169509597123, "learning_rate": 2.6044902216374497e-06, "loss": 0.0009608984109945595, "num_tokens": 84765593.0, "reward": 2.224609375, "reward_std": 0.5478273630142212, "rewards/code_complexity_reward/mean": 0.8667968511581421, "rewards/code_complexity_reward/std": 0.18797273933887482, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.02327641472220421, "step": 473, "step_time": 59.943189572542906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 123.138671875, "completions/mean_terminated_length": 122.37769317626953, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.1978857759386301, "epoch": 0.540478905359179, "frac_reward_zero_std": 0.421875, "grad_norm": 0.03811249881982803, "kl": 0.19112805894110352, "learning_rate": 2.5945437687573816e-06, "loss": 0.0009558314923197031, "num_tokens": 84897472.0, "reward": 2.2374024391174316, "reward_std": 0.5367693305015564, "rewards/code_complexity_reward/mean": 0.8768554925918579, "rewards/code_complexity_reward/std": 0.1680627316236496, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.025862684473395348, "step": 474, "step_time": 68.36760861985385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 122.185546875, "completions/mean_terminated_length": 122.185546875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2021337121259421, "epoch": 0.5416191562143672, "frac_reward_zero_std": 0.421875, "grad_norm": 0.040355272591114044, "kl": 0.1928606026340276, "learning_rate": 2.584595816958621e-06, "loss": 0.0009643569937907159, "num_tokens": 85028399.0, "reward": 2.2247071266174316, "reward_std": 0.5310329794883728, "rewards/code_complexity_reward/mean": 0.8707031011581421, "rewards/code_complexity_reward/std": 0.16587869822978973, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.03805733472108841, "step": 475, "step_time": 53.85859397146851 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 117.458984375, "completions/mean_terminated_length": 116.6868896484375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.203527320176363, "epoch": 0.5427594070695553, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.041813962161540985, "kl": 0.17862305091693997, "learning_rate": 2.574646523958288e-06, "loss": 0.0008929879404604435, "num_tokens": 85156450.0, "reward": 2.28173828125, "reward_std": 0.511062502861023, "rewards/code_complexity_reward/mean": 0.887499988079071, "rewards/code_complexity_reward/std": 0.13954076170921326, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 476, "step_time": 67.38033112604171 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 116.5703125, "completions/mean_terminated_length": 115.79647827148438, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.20232148421928287, "epoch": 0.5438996579247435, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.046735748648643494, "kl": 0.17587921186350286, "learning_rate": 2.564696047494765e-06, "loss": 0.0008793058805167675, "num_tokens": 85283574.0, "reward": 2.294238567352295, "reward_std": 0.5381789207458496, "rewards/code_complexity_reward/mean": 0.890917956829071, "rewards/code_complexity_reward/std": 0.15565195679664612, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03729969263076782, "step": 477, "step_time": 68.06398711632937 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 434.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 120.169921875, "completions/mean_terminated_length": 120.169921875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2029083666857332, "epoch": 0.5450399087799316, "frac_reward_zero_std": 0.296875, "grad_norm": 0.04607435688376427, "kl": 0.20332452282309532, "learning_rate": 2.5547445453252e-06, "loss": 0.001016736146993935, "num_tokens": 85414573.0, "reward": 2.2606444358825684, "reward_std": 0.5358858704566956, "rewards/code_complexity_reward/mean": 0.8843749761581421, "rewards/code_complexity_reward/std": 0.1597760021686554, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.030093414708971977, "step": 478, "step_time": 44.85899585112929 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 116.01171875, "completions/mean_terminated_length": 114.45883178710938, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.1993470301385969, "epoch": 0.5461801596351197, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.043004438281059265, "kl": 0.17677216650918126, "learning_rate": 2.5447921752230003e-06, "loss": 0.0008835801272653043, "num_tokens": 85541855.0, "reward": 2.378662109375, "reward_std": 0.5962169170379639, "rewards/code_complexity_reward/mean": 0.8838866949081421, "rewards/code_complexity_reward/std": 0.17317436635494232, "rewards/code_execution_reward/mean": 0.4140625, "rewards/code_execution_reward/std": 0.49304109811782837, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.027404291555285454, "step": 479, "step_time": 58.11274976003915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 115.787109375, "completions/mean_terminated_length": 115.787109375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.20328013692051172, "epoch": 0.5473204104903079, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.04825602099299431, "kl": 0.19488975568674505, "learning_rate": 2.5348390949753343e-06, "loss": 0.000974498107098043, "num_tokens": 85669982.0, "reward": 2.2859373092651367, "reward_std": 0.5417749285697937, "rewards/code_complexity_reward/mean": 0.8846679925918579, "rewards/code_complexity_reward/std": 0.15837447345256805, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 480, "step_time": 41.777116228826344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 454.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 117.77734375, "completions/mean_terminated_length": 117.77734375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.1944221486337483, "epoch": 0.548460661345496, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04521126300096512, "kl": 0.1963545245816931, "learning_rate": 2.5248854623806297e-06, "loss": 0.0009818535763770342, "num_tokens": 85798868.0, "reward": 2.274169921875, "reward_std": 0.5271592140197754, "rewards/code_complexity_reward/mean": 0.8858398199081421, "rewards/code_complexity_reward/std": 0.15069855749607086, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.039215847849845886, "step": 481, "step_time": 61.066398687660694 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 127.185546875, "completions/mean_terminated_length": 126.43248748779297, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.19581231800839305, "epoch": 0.5496009122006842, "frac_reward_zero_std": 0.390625, "grad_norm": 0.04257240891456604, "kl": 0.21373940189369023, "learning_rate": 2.514931435246071e-06, "loss": 0.0010690197814255953, "num_tokens": 85930855.0, "reward": 2.308837890625, "reward_std": 0.5243239998817444, "rewards/code_complexity_reward/mean": 0.8864257335662842, "rewards/code_complexity_reward/std": 0.12565413117408752, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.039270635694265366, "step": 482, "step_time": 58.05678237602115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 440.0, "completions/mean_length": 121.06640625, "completions/mean_terminated_length": 120.3013687133789, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.20334241539239883, "epoch": 0.5507411630558723, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04284539818763733, "kl": 0.17328209162224084, "learning_rate": 2.504977171385098e-06, "loss": 0.0008662618929520249, "num_tokens": 86060817.0, "reward": 2.2965331077575684, "reward_std": 0.5209330916404724, "rewards/code_complexity_reward/mean": 0.8857421875, "rewards/code_complexity_reward/std": 0.13753432035446167, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 483, "step_time": 58.38384131062776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 451.0, "completions/max_terminated_length": 451.0, "completions/mean_length": 120.03515625, "completions/mean_terminated_length": 120.03515625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.19913944671861827, "epoch": 0.5518814139110604, "frac_reward_zero_std": 0.40625, "grad_norm": 0.04659715294837952, "kl": 0.18584857613313943, "learning_rate": 2.4950228286149028e-06, "loss": 0.0009292158065363765, "num_tokens": 86190935.0, "reward": 2.2522950172424316, "reward_std": 0.5278319716453552, "rewards/code_complexity_reward/mean": 0.8802734017372131, "rewards/code_complexity_reward/std": 0.16302402317523956, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 484, "step_time": 74.16857629921287 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 114.01953125, "completions/mean_terminated_length": 114.01953125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.19120553601533175, "epoch": 0.5530216647662486, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04992764815688133, "kl": 0.1714409238193184, "learning_rate": 2.48506856475393e-06, "loss": 0.0008572826627641916, "num_tokens": 86316637.0, "reward": 2.3455567359924316, "reward_std": 0.5574820637702942, "rewards/code_complexity_reward/mean": 0.8902343511581421, "rewards/code_complexity_reward/std": 0.14417028427124023, "rewards/code_execution_reward/mean": 0.369140625, "rewards/code_execution_reward/std": 0.4830440282821655, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.042219679802656174, "step": 485, "step_time": 50.670319026336074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 485.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 120.23828125, "completions/mean_terminated_length": 120.23828125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.20456409617327154, "epoch": 0.5541619156214367, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.04808012768626213, "kl": 0.19128678424749523, "learning_rate": 2.475114537619371e-06, "loss": 0.0009564837673678994, "num_tokens": 86446743.0, "reward": 2.337207317352295, "reward_std": 0.543820858001709, "rewards/code_complexity_reward/mean": 0.8926757574081421, "rewards/code_complexity_reward/std": 0.13841451704502106, "rewards/code_execution_reward/mean": 0.357421875, "rewards/code_execution_reward/std": 0.4797092080116272, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.032946839928627014, "step": 486, "step_time": 50.05792421475053 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 116.55859375, "completions/mean_terminated_length": 115.78473663330078, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.20600651274435222, "epoch": 0.5553021664766249, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.04536648839712143, "kl": 0.2058132393285632, "learning_rate": 2.4651609050246674e-06, "loss": 0.0010291931685060263, "num_tokens": 86576137.0, "reward": 2.1769533157348633, "reward_std": 0.5357188582420349, "rewards/code_complexity_reward/mean": 0.875781238079071, "rewards/code_complexity_reward/std": 0.19499453902244568, "rewards/code_execution_reward/mean": 0.22265625, "rewards/code_execution_reward/std": 0.41643625497817993, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 487, "step_time": 68.47450216673315 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 113.748046875, "completions/mean_terminated_length": 112.186279296875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2034447961486876, "epoch": 0.556442417331813, "frac_reward_zero_std": 0.4375, "grad_norm": 0.05799882858991623, "kl": 0.2798899515764788, "learning_rate": 2.4552078247770005e-06, "loss": 0.0013983813114464283, "num_tokens": 86700648.0, "reward": 2.267383098602295, "reward_std": 0.5464755892753601, "rewards/code_complexity_reward/mean": 0.8853515386581421, "rewards/code_complexity_reward/std": 0.16661201417446136, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.019055325537919998, "step": 488, "step_time": 50.85825465992093 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 118.46484375, "completions/mean_terminated_length": 116.92157745361328, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.19950762088410556, "epoch": 0.5575826681870011, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.03979875519871712, "kl": 0.17444116657134145, "learning_rate": 2.4452554546748008e-06, "loss": 0.0008724848157726228, "num_tokens": 86829602.0, "reward": 2.342334032058716, "reward_std": 0.5757735967636108, "rewards/code_complexity_reward/mean": 0.8738280534744263, "rewards/code_complexity_reward/std": 0.1667538434267044, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.04631038010120392, "step": 489, "step_time": 79.27966903243214 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 472.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 118.89453125, "completions/mean_terminated_length": 118.89453125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.20410157716833055, "epoch": 0.5587229190421893, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.042523350566625595, "kl": 0.19255458959378302, "learning_rate": 2.4353039525052354e-06, "loss": 0.0009628652478568256, "num_tokens": 86959576.0, "reward": 2.324512004852295, "reward_std": 0.5194054841995239, "rewards/code_complexity_reward/mean": 0.896484375, "rewards/code_complexity_reward/std": 0.12044984102249146, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.49462890625, "rewards/xmlcount_reward_func/std": 0.044600412249565125, "step": 490, "step_time": 58.80864681862295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 124.921875, "completions/mean_terminated_length": 121.87401580810547, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2091535299550742, "epoch": 0.5598631698973774, "frac_reward_zero_std": 0.5, "grad_norm": 0.03937102481722832, "kl": 0.18264956306666136, "learning_rate": 2.425353476041713e-06, "loss": 0.0009131882106885314, "num_tokens": 87091996.0, "reward": 2.221240282058716, "reward_std": 0.4707846939563751, "rewards/code_complexity_reward/mean": 0.8916991949081421, "rewards/code_complexity_reward/std": 0.12506139278411865, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.03510142117738724, "step": 491, "step_time": 62.02264278475195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 124.970703125, "completions/mean_terminated_length": 124.21330261230469, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.20684482180513442, "epoch": 0.5610034207525656, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.03801170736551285, "kl": 0.17654062097426504, "learning_rate": 2.4154041830413803e-06, "loss": 0.000882670865394175, "num_tokens": 87224549.0, "reward": 2.2205567359924316, "reward_std": 0.5386831760406494, "rewards/code_complexity_reward/mean": 0.8698241710662842, "rewards/code_complexity_reward/std": 0.1779492050409317, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 492, "step_time": 50.135662014596164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 114.44140625, "completions/mean_terminated_length": 112.88236236572266, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.1961705128196627, "epoch": 0.5621436716077537, "frac_reward_zero_std": 0.375, "grad_norm": 0.07102346420288086, "kl": 0.20416853355709463, "learning_rate": 2.4054562312426193e-06, "loss": 0.0010207772720605135, "num_tokens": 87349803.0, "reward": 2.2537598609924316, "reward_std": 0.5104479193687439, "rewards/code_complexity_reward/mean": 0.8944336175918579, "rewards/code_complexity_reward/std": 0.14461390674114227, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 493, "step_time": 60.67718891054392 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 503.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 127.509765625, "completions/mean_terminated_length": 127.509765625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2005750765092671, "epoch": 0.5632839224629419, "frac_reward_zero_std": 0.40625, "grad_norm": 0.047411415725946426, "kl": 0.17602034204173833, "learning_rate": 2.395509778362552e-06, "loss": 0.0008800018695183098, "num_tokens": 87484628.0, "reward": 2.2122559547424316, "reward_std": 0.5339294075965881, "rewards/code_complexity_reward/mean": 0.8708007335662842, "rewards/code_complexity_reward/std": 0.17578992247581482, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.031553346663713455, "step": 494, "step_time": 62.301436943002045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 504.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 122.69921875, "completions/mean_terminated_length": 122.69921875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.197824876755476, "epoch": 0.56442417331813, "frac_reward_zero_std": 0.3359375, "grad_norm": 0.045171745121479034, "kl": 0.17628267919644713, "learning_rate": 2.3855649820945313e-06, "loss": 0.0008815093897283077, "num_tokens": 87614890.0, "reward": 2.2413086891174316, "reward_std": 0.5361995697021484, "rewards/code_complexity_reward/mean": 0.868847668170929, "rewards/code_complexity_reward/std": 0.1696988195180893, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 495, "step_time": 60.70090287271887 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 379.0, "completions/mean_length": 113.041015625, "completions/mean_terminated_length": 112.2602767944336, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.1952875015558675, "epoch": 0.5655644241733181, "frac_reward_zero_std": 0.453125, "grad_norm": 0.0412074439227581, "kl": 0.18843856710009277, "learning_rate": 2.375622000105651e-06, "loss": 0.0009423307492397726, "num_tokens": 87742151.0, "reward": 2.3013672828674316, "reward_std": 0.5110214948654175, "rewards/code_complexity_reward/mean": 0.8999999761581421, "rewards/code_complexity_reward/std": 0.1182471439242363, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 496, "step_time": 50.45256731007248 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 458.0, "completions/max_terminated_length": 458.0, "completions/mean_length": 118.134765625, "completions/mean_terminated_length": 118.134765625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.19805099768564105, "epoch": 0.5667046750285063, "frac_reward_zero_std": 0.421875, "grad_norm": 0.044330909848213196, "kl": 0.17642376851290464, "learning_rate": 2.3656809900342383e-06, "loss": 0.0008818942587822676, "num_tokens": 87870516.0, "reward": 2.3341310024261475, "reward_std": 0.5248551964759827, "rewards/code_complexity_reward/mean": 0.9000976085662842, "rewards/code_complexity_reward/std": 0.12265406548976898, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 497, "step_time": 47.029645345173776 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 433.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 112.453125, "completions/mean_terminated_length": 112.453125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.20707908901385963, "epoch": 0.5678449258836944, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04368899390101433, "kl": 0.1835294715128839, "learning_rate": 2.355742109487355e-06, "loss": 0.0009174600127153099, "num_tokens": 87994960.0, "reward": 2.352051019668579, "reward_std": 0.5592976212501526, "rewards/code_complexity_reward/mean": 0.8931640386581421, "rewards/code_complexity_reward/std": 0.15360045433044434, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.024608410894870758, "step": 498, "step_time": 54.18074494134635 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 122.734375, "completions/mean_terminated_length": 121.97260284423828, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.20877380622550845, "epoch": 0.5689851767388826, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04545408487319946, "kl": 0.18420725548639894, "learning_rate": 2.3458055160383055e-06, "loss": 0.0009210672578774393, "num_tokens": 88125500.0, "reward": 2.3134279251098633, "reward_std": 0.5429850816726685, "rewards/code_complexity_reward/mean": 0.8841796517372131, "rewards/code_complexity_reward/std": 0.1432899385690689, "rewards/code_execution_reward/mean": 0.34375, "rewards/code_execution_reward/std": 0.4754233956336975, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.04358936473727226, "step": 499, "step_time": 86.89127167966217 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 117.669921875, "completions/mean_terminated_length": 115.34577941894531, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.20324918441474438, "epoch": 0.5701254275940707, "frac_reward_zero_std": 0.5, "grad_norm": 0.04335007444024086, "kl": 0.1833657359238714, "learning_rate": 2.33587136722413e-06, "loss": 0.0009167992975562811, "num_tokens": 88254015.0, "reward": 2.3480467796325684, "reward_std": 0.5674522519111633, "rewards/code_complexity_reward/mean": 0.8888671398162842, "rewards/code_complexity_reward/std": 0.1564750075340271, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.019055325537919998, "step": 500, "step_time": 52.142669164575636 }, { "epoch": 0.5701254275940707, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0025, "eval_completions/max_length": 195.96, "eval_completions/max_terminated_length": 194.96, "eval_completions/mean_length": 119.525, "eval_completions/mean_terminated_length": 118.81250030517577, "eval_completions/min_length": 74.16, "eval_completions/min_terminated_length": 74.16, "eval_entropy": 0.20535233587026597, "eval_frac_reward_zero_std": 0.43, "eval_kl": 0.18844585955142976, "eval_loss": 0.0009444838506169617, "eval_num_tokens": 88254015.0, "eval_reward": 2.2452500915527343, "eval_reward_std": 0.3622209738567472, "eval_rewards/code_complexity_reward/mean": 0.8858749830722809, "eval_rewards/code_complexity_reward/std": 0.08965978924185038, "eval_rewards/code_execution_reward/mean": 0.2725, "eval_rewards/code_execution_reward/std": 0.2942692422866821, "eval_rewards/code_syntax_reward/mean": 0.49, "eval_rewards/code_syntax_reward/std": 0.02584230363368988, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.496875, "eval_rewards/xmlcount_reward_func/std": 0.008047243803739548, "eval_runtime": 421.4939, "eval_samples_per_second": 0.237, "eval_steps_per_second": 0.031, "step": 500 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 415.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 121.298828125, "completions/mean_terminated_length": 121.298828125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20822372613474727, "epoch": 0.5712656784492588, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.04847276210784912, "kl": 0.17973754170816392, "learning_rate": 2.325939820543114e-06, "loss": 0.0008986563188955188, "num_tokens": 88385684.0, "reward": 2.3021974563598633, "reward_std": 0.5546823143959045, "rewards/code_complexity_reward/mean": 0.87646484375, "rewards/code_complexity_reward/std": 0.16331270337104797, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 501, "step_time": 44.30626406148076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 116.408203125, "completions/mean_terminated_length": 115.63404846191406, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2025200326461345, "epoch": 0.572405929304447, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04604817554354668, "kl": 0.19425758894067258, "learning_rate": 2.3160110334522864e-06, "loss": 0.0009713039617054164, "num_tokens": 88512333.0, "reward": 2.328369140625, "reward_std": 0.5625239014625549, "rewards/code_complexity_reward/mean": 0.8873046636581421, "rewards/code_complexity_reward/std": 0.15957674384117126, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 502, "step_time": 62.61820894386619 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 119.296875, "completions/mean_terminated_length": 118.52837371826172, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21129932021722198, "epoch": 0.5735461801596351, "frac_reward_zero_std": 0.390625, "grad_norm": 0.04627607390284538, "kl": 0.20126484765205532, "learning_rate": 2.3060851633649246e-06, "loss": 0.001006077160127461, "num_tokens": 88641937.0, "reward": 2.1637697219848633, "reward_std": 0.542896568775177, "rewards/code_complexity_reward/mean": 0.8705077767372131, "rewards/code_complexity_reward/std": 0.19592055678367615, "rewards/code_execution_reward/mean": 0.21875, "rewards/code_execution_reward/std": 0.41380295157432556, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.04114582762122154, "step": 503, "step_time": 59.65085758082569 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 121.814453125, "completions/mean_terminated_length": 120.2843246459961, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.2093670645263046, "epoch": 0.5746864310148233, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.050039488822221756, "kl": 0.1851103751687333, "learning_rate": 2.296162367648061e-06, "loss": 0.0009256038465537131, "num_tokens": 88773898.0, "reward": 2.2583985328674316, "reward_std": 0.5634021759033203, "rewards/code_complexity_reward/mean": 0.8780273199081421, "rewards/code_complexity_reward/std": 0.17901286482810974, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.030093414708971977, "step": 504, "step_time": 51.101298911497 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 122.779296875, "completions/mean_terminated_length": 122.01760864257812, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.20369255682453513, "epoch": 0.5758266818700114, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.04395781457424164, "kl": 0.17205830663442612, "learning_rate": 2.2862428036199834e-06, "loss": 0.0008602028246968985, "num_tokens": 88905093.0, "reward": 2.27001953125, "reward_std": 0.546035885810852, "rewards/code_complexity_reward/mean": 0.882128894329071, "rewards/code_complexity_reward/std": 0.16062487661838531, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03647070750594139, "step": 505, "step_time": 51.22200694400817 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 122.740234375, "completions/mean_terminated_length": 122.740234375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.20509834308177233, "epoch": 0.5769669327251995, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04191059619188309, "kl": 0.1823525195941329, "learning_rate": 2.2763266285477476e-06, "loss": 0.0009116511209867895, "num_tokens": 89035452.0, "reward": 2.2720704078674316, "reward_std": 0.5397879481315613, "rewards/code_complexity_reward/mean": 0.881054699420929, "rewards/code_complexity_reward/std": 0.16614796221256256, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 506, "step_time": 64.21926404628903 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 117.6171875, "completions/mean_terminated_length": 116.84539794921875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.19816545024514198, "epoch": 0.5781071835803877, "frac_reward_zero_std": 0.34375, "grad_norm": 0.04814969375729561, "kl": 0.2127476327586919, "learning_rate": 2.2664139996446756e-06, "loss": 0.0010642902925610542, "num_tokens": 89163924.0, "reward": 2.3095216751098633, "reward_std": 0.5647016167640686, "rewards/code_complexity_reward/mean": 0.8837890625, "rewards/code_complexity_reward/std": 0.1607251763343811, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02960825525224209, "step": 507, "step_time": 59.777878154069185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 119.19921875, "completions/mean_terminated_length": 118.43052673339844, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.19879357516765594, "epoch": 0.5792474344355758, "frac_reward_zero_std": 0.359375, "grad_norm": 0.04847878962755203, "kl": 0.1916384499054402, "learning_rate": 2.256505074067872e-06, "loss": 0.0009580126497894526, "num_tokens": 89294650.0, "reward": 2.2416017055511475, "reward_std": 0.5539290308952332, "rewards/code_complexity_reward/mean": 0.8688476085662842, "rewards/code_complexity_reward/std": 0.18013161420822144, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 508, "step_time": 58.258654612116516 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 112.478515625, "completions/mean_terminated_length": 112.478515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.20424170815385878, "epoch": 0.580387685290764, "frac_reward_zero_std": 0.421875, "grad_norm": 0.044714897871017456, "kl": 0.19591438747011125, "learning_rate": 2.246600008915724e-06, "loss": 0.000979224219918251, "num_tokens": 89418659.0, "reward": 2.2958498001098633, "reward_std": 0.5345051884651184, "rewards/code_complexity_reward/mean": 0.8883788585662842, "rewards/code_complexity_reward/std": 0.14593885838985443, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 509, "step_time": 65.46228080615401 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 477.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 119.037109375, "completions/mean_terminated_length": 119.037109375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.20462622307240963, "epoch": 0.5815279361459521, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.043491680175065994, "kl": 0.19248685613274574, "learning_rate": 2.236698961225417e-06, "loss": 0.0009623945225030184, "num_tokens": 89548574.0, "reward": 2.281005859375, "reward_std": 0.5465662479400635, "rewards/code_complexity_reward/mean": 0.88232421875, "rewards/code_complexity_reward/std": 0.16015852987766266, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.033528104424476624, "step": 510, "step_time": 63.018761752173305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 492.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 112.96484375, "completions/mean_terminated_length": 112.96484375, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.20302793523296714, "epoch": 0.5826681870011402, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.049827076494693756, "kl": 0.2101586643839255, "learning_rate": 2.226802087970444e-06, "loss": 0.0010506963590160012, "num_tokens": 89675104.0, "reward": 2.2418456077575684, "reward_std": 0.5230225324630737, "rewards/code_complexity_reward/mean": 0.8907226324081421, "rewards/code_complexity_reward/std": 0.1524646282196045, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.039270635694265366, "step": 511, "step_time": 58.7052404191345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 123.28125, "completions/mean_terminated_length": 123.28125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20653333677910268, "epoch": 0.5838084378563284, "frac_reward_zero_std": 0.375, "grad_norm": 0.04435238987207413, "kl": 0.19416476495098323, "learning_rate": 2.2169095460581116e-06, "loss": 0.0009706871351227164, "num_tokens": 89807152.0, "reward": 2.2677247524261475, "reward_std": 0.531203031539917, "rewards/code_complexity_reward/mean": 0.8820312023162842, "rewards/code_complexity_reward/std": 0.15940622985363007, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02960825525224209, "step": 512, "step_time": 51.721238744445145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 115.68359375, "completions/mean_terminated_length": 114.90802001953125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.1940628511365503, "epoch": 0.5849486887115165, "frac_reward_zero_std": 0.359375, "grad_norm": 0.04634249582886696, "kl": 0.200220278929919, "learning_rate": 2.2070214923270604e-06, "loss": 0.0010010171681642532, "num_tokens": 89933870.0, "reward": 2.2946290969848633, "reward_std": 0.53912752866745, "rewards/code_complexity_reward/mean": 0.8835936784744263, "rewards/code_complexity_reward/std": 0.15229611098766327, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 513, "step_time": 62.79356255475432 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 495.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 111.884765625, "completions/mean_terminated_length": 111.884765625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.19884286308661103, "epoch": 0.5860889395667047, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04710862785577774, "kl": 0.1873002154752612, "learning_rate": 2.197138083544771e-06, "loss": 0.0009364017751067877, "num_tokens": 90057459.0, "reward": 2.291796922683716, "reward_std": 0.49146440625190735, "rewards/code_complexity_reward/mean": 0.8970702886581421, "rewards/code_complexity_reward/std": 0.11675325036048889, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 514, "step_time": 50.93656835146248 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 482.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 120.853515625, "completions/mean_terminated_length": 120.853515625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.200730100274086, "epoch": 0.5872291904218928, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.048276208341121674, "kl": 0.19441863475367427, "learning_rate": 2.1872594764050835e-06, "loss": 0.0009721502428874373, "num_tokens": 90187120.0, "reward": 2.268603563308716, "reward_std": 0.5817464590072632, "rewards/code_complexity_reward/mean": 0.8705078363418579, "rewards/code_complexity_reward/std": 0.18993596732616425, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.026382790878415108, "step": 515, "step_time": 50.4555200105533 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 484.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 115.041015625, "completions/mean_terminated_length": 115.041015625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2032528668642044, "epoch": 0.5883694412770809, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.07790976762771606, "kl": 0.22301266598515213, "learning_rate": 2.177385827525712e-06, "loss": 0.0011151714716106653, "num_tokens": 90315089.0, "reward": 2.335986614227295, "reward_std": 0.5238118767738342, "rewards/code_complexity_reward/mean": 0.9032226204872131, "rewards/code_complexity_reward/std": 0.1254906803369522, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 516, "step_time": 54.13541030045599 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 444.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 119.728515625, "completions/mean_terminated_length": 119.728515625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20420214196201414, "epoch": 0.5895096921322691, "frac_reward_zero_std": 0.40625, "grad_norm": 0.04711003601551056, "kl": 0.18942299706395715, "learning_rate": 2.16751729344576e-06, "loss": 0.0009472399251535535, "num_tokens": 90442822.0, "reward": 2.281787395477295, "reward_std": 0.5534773468971252, "rewards/code_complexity_reward/mean": 0.8833984732627869, "rewards/code_complexity_reward/std": 0.16654594242572784, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03149271756410599, "step": 517, "step_time": 44.87257968913764 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 116.12890625, "completions/mean_terminated_length": 114.57647705078125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.19905595993623137, "epoch": 0.5906499429874572, "frac_reward_zero_std": 0.453125, "grad_norm": 0.042287781834602356, "kl": 0.17753272992558777, "learning_rate": 2.1576540306232418e-06, "loss": 0.0008877252694219351, "num_tokens": 90568844.0, "reward": 2.2716307640075684, "reward_std": 0.5293787121772766, "rewards/code_complexity_reward/mean": 0.88525390625, "rewards/code_complexity_reward/std": 0.15666013956069946, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 518, "step_time": 58.97009496856481 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 396.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 112.865234375, "completions/mean_terminated_length": 112.865234375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2049731151200831, "epoch": 0.5917901938426454, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.040036316961050034, "kl": 0.20201594359241426, "learning_rate": 2.147796195432597e-06, "loss": 0.0010099213104695082, "num_tokens": 90694299.0, "reward": 2.3028321266174316, "reward_std": 0.5161523222923279, "rewards/code_complexity_reward/mean": 0.8966796398162842, "rewards/code_complexity_reward/std": 0.13511867821216583, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 519, "step_time": 50.33181980997324 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 412.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 115.611328125, "completions/mean_terminated_length": 115.611328125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21013785735704005, "epoch": 0.5929304446978335, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.047352030873298645, "kl": 0.18953279685229063, "learning_rate": 2.1379439441622183e-06, "loss": 0.0009477370185777545, "num_tokens": 90821908.0, "reward": 2.225390911102295, "reward_std": 0.5417048335075378, "rewards/code_complexity_reward/mean": 0.8812499642372131, "rewards/code_complexity_reward/std": 0.17903852462768555, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.5, "rewards/xmlcount_reward_func/std": 0.0, "step": 520, "step_time": 55.49295894894749 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 491.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 122.083984375, "completions/mean_terminated_length": 122.083984375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.19695743988268077, "epoch": 0.5940706955530216, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.04494083672761917, "kl": 0.1861697097774595, "learning_rate": 2.1280974330119647e-06, "loss": 0.0009306262363679707, "num_tokens": 90953791.0, "reward": 2.259765625, "reward_std": 0.5535421371459961, "rewards/code_complexity_reward/mean": 0.8770507574081421, "rewards/code_complexity_reward/std": 0.17131838202476501, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 521, "step_time": 51.682381571270525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 396.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 113.921875, "completions/mean_terminated_length": 113.921875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.20496101956814528, "epoch": 0.5952109464082098, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04219317063689232, "kl": 0.21447840565815568, "learning_rate": 2.1182568180906947e-06, "loss": 0.0010724919848144054, "num_tokens": 91078895.0, "reward": 2.296191692352295, "reward_std": 0.5280559659004211, "rewards/code_complexity_reward/mean": 0.8985351324081421, "rewards/code_complexity_reward/std": 0.1462271809577942, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02059757336974144, "step": 522, "step_time": 57.497423333115876 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 118.15625, "completions/mean_terminated_length": 118.15625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.198111868230626, "epoch": 0.5963511972633979, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.0435057207942009, "kl": 0.19761307863518596, "learning_rate": 2.108422255413782e-06, "loss": 0.0009878028649836779, "num_tokens": 91206931.0, "reward": 2.2919435501098633, "reward_std": 0.5317923426628113, "rewards/code_complexity_reward/mean": 0.8865234851837158, "rewards/code_complexity_reward/std": 0.15351083874702454, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 523, "step_time": 48.58209296874702 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 465.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 118.220703125, "completions/mean_terminated_length": 118.220703125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.19917610753327608, "epoch": 0.5974914481185861, "frac_reward_zero_std": 0.390625, "grad_norm": 0.04192797839641571, "kl": 0.17960910429246724, "learning_rate": 2.0985939009006506e-06, "loss": 0.0008979298290796578, "num_tokens": 91335120.0, "reward": 2.237548828125, "reward_std": 0.5387367606163025, "rewards/code_complexity_reward/mean": 0.8726562261581421, "rewards/code_complexity_reward/std": 0.1668001264333725, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.025282343849539757, "step": 524, "step_time": 64.04792027547956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 119.85546875, "completions/mean_terminated_length": 119.08805847167969, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21046490385197103, "epoch": 0.5986316989737742, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04352264106273651, "kl": 0.18542814149986953, "learning_rate": 2.0887719103722987e-06, "loss": 0.0009272024617530406, "num_tokens": 91464026.0, "reward": 2.241992235183716, "reward_std": 0.5021620392799377, "rewards/code_complexity_reward/mean": 0.8907226324081421, "rewards/code_complexity_reward/std": 0.13576219975948334, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.03805733472108841, "step": 525, "step_time": 60.161574536934495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 507.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 116.57421875, "completions/mean_terminated_length": 116.57421875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.20675222855061293, "epoch": 0.5997719498289624, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04657342657446861, "kl": 0.19682013185229152, "learning_rate": 2.0789564395488252e-06, "loss": 0.0009840677957981825, "num_tokens": 91593380.0, "reward": 2.2513673305511475, "reward_std": 0.48106321692466736, "rewards/code_complexity_reward/mean": 0.8998046517372131, "rewards/code_complexity_reward/std": 0.11373382061719894, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 526, "step_time": 49.414467403665185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 421.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 118.58203125, "completions/mean_terminated_length": 118.58203125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.20497842389158905, "epoch": 0.6009122006841505, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04693586006760597, "kl": 0.21429316652938724, "learning_rate": 2.0691476440469674e-06, "loss": 0.0010716223623603582, "num_tokens": 91721970.0, "reward": 2.2969727516174316, "reward_std": 0.5367060899734497, "rewards/code_complexity_reward/mean": 0.8859374523162842, "rewards/code_complexity_reward/std": 0.15065902471542358, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02697930857539177, "step": 527, "step_time": 44.107126450166106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 121.314453125, "completions/mean_terminated_length": 120.5499038696289, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.21099975006654859, "epoch": 0.6020524515393386, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.04494306817650795, "kl": 0.19873198820278049, "learning_rate": 2.059345679377627e-06, "loss": 0.0009935261914506555, "num_tokens": 91853971.0, "reward": 2.3021485805511475, "reward_std": 0.5548357963562012, "rewards/code_complexity_reward/mean": 0.8826172351837158, "rewards/code_complexity_reward/std": 0.16175585985183716, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.04533616453409195, "step": 528, "step_time": 71.31612950470299 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 116.376953125, "completions/mean_terminated_length": 115.60273742675781, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.20515693887136877, "epoch": 0.6031927023945268, "frac_reward_zero_std": 0.484375, "grad_norm": 0.038346149027347565, "kl": 0.1775742582976818, "learning_rate": 2.0495507009434127e-06, "loss": 0.0008879265515133739, "num_tokens": 91983020.0, "reward": 2.31396484375, "reward_std": 0.5452003479003906, "rewards/code_complexity_reward/mean": 0.8910155892372131, "rewards/code_complexity_reward/std": 0.14241193234920502, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.03380218520760536, "step": 529, "step_time": 60.038116762414575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 111.51171875, "completions/mean_terminated_length": 110.72798156738281, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.19926273031160235, "epoch": 0.6043329532497149, "frac_reward_zero_std": 0.5, "grad_norm": 0.04944264516234398, "kl": 0.19523620908148587, "learning_rate": 2.0397628640361674e-06, "loss": 0.0009762881090864539, "num_tokens": 92108406.0, "reward": 2.2879395484924316, "reward_std": 0.5219167470932007, "rewards/code_complexity_reward/mean": 0.8949218988418579, "rewards/code_complexity_reward/std": 0.14545829594135284, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 530, "step_time": 63.08163022901863 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 118.34375, "completions/mean_terminated_length": 117.5733871459961, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.19916955847293139, "epoch": 0.6054732041049031, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.039822883903980255, "kl": 0.18594977527391165, "learning_rate": 2.0299823238345125e-06, "loss": 0.000929471047129482, "num_tokens": 92239150.0, "reward": 2.2867677211761475, "reward_std": 0.552145779132843, "rewards/code_complexity_reward/mean": 0.8864257335662842, "rewards/code_complexity_reward/std": 0.1653328388929367, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02514021471142769, "step": 531, "step_time": 51.62280264496803 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 122.751953125, "completions/mean_terminated_length": 121.22549438476562, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.21360162226483226, "epoch": 0.6066134549600912, "frac_reward_zero_std": 0.3671875, "grad_norm": 0.050697002559900284, "kl": 0.19757053069770336, "learning_rate": 2.0202092354013885e-06, "loss": 0.000987856648862362, "num_tokens": 92371351.0, "reward": 2.268603801727295, "reward_std": 0.5492584109306335, "rewards/code_complexity_reward/mean": 0.8814452886581421, "rewards/code_complexity_reward/std": 0.1647433489561081, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09882812201976776, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.494384765625, "rewards/xmlcount_reward_func/std": 0.044225916266441345, "step": 532, "step_time": 63.052874591201544 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 115.474609375, "completions/mean_terminated_length": 114.6986312866211, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20670045143924654, "epoch": 0.6077537058152793, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.04564490541815758, "kl": 0.18815121054649353, "learning_rate": 2.0104437536815884e-06, "loss": 0.0009407930774614215, "num_tokens": 92498214.0, "reward": 2.248974561691284, "reward_std": 0.5370110273361206, "rewards/code_complexity_reward/mean": 0.8916015625, "rewards/code_complexity_reward/std": 0.16388081014156342, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.04004169628024101, "step": 533, "step_time": 61.1323479404673 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 473.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 118.958984375, "completions/mean_terminated_length": 118.958984375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21052496787160635, "epoch": 0.6088939566704675, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.045574091374874115, "kl": 0.20711123349610716, "learning_rate": 2.0006860334993105e-06, "loss": 0.0010356666753068566, "num_tokens": 92628097.0, "reward": 2.274169921875, "reward_std": 0.5235506296157837, "rewards/code_complexity_reward/mean": 0.8869140148162842, "rewards/code_complexity_reward/std": 0.1521684229373932, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.026328405365347862, "step": 534, "step_time": 67.22467786073685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 111.9609375, "completions/mean_terminated_length": 111.9609375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20235288399271667, "epoch": 0.6100342075256556, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.046888839453458786, "kl": 0.19685357972048223, "learning_rate": 1.990936229555697e-06, "loss": 0.0009841825813055038, "num_tokens": 92752621.0, "reward": 2.2928223609924316, "reward_std": 0.5274055004119873, "rewards/code_complexity_reward/mean": 0.896777331829071, "rewards/code_complexity_reward/std": 0.14219540357589722, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.040798187255859375, "step": 535, "step_time": 42.363345484249294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 119.115234375, "completions/mean_terminated_length": 117.57451629638672, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.20984520483762026, "epoch": 0.6111744583808438, "frac_reward_zero_std": 0.359375, "grad_norm": 0.04859217256307602, "kl": 0.19255335116758943, "learning_rate": 1.981194496426389e-06, "loss": 0.0009626021492294967, "num_tokens": 92881164.0, "reward": 2.228076457977295, "reward_std": 0.5442113280296326, "rewards/code_complexity_reward/mean": 0.8724609613418579, "rewards/code_complexity_reward/std": 0.18065275251865387, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 536, "step_time": 57.37546119093895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 109.916015625, "completions/mean_terminated_length": 109.12915802001953, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.19557616231031716, "epoch": 0.6123147092360319, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04872291535139084, "kl": 0.18496428930666298, "learning_rate": 1.971460988559065e-06, "loss": 0.0009245573310181499, "num_tokens": 93004729.0, "reward": 2.3080568313598633, "reward_std": 0.54327392578125, "rewards/code_complexity_reward/mean": 0.895800769329071, "rewards/code_complexity_reward/std": 0.14958174526691437, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.025244520977139473, "step": 537, "step_time": 53.44624068029225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 462.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 124.212890625, "completions/mean_terminated_length": 124.212890625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.19782803766429424, "epoch": 0.61345496009122, "frac_reward_zero_std": 0.390625, "grad_norm": 0.04540504515171051, "kl": 0.17348143039271235, "learning_rate": 1.9617358602710034e-06, "loss": 0.0008674884447827935, "num_tokens": 93138290.0, "reward": 2.3036623001098633, "reward_std": 0.5314935445785522, "rewards/code_complexity_reward/mean": 0.8837890625, "rewards/code_complexity_reward/std": 0.13964904844760895, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 538, "step_time": 56.02558157220483 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 494.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 118.802734375, "completions/mean_terminated_length": 118.802734375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.20419717649929225, "epoch": 0.6145952109464082, "frac_reward_zero_std": 0.5, "grad_norm": 0.040872808545827866, "kl": 0.184473006404005, "learning_rate": 1.9520192657466286e-06, "loss": 0.0009222574299201369, "num_tokens": 93267477.0, "reward": 2.2770020961761475, "reward_std": 0.5543588995933533, "rewards/code_complexity_reward/mean": 0.8841796517372131, "rewards/code_complexity_reward/std": 0.16875191032886505, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03521692752838135, "step": 539, "step_time": 50.0024071438238 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 119.013671875, "completions/mean_terminated_length": 119.013671875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.19726030807942152, "epoch": 0.6157354618015963, "frac_reward_zero_std": 0.421875, "grad_norm": 0.050635844469070435, "kl": 0.20265634928364307, "learning_rate": 1.9423113590350666e-06, "loss": 0.0010131653398275375, "num_tokens": 93398396.0, "reward": 2.247851848602295, "reward_std": 0.5169559717178345, "rewards/code_complexity_reward/mean": 0.890429675579071, "rewards/code_complexity_reward/std": 0.15275105834007263, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.033960822969675064, "step": 540, "step_time": 43.32531864847988 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 116.30078125, "completions/mean_terminated_length": 115.52642059326172, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.20809697406366467, "epoch": 0.6168757126567845, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.04219748079776764, "kl": 0.19445714156609029, "learning_rate": 1.9326122940477102e-06, "loss": 0.0009720231173560023, "num_tokens": 93525766.0, "reward": 2.2555177211761475, "reward_std": 0.5074157118797302, "rewards/code_complexity_reward/mean": 0.8976562023162842, "rewards/code_complexity_reward/std": 0.13665121793746948, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.027517380192875862, "step": 541, "step_time": 60.94381522294134 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 117.115234375, "completions/mean_terminated_length": 116.34246826171875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20788667490705848, "epoch": 0.6180159635119726, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.04158759117126465, "kl": 0.19724529678933322, "learning_rate": 1.9229222245557675e-06, "loss": 0.0009863232262432575, "num_tokens": 93653069.0, "reward": 2.2787599563598633, "reward_std": 0.546326756477356, "rewards/code_complexity_reward/mean": 0.889843761920929, "rewards/code_complexity_reward/std": 0.1602480262517929, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03691261634230614, "step": 542, "step_time": 49.796050718054175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 483.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 116.466796875, "completions/mean_terminated_length": 116.466796875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20928800315596163, "epoch": 0.6191562143671607, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.11907359212636948, "kl": 0.39858908497262746, "learning_rate": 1.9132413041878356e-06, "loss": 0.0019966403488069773, "num_tokens": 93780544.0, "reward": 2.2703614234924316, "reward_std": 0.5267053246498108, "rewards/code_complexity_reward/mean": 0.8866211175918579, "rewards/code_complexity_reward/std": 0.15587548911571503, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.01981583796441555, "step": 543, "step_time": 57.04909439291805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 112.212890625, "completions/mean_terminated_length": 111.43052673339844, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21068232227116823, "epoch": 0.6202964652223489, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04599347338080406, "kl": 0.19718560331966728, "learning_rate": 1.903569686427454e-06, "loss": 0.000986021477729082, "num_tokens": 93904893.0, "reward": 2.3062989711761475, "reward_std": 0.5367608070373535, "rewards/code_complexity_reward/mean": 0.8998047113418579, "rewards/code_complexity_reward/std": 0.14117884635925293, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.031553346663713455, "step": 544, "step_time": 57.60066555812955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 118.59765625, "completions/mean_terminated_length": 117.82778930664062, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.20821587461978197, "epoch": 0.621436716077537, "frac_reward_zero_std": 0.453125, "grad_norm": 0.049752090126276016, "kl": 0.1830353079130873, "learning_rate": 1.8939075246106809e-06, "loss": 0.0009148797253146768, "num_tokens": 94034275.0, "reward": 2.2309083938598633, "reward_std": 0.5461643934249878, "rewards/code_complexity_reward/mean": 0.877148449420929, "rewards/code_complexity_reward/std": 0.16674409806728363, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.04008939489722252, "step": 545, "step_time": 68.80869743973017 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 454.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 118.4296875, "completions/mean_terminated_length": 118.4296875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2049155430868268, "epoch": 0.6225769669327252, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04370631277561188, "kl": 0.18930393259506673, "learning_rate": 1.8842549719236544e-06, "loss": 0.0009468484204262495, "num_tokens": 94163855.0, "reward": 2.2876954078674316, "reward_std": 0.5481645464897156, "rewards/code_complexity_reward/mean": 0.8905273675918579, "rewards/code_complexity_reward/std": 0.15719257295131683, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.021983280777931213, "step": 546, "step_time": 48.29742703028023 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 393.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 118.861328125, "completions/mean_terminated_length": 118.861328125, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.20786532200872898, "epoch": 0.6237172177879133, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.04430852830410004, "kl": 0.19431215443182737, "learning_rate": 1.874612181400169e-06, "loss": 0.0009715239284560084, "num_tokens": 94297024.0, "reward": 2.263232707977295, "reward_std": 0.5368987321853638, "rewards/code_complexity_reward/mean": 0.8902343511581421, "rewards/code_complexity_reward/std": 0.15687862038612366, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.03160629794001579, "step": 547, "step_time": 72.38496452290565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 121.69921875, "completions/mean_terminated_length": 120.93541717529297, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.20443466561846435, "epoch": 0.6248574686431014, "frac_reward_zero_std": 0.40625, "grad_norm": 0.052284564822912216, "kl": 0.18695710157044232, "learning_rate": 1.8649793059192484e-06, "loss": 0.00093474006280303, "num_tokens": 94427018.0, "reward": 2.256103754043579, "reward_std": 0.541786789894104, "rewards/code_complexity_reward/mean": 0.8818359375, "rewards/code_complexity_reward/std": 0.16772960126399994, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.03260336071252823, "step": 548, "step_time": 50.5119391893968 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 117.25, "completions/mean_terminated_length": 116.47749328613281, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.20640884921886027, "epoch": 0.6259977194982896, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.042701512575149536, "kl": 0.20245035702828318, "learning_rate": 1.8553564982027183e-06, "loss": 0.0010121262166649103, "num_tokens": 94552794.0, "reward": 2.3783693313598633, "reward_std": 0.5793576240539551, "rewards/code_complexity_reward/mean": 0.888867199420929, "rewards/code_complexity_reward/std": 0.15926410257816315, "rewards/code_execution_reward/mean": 0.40625, "rewards/code_execution_reward/std": 0.49161264300346375, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03607473522424698, "step": 549, "step_time": 57.858662048354745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 114.958984375, "completions/mean_terminated_length": 113.40196990966797, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.2067343129310757, "epoch": 0.6271379703534777, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.03804142028093338, "kl": 0.19907039450481534, "learning_rate": 1.8457439108127914e-06, "loss": 0.000995362177491188, "num_tokens": 94680793.0, "reward": 2.2735841274261475, "reward_std": 0.5509658455848694, "rewards/code_complexity_reward/mean": 0.8890625238418579, "rewards/code_complexity_reward/std": 0.16349144279956818, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.04004169628024101, "step": 550, "step_time": 60.15547319594771 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 289.0, "completions/max_terminated_length": 289.0, "completions/mean_length": 115.23828125, "completions/mean_terminated_length": 115.23828125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.19964360096491873, "epoch": 0.6282782212086659, "frac_reward_zero_std": 0.4453125, "grad_norm": 1.3661240339279175, "kl": 1.2800506512867287, "learning_rate": 1.8361416961496412e-06, "loss": 0.006409589201211929, "num_tokens": 94808667.0, "reward": 2.3050782680511475, "reward_std": 0.5329881906509399, "rewards/code_complexity_reward/mean": 0.897265613079071, "rewards/code_complexity_reward/std": 0.14084015786647797, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03562243655323982, "step": 551, "step_time": 46.88445772603154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 468.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 118.9296875, "completions/mean_terminated_length": 118.9296875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2007942763157189, "epoch": 0.629418472063854, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.047557372599840164, "kl": 0.18208052043337375, "learning_rate": 1.8265500064489915e-06, "loss": 0.0009103791089728475, "num_tokens": 94937823.0, "reward": 2.243213176727295, "reward_std": 0.5163795351982117, "rewards/code_complexity_reward/mean": 0.8841796517372131, "rewards/code_complexity_reward/std": 0.15880458056926727, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 552, "step_time": 51.613195007666945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 407.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 111.859375, "completions/mean_terminated_length": 111.859375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21768906502984464, "epoch": 0.6305587229190421, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04336027801036835, "kl": 0.20329046971164644, "learning_rate": 1.816968993779701e-06, "loss": 0.0010162713006138802, "num_tokens": 95061487.0, "reward": 2.3128418922424316, "reward_std": 0.5228697061538696, "rewards/code_complexity_reward/mean": 0.9027343988418579, "rewards/code_complexity_reward/std": 0.13305217027664185, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 553, "step_time": 41.13446932006627 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 383.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 115.158203125, "completions/mean_terminated_length": 115.158203125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.20500368135981262, "epoch": 0.6316989737742303, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.04651375114917755, "kl": 0.18891283741686493, "learning_rate": 1.8073988100413515e-06, "loss": 0.0009446208132430911, "num_tokens": 95188036.0, "reward": 2.322558879852295, "reward_std": 0.5808350443840027, "rewards/code_complexity_reward/mean": 0.880664050579071, "rewards/code_complexity_reward/std": 0.17319701611995697, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03567269444465637, "step": 554, "step_time": 40.33755754400045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 113.859375, "completions/mean_terminated_length": 113.08023071289062, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21226418809965253, "epoch": 0.6328392246294184, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04211435094475746, "kl": 0.20029227284248918, "learning_rate": 1.7978396069618426e-06, "loss": 0.0010011999402195215, "num_tokens": 95313608.0, "reward": 2.262988328933716, "reward_std": 0.5340758562088013, "rewards/code_complexity_reward/mean": 0.8919922113418579, "rewards/code_complexity_reward/std": 0.15886810421943665, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.03805733472108841, "step": 555, "step_time": 56.00666588637978 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 115.82421875, "completions/mean_terminated_length": 115.82421875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20112039940431714, "epoch": 0.6339794754846066, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.04068654030561447, "kl": 0.1803446866106242, "learning_rate": 1.7882915360949802e-06, "loss": 0.0009018216514959931, "num_tokens": 95440830.0, "reward": 2.2821288108825684, "reward_std": 0.5298805832862854, "rewards/code_complexity_reward/mean": 0.8952147960662842, "rewards/code_complexity_reward/std": 0.1462898999452591, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 556, "step_time": 66.70827361289412 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 417.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 114.05859375, "completions/mean_terminated_length": 114.05859375, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.20612773392349482, "epoch": 0.6351197263397947, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04421505704522133, "kl": 0.197975168004632, "learning_rate": 1.7787547488180816e-06, "loss": 0.000989990308880806, "num_tokens": 95566752.0, "reward": 2.2523438930511475, "reward_std": 0.5175934433937073, "rewards/code_complexity_reward/mean": 0.8957030773162842, "rewards/code_complexity_reward/std": 0.14956261217594147, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09882812201976776, "rewards/reasoning_present_reward_func/std": 0.010772226378321648, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.04465661570429802, "step": 557, "step_time": 51.64337640441954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 119.087890625, "completions/mean_terminated_length": 117.54706573486328, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21264623943716288, "epoch": 0.636259977194983, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.045127272605895996, "kl": 0.18778826179914176, "learning_rate": 1.7692293963295679e-06, "loss": 0.0009388902690261602, "num_tokens": 95693469.0, "reward": 2.301513671875, "reward_std": 0.5901809334754944, "rewards/code_complexity_reward/mean": 0.8738280534744263, "rewards/code_complexity_reward/std": 0.18637177348136902, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 558, "step_time": 51.785086097195745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 429.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 124.748046875, "completions/mean_terminated_length": 124.748046875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.22250363859348, "epoch": 0.637400228050171, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04462701082229614, "kl": 0.20123346720356494, "learning_rate": 1.7597156296465734e-06, "loss": 0.0010058954358100891, "num_tokens": 95827252.0, "reward": 2.2035157680511475, "reward_std": 0.5416776537895203, "rewards/code_complexity_reward/mean": 0.8737304210662842, "rewards/code_complexity_reward/std": 0.17780007421970367, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.03885247930884361, "step": 559, "step_time": 51.23126363847405 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 441.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 115.830078125, "completions/mean_terminated_length": 115.830078125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.1992715287487954, "epoch": 0.6385404789053591, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04305538535118103, "kl": 0.20353249402251095, "learning_rate": 1.7502135996025454e-06, "loss": 0.0010179596720263362, "num_tokens": 95954193.0, "reward": 2.265429735183716, "reward_std": 0.5212731957435608, "rewards/code_complexity_reward/mean": 0.891894519329071, "rewards/code_complexity_reward/std": 0.14823727309703827, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.03380218520760536, "step": 560, "step_time": 51.539345018565655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 473.0, "completions/mean_length": 118.21875, "completions/mean_terminated_length": 117.4481430053711, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2094325884245336, "epoch": 0.6396807297605474, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.04504097253084183, "kl": 0.19671507482416928, "learning_rate": 1.7407234568448583e-06, "loss": 0.0009833048097789288, "num_tokens": 96083597.0, "reward": 2.2560548782348633, "reward_std": 0.560192883014679, "rewards/code_complexity_reward/mean": 0.876953125, "rewards/code_complexity_reward/std": 0.17488069832324982, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.04182815924286842, "step": 561, "step_time": 51.00856740400195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 113.361328125, "completions/mean_terminated_length": 111.79804229736328, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.19456169474869967, "epoch": 0.6408209806157354, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.046291954815387726, "kl": 0.20707298093475401, "learning_rate": 1.7312453518324232e-06, "loss": 0.001035399385727942, "num_tokens": 96210354.0, "reward": 2.306201219558716, "reward_std": 0.5865788459777832, "rewards/code_complexity_reward/mean": 0.8818358778953552, "rewards/code_complexity_reward/std": 0.1843753606081009, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.04497045651078224, "step": 562, "step_time": 68.05191555526108 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 115.39453125, "completions/mean_terminated_length": 112.27165222167969, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.20917845238000154, "epoch": 0.6419612314709237, "frac_reward_zero_std": 0.5, "grad_norm": 0.04377875477075577, "kl": 0.18719327251892537, "learning_rate": 1.7217794348332989e-06, "loss": 0.0009360511903651059, "num_tokens": 96337192.0, "reward": 2.2665529251098633, "reward_std": 0.5562275648117065, "rewards/code_complexity_reward/mean": 0.8916015625, "rewards/code_complexity_reward/std": 0.17381912469863892, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02257700450718403, "step": 563, "step_time": 87.50286492239684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 110.78515625, "completions/mean_terminated_length": 110.78515625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.20358546962961555, "epoch": 0.6431014823261118, "frac_reward_zero_std": 0.515625, "grad_norm": 0.04561629891395569, "kl": 0.20565124694257975, "learning_rate": 1.712325855922316e-06, "loss": 0.0010281954891979694, "num_tokens": 96462330.0, "reward": 2.2912111282348633, "reward_std": 0.5391302108764648, "rewards/code_complexity_reward/mean": 0.8900390863418579, "rewards/code_complexity_reward/std": 0.15072686970233917, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03968288004398346, "step": 564, "step_time": 51.536716564558446 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 459.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 118.470703125, "completions/mean_terminated_length": 118.470703125, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.20291268778964877, "epoch": 0.6442417331812998, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.0432577021420002, "kl": 0.20387234317604452, "learning_rate": 1.7028847649786907e-06, "loss": 0.001019404619000852, "num_tokens": 96591387.0, "reward": 2.251757860183716, "reward_std": 0.5155192613601685, "rewards/code_complexity_reward/mean": 0.8921874761581421, "rewards/code_complexity_reward/std": 0.14051032066345215, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4931640625, "rewards/xmlcount_reward_func/std": 0.05021624639630318, "step": 565, "step_time": 55.13174018356949 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 442.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 114.779296875, "completions/mean_terminated_length": 114.779296875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.20423678727820516, "epoch": 0.645381984036488, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.03747190535068512, "kl": 0.18915884708985686, "learning_rate": 1.6934563116836555e-06, "loss": 0.000945781241171062, "num_tokens": 96718982.0, "reward": 2.3165528774261475, "reward_std": 0.5573566555976868, "rewards/code_complexity_reward/mean": 0.8929687142372131, "rewards/code_complexity_reward/std": 0.15549105405807495, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.027560751885175705, "step": 566, "step_time": 46.322850544936955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 113.423828125, "completions/mean_terminated_length": 111.86079406738281, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21004583779722452, "epoch": 0.6465222348916762, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.09015441685914993, "kl": 0.38400436099618673, "learning_rate": 1.6840406455180801e-06, "loss": 0.0019220359390601516, "num_tokens": 96844223.0, "reward": 2.2833497524261475, "reward_std": 0.5652916431427002, "rewards/code_complexity_reward/mean": 0.88818359375, "rewards/code_complexity_reward/std": 0.1755952090024948, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.03260336071252823, "step": 567, "step_time": 56.47447831835598 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 462.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 121.52734375, "completions/mean_terminated_length": 121.52734375, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.20934968465007842, "epoch": 0.6476624857468644, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.04544399678707123, "kl": 0.1967317204689607, "learning_rate": 1.6746379157601062e-06, "loss": 0.0009835786186158657, "num_tokens": 96973561.0, "reward": 2.2772459983825684, "reward_std": 0.5548684597015381, "rewards/code_complexity_reward/mean": 0.8761718273162842, "rewards/code_complexity_reward/std": 0.16598927974700928, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09882812201976776, "rewards/reasoning_present_reward_func/std": 0.010772225446999073, "rewards/xmlcount_reward_func/mean": 0.49267578125, "rewards/xmlcount_reward_func/std": 0.051944274455308914, "step": 568, "step_time": 55.39863607194275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 117.701171875, "completions/mean_terminated_length": 116.92955017089844, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.21152661857195199, "epoch": 0.6488027366020525, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04429265111684799, "kl": 0.19290914246812463, "learning_rate": 1.6652482714827783e-06, "loss": 0.0009647633414715528, "num_tokens": 97102512.0, "reward": 2.275927782058716, "reward_std": 0.5408262610435486, "rewards/code_complexity_reward/mean": 0.8897460699081421, "rewards/code_complexity_reward/std": 0.16376003623008728, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.031553346663713455, "step": 569, "step_time": 49.622682102024555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 112.6484375, "completions/mean_terminated_length": 111.86692810058594, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.2064601848833263, "epoch": 0.6499429874572406, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.03955594077706337, "kl": 0.1863146839896217, "learning_rate": 1.6558718615516787e-06, "loss": 0.0009315839852206409, "num_tokens": 97228376.0, "reward": 2.3272950649261475, "reward_std": 0.5386353731155396, "rewards/code_complexity_reward/mean": 0.9032226800918579, "rewards/code_complexity_reward/std": 0.14546088874340057, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 570, "step_time": 59.75282961130142 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 486.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 123.791015625, "completions/mean_terminated_length": 123.791015625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20572684472426772, "epoch": 0.6510832383124288, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.03974100947380066, "kl": 0.18054240278434008, "learning_rate": 1.6465088346225719e-06, "loss": 0.000902605417650193, "num_tokens": 97360957.0, "reward": 2.276611328125, "reward_std": 0.5503677725791931, "rewards/code_complexity_reward/mean": 0.883593738079071, "rewards/code_complexity_reward/std": 0.16470448672771454, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03250797092914581, "step": 571, "step_time": 70.81030938960612 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 120.400390625, "completions/mean_terminated_length": 119.63404846191406, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21286761248484254, "epoch": 0.6522234891676169, "frac_reward_zero_std": 0.453125, "grad_norm": 0.042900197207927704, "kl": 0.20300188148394227, "learning_rate": 1.6371593391390427e-06, "loss": 0.0010149665176868439, "num_tokens": 97492402.0, "reward": 2.2333984375, "reward_std": 0.5164232850074768, "rewards/code_complexity_reward/mean": 0.887011706829071, "rewards/code_complexity_reward/std": 0.16091182827949524, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 572, "step_time": 53.287852216511965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 117.0546875, "completions/mean_terminated_length": 116.28179931640625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.20648158458061516, "epoch": 0.6533637400228051, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04267330467700958, "kl": 0.18380814616102725, "learning_rate": 1.6278235233301482e-06, "loss": 0.0009189550764858723, "num_tokens": 97620782.0, "reward": 2.283203125, "reward_std": 0.5232870578765869, "rewards/code_complexity_reward/mean": 0.8931640386581421, "rewards/code_complexity_reward/std": 0.14959877729415894, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 573, "step_time": 57.38211890310049 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 475.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 115.3515625, "completions/mean_terminated_length": 115.3515625, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.20394620089791715, "epoch": 0.6545039908779932, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.047887884080410004, "kl": 0.1979789778124541, "learning_rate": 1.6185015352080614e-06, "loss": 0.000990204163827002, "num_tokens": 97747994.0, "reward": 2.2161622047424316, "reward_std": 0.5308683514595032, "rewards/code_complexity_reward/mean": 0.8775390386581421, "rewards/code_complexity_reward/std": 0.17518064379692078, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.03438635915517807, "step": 574, "step_time": 48.118670573458076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 457.0, "completions/max_terminated_length": 457.0, "completions/mean_length": 116.3671875, "completions/mean_terminated_length": 116.3671875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21346852579154074, "epoch": 0.6556442417331813, "frac_reward_zero_std": 0.40625, "grad_norm": 0.046735767275094986, "kl": 0.19240877742413431, "learning_rate": 1.6091935225657312e-06, "loss": 0.0009618955664336681, "num_tokens": 97877086.0, "reward": 2.309863567352295, "reward_std": 0.5205374956130981, "rewards/code_complexity_reward/mean": 0.8999999761581421, "rewards/code_complexity_reward/std": 0.13322728872299194, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.031092895194888115, "step": 575, "step_time": 46.08725989796221 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 113.89453125, "completions/mean_terminated_length": 113.89453125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.19858434819616377, "epoch": 0.6567844925883695, "frac_reward_zero_std": 0.4375, "grad_norm": 0.045821890234947205, "kl": 0.18775547191035002, "learning_rate": 1.599899632974535e-06, "loss": 0.0009388511534780264, "num_tokens": 98002932.0, "reward": 2.2625977993011475, "reward_std": 0.521916925907135, "rewards/code_complexity_reward/mean": 0.8905273079872131, "rewards/code_complexity_reward/std": 0.14569193124771118, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.033960822969675064, "step": 576, "step_time": 41.47132898773998 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 432.0, "completions/max_terminated_length": 432.0, "completions/mean_length": 115.73828125, "completions/mean_terminated_length": 115.73828125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21224628016352654, "epoch": 0.6579247434435576, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.048664774745702744, "kl": 0.1897229622118175, "learning_rate": 1.5906200137819378e-06, "loss": 0.000948671018704772, "num_tokens": 98130770.0, "reward": 2.3323731422424316, "reward_std": 0.5521954894065857, "rewards/code_complexity_reward/mean": 0.8966796398162842, "rewards/code_complexity_reward/std": 0.1459120362997055, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02960825525224209, "step": 577, "step_time": 50.70920352358371 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 120.51171875, "completions/mean_terminated_length": 119.74559783935547, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2105766599997878, "epoch": 0.6590649942987458, "frac_reward_zero_std": 0.484375, "grad_norm": 0.040185339748859406, "kl": 0.18958499934524298, "learning_rate": 1.581354812109162e-06, "loss": 0.0009478017454966903, "num_tokens": 98261476.0, "reward": 2.231933832168579, "reward_std": 0.5317426919937134, "rewards/code_complexity_reward/mean": 0.8840820789337158, "rewards/code_complexity_reward/std": 0.17217253148555756, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 578, "step_time": 59.02408943977207 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 113.1171875, "completions/mean_terminated_length": 112.33659362792969, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21268249489367008, "epoch": 0.6602052451539339, "frac_reward_zero_std": 0.453125, "grad_norm": 0.043688561767339706, "kl": 0.19897344964556396, "learning_rate": 1.572104174848848e-06, "loss": 0.0009948199149221182, "num_tokens": 98388764.0, "reward": 2.318310499191284, "reward_std": 0.578164279460907, "rewards/code_complexity_reward/mean": 0.8856445550918579, "rewards/code_complexity_reward/std": 0.17437046766281128, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.04008939489722252, "step": 579, "step_time": 50.39408098347485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 120.79296875, "completions/mean_terminated_length": 119.25882720947266, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20600386848673224, "epoch": 0.661345496009122, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04037346690893173, "kl": 0.1793088677804917, "learning_rate": 1.562868248662732e-06, "loss": 0.0008963251602835953, "num_tokens": 98518930.0, "reward": 2.2618165016174316, "reward_std": 0.532821536064148, "rewards/code_complexity_reward/mean": 0.886425793170929, "rewards/code_complexity_reward/std": 0.1594582200050354, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.028043000027537346, "step": 580, "step_time": 58.728323098272085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 120.24609375, "completions/mean_terminated_length": 119.47945404052734, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2109548996668309, "epoch": 0.6624857468643102, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.0437057763338089, "kl": 0.1828673155978322, "learning_rate": 1.553647179979314e-06, "loss": 0.0009142899652943015, "num_tokens": 98647804.0, "reward": 2.252734661102295, "reward_std": 0.5014898180961609, "rewards/code_complexity_reward/mean": 0.893359363079071, "rewards/code_complexity_reward/std": 0.13966286182403564, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.04600567743182182, "step": 581, "step_time": 49.92625084053725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 117.87109375, "completions/mean_terminated_length": 117.87109375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.20622032159008086, "epoch": 0.6636259977194983, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.0446094274520874, "kl": 0.21468029532115906, "learning_rate": 1.5444411149915427e-06, "loss": 0.0010735903633758426, "num_tokens": 98776758.0, "reward": 2.269336223602295, "reward_std": 0.5378821492195129, "rewards/code_complexity_reward/mean": 0.8853515386581421, "rewards/code_complexity_reward/std": 0.16032706201076508, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.019055325537919998, "step": 582, "step_time": 62.63162047602236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 118.2734375, "completions/mean_terminated_length": 117.50293731689453, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21097185905091465, "epoch": 0.6647662485746865, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.0439179427921772, "kl": 0.18760294327512383, "learning_rate": 1.5352501996544935e-06, "loss": 0.0009379368857480586, "num_tokens": 98905622.0, "reward": 2.2256836891174316, "reward_std": 0.5234935879707336, "rewards/code_complexity_reward/mean": 0.8866211175918579, "rewards/code_complexity_reward/std": 0.1680779606103897, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.042552899569272995, "step": 583, "step_time": 58.309276348911226 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 439.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 118.453125, "completions/mean_terminated_length": 118.453125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.19968533469364047, "epoch": 0.6659064994298746, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.04098726809024811, "kl": 0.19028269569389522, "learning_rate": 1.5260745796830545e-06, "loss": 0.0009515002020634711, "num_tokens": 99034906.0, "reward": 2.330859422683716, "reward_std": 0.5418650507926941, "rewards/code_complexity_reward/mean": 0.8948242664337158, "rewards/code_complexity_reward/std": 0.14078882336616516, "rewards/code_execution_reward/mean": 0.349609375, "rewards/code_execution_reward/std": 0.47731292247772217, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.04188237711787224, "step": 584, "step_time": 51.92757675703615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 115.931640625, "completions/mean_terminated_length": 115.931640625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21503691747784615, "epoch": 0.6670467502850627, "frac_reward_zero_std": 0.4375, "grad_norm": 0.045676954090595245, "kl": 0.21066413563676178, "learning_rate": 1.51691440054962e-06, "loss": 0.0010535677429288626, "num_tokens": 99162259.0, "reward": 2.2404298782348633, "reward_std": 0.532676100730896, "rewards/code_complexity_reward/mean": 0.88623046875, "rewards/code_complexity_reward/std": 0.16519823670387268, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.029158055782318115, "step": 585, "step_time": 50.25724629499018 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 492.0, "completions/mean_length": 115.92578125, "completions/mean_terminated_length": 115.15068054199219, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21283286809921265, "epoch": 0.6681870011402509, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04298173636198044, "kl": 0.1897781506413594, "learning_rate": 1.5077698074817793e-06, "loss": 0.0009490360971540213, "num_tokens": 99290809.0, "reward": 2.3119144439697266, "reward_std": 0.5683254599571228, "rewards/code_complexity_reward/mean": 0.8917968273162842, "rewards/code_complexity_reward/std": 0.1707044541835785, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.031142795458436012, "step": 586, "step_time": 60.429373260587454 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 401.0, "completions/max_terminated_length": 401.0, "completions/mean_length": 119.53125, "completions/mean_terminated_length": 119.53125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21139764902181923, "epoch": 0.669327251995439, "frac_reward_zero_std": 0.421875, "grad_norm": 0.04530602693557739, "kl": 0.18400629190728068, "learning_rate": 1.49864094546002e-06, "loss": 0.0009199806954711676, "num_tokens": 99419297.0, "reward": 2.2693848609924316, "reward_std": 0.5407006144523621, "rewards/code_complexity_reward/mean": 0.8795897960662842, "rewards/code_complexity_reward/std": 0.166342094540596, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.028556859120726585, "step": 587, "step_time": 49.99597483314574 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 426.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 117.1953125, "completions/mean_terminated_length": 117.1953125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2193076096009463, "epoch": 0.6704675028506272, "frac_reward_zero_std": 0.515625, "grad_norm": 0.04041847214102745, "kl": 0.20293773035518825, "learning_rate": 1.489527959215421e-06, "loss": 0.0010146299609914422, "num_tokens": 99546369.0, "reward": 2.3595705032348633, "reward_std": 0.5464045405387878, "rewards/code_complexity_reward/mean": 0.890625, "rewards/code_complexity_reward/std": 0.14923155307769775, "rewards/code_execution_reward/mean": 0.3828125, "rewards/code_execution_reward/std": 0.486548513174057, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 588, "step_time": 43.21804660279304 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 107.6015625, "completions/mean_terminated_length": 106.81017303466797, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.19986138073727489, "epoch": 0.6716077537058153, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.04569625481963158, "kl": 0.19897017115727067, "learning_rate": 1.4804309932273669e-06, "loss": 0.0009947697399184108, "num_tokens": 99669061.0, "reward": 2.329150438308716, "reward_std": 0.5420535802841187, "rewards/code_complexity_reward/mean": 0.9027343392372131, "rewards/code_complexity_reward/std": 0.139584019780159, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.04008939489722252, "step": 589, "step_time": 57.11886035837233 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 112.896484375, "completions/mean_terminated_length": 112.896484375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20811396837234497, "epoch": 0.6727480045610034, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.04040137305855751, "kl": 0.2164451308781281, "learning_rate": 1.471350191721254e-06, "loss": 0.0010823990451171994, "num_tokens": 99795460.0, "reward": 2.2736330032348633, "reward_std": 0.5029444694519043, "rewards/code_complexity_reward/mean": 0.908203125, "rewards/code_complexity_reward/std": 0.13363462686538696, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 590, "step_time": 48.68236614204943 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 475.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 126.220703125, "completions/mean_terminated_length": 126.220703125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21456903498619795, "epoch": 0.6738882554161916, "frac_reward_zero_std": 0.3203125, "grad_norm": 0.047865577042102814, "kl": 0.18022927222773433, "learning_rate": 1.462285698666199e-06, "loss": 0.0009010575013235211, "num_tokens": 99928021.0, "reward": 2.2565431594848633, "reward_std": 0.5669649839401245, "rewards/code_complexity_reward/mean": 0.8701171875, "rewards/code_complexity_reward/std": 0.18863394856452942, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.021983280777931213, "step": 591, "step_time": 48.25568618811667 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 115.361328125, "completions/mean_terminated_length": 115.361328125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21179337962530553, "epoch": 0.6750285062713797, "frac_reward_zero_std": 0.421875, "grad_norm": 0.05230681598186493, "kl": 0.20260237983893603, "learning_rate": 1.4532376577727662e-06, "loss": 0.0010128874564543366, "num_tokens": 100055306.0, "reward": 2.268603563308716, "reward_std": 0.5257571339607239, "rewards/code_complexity_reward/mean": 0.8910155892372131, "rewards/code_complexity_reward/std": 0.1526585817337036, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.025282343849539757, "step": 592, "step_time": 51.18949353694916 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 489.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 117.791015625, "completions/mean_terminated_length": 117.791015625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21275977534241974, "epoch": 0.6761687571265679, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04415535181760788, "kl": 0.2225301533471793, "learning_rate": 1.4442062124906764e-06, "loss": 0.001112743397243321, "num_tokens": 100185147.0, "reward": 2.2138671875, "reward_std": 0.4988667368888855, "rewards/code_complexity_reward/mean": 0.8958007097244263, "rewards/code_complexity_reward/std": 0.14941811561584473, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03480498120188713, "step": 593, "step_time": 54.89004725776613 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 499.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 116.46484375, "completions/mean_terminated_length": 116.46484375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21251396974548697, "epoch": 0.677309007981756, "frac_reward_zero_std": 0.484375, "grad_norm": 0.041634708642959595, "kl": 0.1796561343362555, "learning_rate": 1.4351915060065488e-06, "loss": 0.0009028486674651504, "num_tokens": 100311701.0, "reward": 2.261474609375, "reward_std": 0.5268605947494507, "rewards/code_complexity_reward/mean": 0.88720703125, "rewards/code_complexity_reward/std": 0.1550752967596054, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 594, "step_time": 57.00105497799814 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 123.923828125, "completions/mean_terminated_length": 122.40196990966797, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2091953339986503, "epoch": 0.6784492588369442, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.04241528362035751, "kl": 0.1902667161775753, "learning_rate": 1.4261936812416124e-06, "loss": 0.0009511595126241446, "num_tokens": 100442242.0, "reward": 2.249755859375, "reward_std": 0.5574339032173157, "rewards/code_complexity_reward/mean": 0.8809570074081421, "rewards/code_complexity_reward/std": 0.17833667993545532, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.024042511358857155, "step": 595, "step_time": 51.326512484811246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009765625, "completions/max_length": 512.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 123.796875, "completions/mean_terminated_length": 119.96844482421875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.20957383094355464, "epoch": 0.6795895096921323, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04700101539492607, "kl": 0.188562877709046, "learning_rate": 1.4172128808494572e-06, "loss": 0.0009432386723347008, "num_tokens": 100573334.0, "reward": 2.286181688308716, "reward_std": 0.548892080783844, "rewards/code_complexity_reward/mean": 0.8858398199081421, "rewards/code_complexity_reward/std": 0.1571507453918457, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03149271756410599, "step": 596, "step_time": 61.2411751486361 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 114.138671875, "completions/mean_terminated_length": 114.138671875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.20832498697564006, "epoch": 0.6807297605473204, "frac_reward_zero_std": 0.5, "grad_norm": 0.04221472516655922, "kl": 0.19250980578362942, "learning_rate": 1.408249247213762e-06, "loss": 0.000962752616032958, "num_tokens": 100699145.0, "reward": 2.3519043922424316, "reward_std": 0.5390545129776001, "rewards/code_complexity_reward/mean": 0.9007812142372131, "rewards/code_complexity_reward/std": 0.13384781777858734, "rewards/code_execution_reward/mean": 0.361328125, "rewards/code_execution_reward/std": 0.48085519671440125, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 597, "step_time": 47.13580953236669 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 391.0, "completions/max_terminated_length": 391.0, "completions/mean_length": 117.66796875, "completions/mean_terminated_length": 117.66796875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21600811649113894, "epoch": 0.6818700114025086, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04208584502339363, "kl": 0.1914057086687535, "learning_rate": 1.3993029224460364e-06, "loss": 0.0009569242829456925, "num_tokens": 100828423.0, "reward": 2.2990236282348633, "reward_std": 0.5251889228820801, "rewards/code_complexity_reward/mean": 0.897167980670929, "rewards/code_complexity_reward/std": 0.13281089067459106, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.032150521874427795, "step": 598, "step_time": 50.901657382026315 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 121.66015625, "completions/mean_terminated_length": 120.89627838134766, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21175570087507367, "epoch": 0.6830102622576967, "frac_reward_zero_std": 0.453125, "grad_norm": 0.03864992409944534, "kl": 0.17801312054507434, "learning_rate": 1.390374048383379e-06, "loss": 0.0008900503744371235, "num_tokens": 100958625.0, "reward": 2.2754883766174316, "reward_std": 0.5632736682891846, "rewards/code_complexity_reward/mean": 0.8795897960662842, "rewards/code_complexity_reward/std": 0.17930728197097778, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02059757336974144, "step": 599, "step_time": 50.067975390702486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 115.998046875, "completions/mean_terminated_length": 115.22309112548828, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20176626881584525, "epoch": 0.6841505131128849, "frac_reward_zero_std": 0.5, "grad_norm": 0.04991946741938591, "kl": 0.19702109810896218, "learning_rate": 1.3814627665862112e-06, "loss": 0.0009850432397797704, "num_tokens": 101086172.0, "reward": 2.2843260765075684, "reward_std": 0.5203089118003845, "rewards/code_complexity_reward/mean": 0.896484375, "rewards/code_complexity_reward/std": 0.14309734106063843, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 600, "step_time": 52.07982299569994 }, { "epoch": 0.6841505131128849, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 191.7, "eval_completions/max_terminated_length": 191.7, "eval_completions/mean_length": 119.5075, "eval_completions/mean_terminated_length": 119.5075, "eval_completions/min_length": 75.26, "eval_completions/min_terminated_length": 75.26, "eval_entropy": 0.20738212570548056, "eval_frac_reward_zero_std": 0.38, "eval_kl": 0.18362467050552367, "eval_loss": 0.0009176870808005333, "eval_num_tokens": 101086172.0, "eval_reward": 2.2360000872612, "eval_reward_std": 0.38027198530733586, "eval_rewards/code_complexity_reward/mean": 0.8899999821186065, "eval_rewards/code_complexity_reward/std": 0.0744856046885252, "eval_rewards/code_execution_reward/mean": 0.255, "eval_rewards/code_execution_reward/std": 0.3185763734579086, "eval_rewards/code_syntax_reward/mean": 0.49375, "eval_rewards/code_syntax_reward/std": 0.01767766922712326, "eval_rewards/reasoning_present_reward_func/mean": 0.09975000157952309, "eval_rewards/reasoning_present_reward_func/std": 0.000707106813788414, "eval_rewards/xmlcount_reward_func/mean": 0.4975, "eval_rewards/xmlcount_reward_func/std": 0.007071067616343498, "eval_runtime": 409.7983, "eval_samples_per_second": 0.244, "eval_steps_per_second": 0.032, "step": 600 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 418.0, "completions/max_terminated_length": 418.0, "completions/mean_length": 117.6484375, "completions/mean_terminated_length": 117.6484375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21093214815482497, "epoch": 0.685290763968073, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.042373474687337875, "kl": 0.19621535786427557, "learning_rate": 1.3725692183360528e-06, "loss": 0.0009808092145249248, "num_tokens": 101213860.0, "reward": 2.2349610328674316, "reward_std": 0.5064321160316467, "rewards/code_complexity_reward/mean": 0.8919922113418579, "rewards/code_complexity_reward/std": 0.1461988389492035, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.028128057718276978, "step": 601, "step_time": 54.078924112953246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 131.591796875, "completions/mean_terminated_length": 129.34971618652344, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21749283466488123, "epoch": 0.6864310148232611, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.043704621493816376, "kl": 0.18472183600533754, "learning_rate": 1.3636935446332628e-06, "loss": 0.0009235702455043793, "num_tokens": 101355387.0, "reward": 2.2027833461761475, "reward_std": 0.5569507479667664, "rewards/code_complexity_reward/mean": 0.8702148199081421, "rewards/code_complexity_reward/std": 0.1895420253276825, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.04004169628024101, "step": 602, "step_time": 54.793247745372355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 447.0, "completions/mean_length": 120.33203125, "completions/mean_terminated_length": 119.56555938720703, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.22085502208210528, "epoch": 0.6875712656784493, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04034699127078056, "kl": 0.19818077143281698, "learning_rate": 1.3548358861948196e-06, "loss": 0.0009906090563163161, "num_tokens": 101485713.0, "reward": 2.190722703933716, "reward_std": 0.5037816166877747, "rewards/code_complexity_reward/mean": 0.8889648914337158, "rewards/code_complexity_reward/std": 0.1644248217344284, "rewards/code_execution_reward/mean": 0.21875, "rewards/code_execution_reward/std": 0.41380295157432556, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.028043000027537346, "step": 603, "step_time": 67.32321630511433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 419.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 116.794921875, "completions/mean_terminated_length": 116.794921875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.20344554795883596, "epoch": 0.6887115165336374, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04083910211920738, "kl": 0.18343870993703604, "learning_rate": 1.3459963834520806e-06, "loss": 0.0009171201381832361, "num_tokens": 101613388.0, "reward": 2.2896971702575684, "reward_std": 0.5289306640625, "rewards/code_complexity_reward/mean": 0.896191418170929, "rewards/code_complexity_reward/std": 0.1420087367296219, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.025244520977139473, "step": 604, "step_time": 43.85742867272347 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 119.326171875, "completions/mean_terminated_length": 118.55773162841797, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.20451722526922822, "epoch": 0.6898517673888256, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.036206893622875214, "kl": 0.1898713030386716, "learning_rate": 1.3371751765485568e-06, "loss": 0.0009493920370005071, "num_tokens": 101742299.0, "reward": 2.3083009719848633, "reward_std": 0.5608295798301697, "rewards/code_complexity_reward/mean": 0.885546863079071, "rewards/code_complexity_reward/std": 0.16243647038936615, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02697930857539177, "step": 605, "step_time": 57.34310621395707 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 119.19921875, "completions/mean_terminated_length": 119.19921875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20974254491738975, "epoch": 0.6909920182440137, "frac_reward_zero_std": 0.390625, "grad_norm": 0.16510780155658722, "kl": 0.36846011970192194, "learning_rate": 1.3283724053376985e-06, "loss": 0.0018454152159392834, "num_tokens": 101871169.0, "reward": 2.311328172683716, "reward_std": 0.5621771216392517, "rewards/code_complexity_reward/mean": 0.8870117664337158, "rewards/code_complexity_reward/std": 0.16422222554683685, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 606, "step_time": 58.24346481449902 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 474.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 122.259765625, "completions/mean_terminated_length": 122.259765625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.20964674814604223, "epoch": 0.6921322690992018, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.03800711780786514, "kl": 0.19286370056215674, "learning_rate": 1.319588209380664e-06, "loss": 0.0009643103112466633, "num_tokens": 102000130.0, "reward": 2.2357423305511475, "reward_std": 0.5146787166595459, "rewards/code_complexity_reward/mean": 0.890625, "rewards/code_complexity_reward/std": 0.15684013068675995, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 607, "step_time": 66.96354355756193 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 116.005859375, "completions/mean_terminated_length": 115.23091888427734, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21144345495849848, "epoch": 0.69327251995439, "frac_reward_zero_std": 0.5, "grad_norm": 0.041200943291187286, "kl": 0.19789389823563397, "learning_rate": 1.3108227279441243e-06, "loss": 0.0009895386174321175, "num_tokens": 102127801.0, "reward": 2.2518556118011475, "reward_std": 0.5161046385765076, "rewards/code_complexity_reward/mean": 0.9022461175918579, "rewards/code_complexity_reward/std": 0.14691822230815887, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.04182815924286842, "step": 608, "step_time": 57.75842207763344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 432.0, "completions/mean_length": 122.734375, "completions/mean_terminated_length": 121.20784759521484, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2093745197635144, "epoch": 0.6944127708095781, "frac_reward_zero_std": 0.3359375, "grad_norm": 0.05089733749628067, "kl": 0.18686977319885045, "learning_rate": 1.3020760999980386e-06, "loss": 0.0009342739940620959, "num_tokens": 102258337.0, "reward": 2.2626953125, "reward_std": 0.5683558583259583, "rewards/code_complexity_reward/mean": 0.8726562261581421, "rewards/code_complexity_reward/std": 0.17965060472488403, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.019055325537919998, "step": 609, "step_time": 58.982075398787856 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 109.22265625, "completions/mean_terminated_length": 109.22265625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2079720599576831, "epoch": 0.6955530216647663, "frac_reward_zero_std": 0.53125, "grad_norm": 0.04141247272491455, "kl": 0.18732947460375726, "learning_rate": 1.2933484642134631e-06, "loss": 0.0009366876911371946, "num_tokens": 102381827.0, "reward": 2.3056640625, "reward_std": 0.5164487957954407, "rewards/code_complexity_reward/mean": 0.9092773199081421, "rewards/code_complexity_reward/std": 0.1295289695262909, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 610, "step_time": 42.57374172937125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 113.228515625, "completions/mean_terminated_length": 113.228515625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21201977017335594, "epoch": 0.6966932725199544, "frac_reward_zero_std": 0.515625, "grad_norm": 0.04356147348880768, "kl": 0.19836835737805814, "learning_rate": 1.2846399589603453e-06, "loss": 0.0009917248971760273, "num_tokens": 102506024.0, "reward": 2.321289300918579, "reward_std": 0.5262050628662109, "rewards/code_complexity_reward/mean": 0.90185546875, "rewards/code_complexity_reward/std": 0.1288648098707199, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.028089813888072968, "step": 611, "step_time": 39.37341091129929 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 117.892578125, "completions/mean_terminated_length": 117.12133026123047, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22253412497229874, "epoch": 0.6978335233751425, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.041914552450180054, "kl": 0.20028420921880752, "learning_rate": 1.2759507223053341e-06, "loss": 0.001001705531962216, "num_tokens": 102636089.0, "reward": 2.232959270477295, "reward_std": 0.5444358587265015, "rewards/code_complexity_reward/mean": 0.88671875, "rewards/code_complexity_reward/std": 0.17368283867835999, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.04631038010120392, "step": 612, "step_time": 51.733739216811955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 489.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 116.76171875, "completions/mean_terminated_length": 116.76171875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.21581344725564122, "epoch": 0.6989737742303307, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04679529741406441, "kl": 0.19073901395313442, "learning_rate": 1.2672808920095914e-06, "loss": 0.0009538892190903425, "num_tokens": 102766071.0, "reward": 2.265869140625, "reward_std": 0.5081126689910889, "rewards/code_complexity_reward/mean": 0.8931640386581421, "rewards/code_complexity_reward/std": 0.14825187623500824, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.025282343849539757, "step": 613, "step_time": 54.246393457986414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 112.751953125, "completions/mean_terminated_length": 112.751953125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2109328715596348, "epoch": 0.7001140250855188, "frac_reward_zero_std": 0.453125, "grad_norm": 0.047752898186445236, "kl": 0.18733093538321555, "learning_rate": 1.2586306055266007e-06, "loss": 0.0009365258738398552, "num_tokens": 102892184.0, "reward": 2.3229494094848633, "reward_std": 0.5366286635398865, "rewards/code_complexity_reward/mean": 0.89599609375, "rewards/code_complexity_reward/std": 0.13862118124961853, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.03484956547617912, "step": 614, "step_time": 42.16377037577331 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 325.0, "completions/max_terminated_length": 325.0, "completions/mean_length": 121.65625, "completions/mean_terminated_length": 121.65625, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.2240888811647892, "epoch": 0.701254275940707, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.04558137431740761, "kl": 0.1880849493900314, "learning_rate": 1.2500000000000007e-06, "loss": 0.0009404792217537761, "num_tokens": 103023276.0, "reward": 2.27294921875, "reward_std": 0.5493709444999695, "rewards/code_complexity_reward/mean": 0.8827148675918579, "rewards/code_complexity_reward/std": 0.16760383546352386, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.04465661570429802, "step": 615, "step_time": 55.79206874407828 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 121.076171875, "completions/mean_terminated_length": 119.54314422607422, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21522735059261322, "epoch": 0.7023945267958951, "frac_reward_zero_std": 0.421875, "grad_norm": 0.05607999488711357, "kl": 0.21606221329420805, "learning_rate": 1.2413892122613968e-06, "loss": 0.0010804148623719811, "num_tokens": 103155399.0, "reward": 2.2350587844848633, "reward_std": 0.5749235153198242, "rewards/code_complexity_reward/mean": 0.8740234375, "rewards/code_complexity_reward/std": 0.19511905312538147, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.025821086019277573, "step": 616, "step_time": 55.07367256190628 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 114.083984375, "completions/mean_terminated_length": 112.5235366821289, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.20654511963948607, "epoch": 0.7035347776510832, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04165871813893318, "kl": 0.19451805076096207, "learning_rate": 1.2327983788282033e-06, "loss": 0.0009728380828164518, "num_tokens": 103282814.0, "reward": 2.2265625, "reward_std": 0.5417704582214355, "rewards/code_complexity_reward/mean": 0.884570300579071, "rewards/code_complexity_reward/std": 0.18000927567481995, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.022032126784324646, "step": 617, "step_time": 54.702095499262214 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 118.826171875, "completions/mean_terminated_length": 117.28431701660156, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.20863183005712926, "epoch": 0.7046750285062714, "frac_reward_zero_std": 0.390625, "grad_norm": 0.04569806158542633, "kl": 0.20856750255916268, "learning_rate": 1.2242276359014724e-06, "loss": 0.00104317138902843, "num_tokens": 103409813.0, "reward": 2.218554973602295, "reward_std": 0.525653064250946, "rewards/code_complexity_reward/mean": 0.888671875, "rewards/code_complexity_reward/std": 0.1623837649822235, "rewards/code_execution_reward/mean": 0.25, "rewards/code_execution_reward/std": 0.43343618512153625, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.494140625, "rewards/xmlcount_reward_func/std": 0.04589129984378815, "step": 618, "step_time": 59.248330575414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 118.291015625, "completions/mean_terminated_length": 117.52054595947266, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21595337032340467, "epoch": 0.7058152793614595, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04154850170016289, "kl": 0.19217635155655444, "learning_rate": 1.215677119363736e-06, "loss": 0.0009609407279640436, "num_tokens": 103540702.0, "reward": 2.2545411586761475, "reward_std": 0.5205795168876648, "rewards/code_complexity_reward/mean": 0.8939453363418579, "rewards/code_complexity_reward/std": 0.14759141206741333, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03521692752838135, "step": 619, "step_time": 61.86364107020199 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 366.0, "completions/max_terminated_length": 366.0, "completions/mean_length": 113.25, "completions/mean_terminated_length": 113.25, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22138028079643846, "epoch": 0.7069555302166477, "frac_reward_zero_std": 0.40625, "grad_norm": 0.04531016945838928, "kl": 0.1976558342576027, "learning_rate": 1.2071469647768564e-06, "loss": 0.0009882212616503239, "num_tokens": 103666714.0, "reward": 2.2815918922424316, "reward_std": 0.498269259929657, "rewards/code_complexity_reward/mean": 0.9046875238418579, "rewards/code_complexity_reward/std": 0.11934876441955566, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02514021471142769, "step": 620, "step_time": 39.66229255218059 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 435.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 115.056640625, "completions/mean_terminated_length": 115.056640625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.20960443676449358, "epoch": 0.7080957810718358, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.046164605766534805, "kl": 0.200406325282529, "learning_rate": 1.198637307379867e-06, "loss": 0.0010017813183367252, "num_tokens": 103793355.0, "reward": 2.2026853561401367, "reward_std": 0.542486846446991, "rewards/code_complexity_reward/mean": 0.876757800579071, "rewards/code_complexity_reward/std": 0.1894911229610443, "rewards/code_execution_reward/mean": 0.248046875, "rewards/code_execution_reward/std": 0.4323015511035919, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.030623575672507286, "step": 621, "step_time": 45.318390749394894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 460.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 115.291015625, "completions/mean_terminated_length": 115.291015625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21009586239233613, "epoch": 0.7092360319270239, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.04026442766189575, "kl": 0.2199591773096472, "learning_rate": 1.190148282086837e-06, "loss": 0.001099850982427597, "num_tokens": 103919540.0, "reward": 2.3017091751098633, "reward_std": 0.5081077218055725, "rewards/code_complexity_reward/mean": 0.9093749523162842, "rewards/code_complexity_reward/std": 0.11321662366390228, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.038484130054712296, "step": 622, "step_time": 55.40420723054558 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 393.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 115.44921875, "completions/mean_terminated_length": 115.44921875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2098028443288058, "epoch": 0.7103762827822121, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04096947982907295, "kl": 0.20901071652770042, "learning_rate": 1.1816800234847304e-06, "loss": 0.0010449484689161181, "num_tokens": 104046898.0, "reward": 2.2715821266174316, "reward_std": 0.5177741050720215, "rewards/code_complexity_reward/mean": 0.899218738079071, "rewards/code_complexity_reward/std": 0.14581364393234253, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03480498120188713, "step": 623, "step_time": 51.54913973994553 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 122.4609375, "completions/mean_terminated_length": 120.93334197998047, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21341692027635872, "epoch": 0.7115165336374002, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.042904019355773926, "kl": 0.19970400305464864, "learning_rate": 1.1732326658312693e-06, "loss": 0.0009986039949581027, "num_tokens": 104176738.0, "reward": 2.282275676727295, "reward_std": 0.5592485070228577, "rewards/code_complexity_reward/mean": 0.88525390625, "rewards/code_complexity_reward/std": 0.1697596162557602, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.04004169628024101, "step": 624, "step_time": 59.73915899079293 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 112.13671875, "completions/mean_terminated_length": 110.56863403320312, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.20653290464542806, "epoch": 0.7126567844925884, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.048394594341516495, "kl": 0.23474089172668755, "learning_rate": 1.1648063430528084e-06, "loss": 0.0011740627232939005, "num_tokens": 104301920.0, "reward": 2.3222169876098633, "reward_std": 0.5727896094322205, "rewards/code_complexity_reward/mean": 0.8935546875, "rewards/code_complexity_reward/std": 0.1708945631980896, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02960825525224209, "step": 625, "step_time": 52.858356235548854 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 110.771484375, "completions/mean_terminated_length": 109.19804382324219, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.20436089183203876, "epoch": 0.7137970353477765, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04678674042224884, "kl": 0.19425602909177542, "learning_rate": 1.1564011887422098e-06, "loss": 0.0009712455794215202, "num_tokens": 104423987.0, "reward": 2.285449504852295, "reward_std": 0.5743290781974792, "rewards/code_complexity_reward/mean": 0.8885741829872131, "rewards/code_complexity_reward/std": 0.17874129116535187, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03562243655323982, "step": 626, "step_time": 50.25099242851138 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 474.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 116.814453125, "completions/mean_terminated_length": 116.814453125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21420301403850317, "epoch": 0.7149372862029647, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.04092530533671379, "kl": 0.18675923021510243, "learning_rate": 1.1480173361567287e-06, "loss": 0.0009338571690022945, "num_tokens": 104551160.0, "reward": 2.318164110183716, "reward_std": 0.5425123572349548, "rewards/code_complexity_reward/mean": 0.8963866829872131, "rewards/code_complexity_reward/std": 0.15067243576049805, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 627, "step_time": 56.693802998401225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 120.02734375, "completions/mean_terminated_length": 120.02734375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.20816903025843203, "epoch": 0.7160775370581528, "frac_reward_zero_std": 0.46875, "grad_norm": 0.043337076902389526, "kl": 0.17841266316827387, "learning_rate": 1.1396549182158933e-06, "loss": 0.0008920262916944921, "num_tokens": 104680638.0, "reward": 2.2637696266174316, "reward_std": 0.5301094651222229, "rewards/code_complexity_reward/mean": 0.8929687142372131, "rewards/code_complexity_reward/std": 0.16202417016029358, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 628, "step_time": 49.94101936277002 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 494.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 120.0078125, "completions/mean_terminated_length": 120.0078125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21358178975060582, "epoch": 0.7172177879133409, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.04431420937180519, "kl": 0.17959491745568812, "learning_rate": 1.1313140674994053e-06, "loss": 0.0008980946149677038, "num_tokens": 104809934.0, "reward": 2.2718751430511475, "reward_std": 0.4867169260978699, "rewards/code_complexity_reward/mean": 0.8955078125, "rewards/code_complexity_reward/std": 0.11828576028347015, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.019055325537919998, "step": 629, "step_time": 56.500056363642216 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 120.01953125, "completions/mean_terminated_length": 120.01953125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.2183632820378989, "epoch": 0.7183580387685291, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.04868370294570923, "kl": 0.18495361565146595, "learning_rate": 1.1229949162450331e-06, "loss": 0.0009246289264410734, "num_tokens": 104940844.0, "reward": 2.2437987327575684, "reward_std": 0.552807629108429, "rewards/code_complexity_reward/mean": 0.8759765028953552, "rewards/code_complexity_reward/std": 0.18346261978149414, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.024042511358857155, "step": 630, "step_time": 47.63433662150055 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 485.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 112.681640625, "completions/mean_terminated_length": 112.681640625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21857498330064118, "epoch": 0.7194982896237172, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.045975781977176666, "kl": 0.17599978588987142, "learning_rate": 1.1146975963465179e-06, "loss": 0.0008798744529485703, "num_tokens": 105065909.0, "reward": 2.3019044399261475, "reward_std": 0.537941575050354, "rewards/code_complexity_reward/mean": 0.89453125, "rewards/code_complexity_reward/std": 0.15040510892868042, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 631, "step_time": 57.40409577265382 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 118.26171875, "completions/mean_terminated_length": 117.49119567871094, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21571020386181772, "epoch": 0.7206385404789054, "frac_reward_zero_std": 0.421875, "grad_norm": 0.050102680921554565, "kl": 0.20689155207946897, "learning_rate": 1.106422239351481e-06, "loss": 0.0010343518806621432, "num_tokens": 105193959.0, "reward": 2.2386720180511475, "reward_std": 0.5191735625267029, "rewards/code_complexity_reward/mean": 0.8924804925918579, "rewards/code_complexity_reward/std": 0.15698698163032532, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.04470740631222725, "step": 632, "step_time": 54.76463741064072 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 327.0, "completions/max_terminated_length": 327.0, "completions/mean_length": 110.599609375, "completions/mean_terminated_length": 110.599609375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.20783394714817405, "epoch": 0.7217787913340935, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04166026785969734, "kl": 0.19804531952831894, "learning_rate": 1.0981689764593384e-06, "loss": 0.000990541884675622, "num_tokens": 105319350.0, "reward": 2.2936525344848633, "reward_std": 0.5173990726470947, "rewards/code_complexity_reward/mean": 0.8984375, "rewards/code_complexity_reward/std": 0.13471540808677673, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 633, "step_time": 41.64607692975551 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 122.900390625, "completions/mean_terminated_length": 120.6070785522461, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22059549717232585, "epoch": 0.7229190421892816, "frac_reward_zero_std": 0.390625, "grad_norm": 0.043047744780778885, "kl": 0.19863275531679392, "learning_rate": 1.0899379385192222e-06, "loss": 0.000993094639852643, "num_tokens": 105448967.0, "reward": 2.270068645477295, "reward_std": 0.5474177598953247, "rewards/code_complexity_reward/mean": 0.8861327767372131, "rewards/code_complexity_reward/std": 0.16449259221553802, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09902344644069672, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.492919921875, "rewards/xmlcount_reward_func/std": 0.0516832061111927, "step": 634, "step_time": 61.06926984898746 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 118.7578125, "completions/mean_terminated_length": 117.9882583618164, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.21428279508836567, "epoch": 0.7240592930444698, "frac_reward_zero_std": 0.5, "grad_norm": 0.04261765629053116, "kl": 0.22142145596444607, "learning_rate": 1.0817292560279038e-06, "loss": 0.0011068836320191622, "num_tokens": 105579003.0, "reward": 2.2692384719848633, "reward_std": 0.5137550234794617, "rewards/code_complexity_reward/mean": 0.9012694954872131, "rewards/code_complexity_reward/std": 0.13771744072437286, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.494140625, "rewards/xmlcount_reward_func/std": 0.04784845933318138, "step": 635, "step_time": 61.890731601044536 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 445.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 122.775390625, "completions/mean_terminated_length": 122.775390625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21216427558101714, "epoch": 0.7251995438996579, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.04801640659570694, "kl": 0.18725601863116026, "learning_rate": 1.0735430591277268e-06, "loss": 0.0009360210387967527, "num_tokens": 105710328.0, "reward": 2.242968797683716, "reward_std": 0.5540894269943237, "rewards/code_complexity_reward/mean": 0.8773437142372131, "rewards/code_complexity_reward/std": 0.18062832951545715, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 636, "step_time": 52.13508326001465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 468.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 122.09765625, "completions/mean_terminated_length": 122.09765625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.20913374330848455, "epoch": 0.7263397947548461, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04582206904888153, "kl": 0.1940056390594691, "learning_rate": 1.0653794776045435e-06, "loss": 0.0009702413808554411, "num_tokens": 105842970.0, "reward": 2.246875047683716, "reward_std": 0.5661090612411499, "rewards/code_complexity_reward/mean": 0.877734363079071, "rewards/code_complexity_reward/std": 0.1801345944404602, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.042552899569272995, "step": 637, "step_time": 64.5843236381188 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 122.701171875, "completions/mean_terminated_length": 121.17451477050781, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21963675180450082, "epoch": 0.7274800456100342, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.047059979289770126, "kl": 0.17745111719705164, "learning_rate": 1.0572386408856553e-06, "loss": 0.0008871458703652024, "num_tokens": 105973065.0, "reward": 2.259814739227295, "reward_std": 0.5259532332420349, "rewards/code_complexity_reward/mean": 0.8871093392372131, "rewards/code_complexity_reward/std": 0.15596377849578857, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 638, "step_time": 56.29531300906092 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 119.234375, "completions/mean_terminated_length": 119.234375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21166898892261088, "epoch": 0.7286202964652223, "frac_reward_zero_std": 0.4375, "grad_norm": 0.049074266105890274, "kl": 0.19275012076832354, "learning_rate": 1.0491206780377636e-06, "loss": 0.0009634458110667765, "num_tokens": 106101325.0, "reward": 2.2383790016174316, "reward_std": 0.5336974263191223, "rewards/code_complexity_reward/mean": 0.8839843273162842, "rewards/code_complexity_reward/std": 0.16899420320987701, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 639, "step_time": 44.91130790952593 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 489.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 110.5859375, "completions/mean_terminated_length": 110.5859375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.20411920687183738, "epoch": 0.7297605473204105, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04880132898688316, "kl": 0.2009009876055643, "learning_rate": 1.0410257177649217e-06, "loss": 0.0010046690003946424, "num_tokens": 106224581.0, "reward": 2.2437989711761475, "reward_std": 0.5289901494979858, "rewards/code_complexity_reward/mean": 0.893261730670929, "rewards/code_complexity_reward/std": 0.16168905794620514, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02960825525224209, "step": 640, "step_time": 48.36463121138513 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 475.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 114.9140625, "completions/mean_terminated_length": 114.9140625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21004499169066548, "epoch": 0.7309007981755986, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.043726883828639984, "kl": 0.21027858508750796, "learning_rate": 1.0329538884064948e-06, "loss": 0.0010512100998312235, "num_tokens": 106350705.0, "reward": 2.2535157203674316, "reward_std": 0.5534206032752991, "rewards/code_complexity_reward/mean": 0.8807617425918579, "rewards/code_complexity_reward/std": 0.17845280468463898, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03480498120188713, "step": 641, "step_time": 64.60980403609574 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 352.0, "completions/mean_length": 110.650390625, "completions/mean_terminated_length": 109.8649673461914, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2186566460877657, "epoch": 0.7320410490307868, "frac_reward_zero_std": 0.5, "grad_norm": 0.044289037585258484, "kl": 0.1907195373205468, "learning_rate": 1.0249053179351257e-06, "loss": 0.0009536681463941932, "num_tokens": 106473966.0, "reward": 2.2767090797424316, "reward_std": 0.5283598899841309, "rewards/code_complexity_reward/mean": 0.8986327648162842, "rewards/code_complexity_reward/std": 0.14614446461200714, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.025244520977139473, "step": 642, "step_time": 57.272872186265886 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 395.0, "completions/max_terminated_length": 395.0, "completions/mean_length": 117.583984375, "completions/mean_terminated_length": 117.583984375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21858490677550435, "epoch": 0.7331812998859749, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.04364917799830437, "kl": 0.20920162193942815, "learning_rate": 1.0168801339547046e-06, "loss": 0.0010458327597007155, "num_tokens": 106604421.0, "reward": 2.3195314407348633, "reward_std": 0.526278018951416, "rewards/code_complexity_reward/mean": 0.9025390148162842, "rewards/code_complexity_reward/std": 0.13739751279354095, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.04044608399271965, "step": 643, "step_time": 42.878269638866186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 122.2109375, "completions/mean_terminated_length": 121.4481430053711, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21047398750670254, "epoch": 0.734321550741163, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.03873355686664581, "kl": 0.17205191089306027, "learning_rate": 1.0088784636983472e-06, "loss": 0.0008601705194450915, "num_tokens": 106735713.0, "reward": 2.299560546875, "reward_std": 0.5234833359718323, "rewards/code_complexity_reward/mean": 0.8887695074081421, "rewards/code_complexity_reward/std": 0.1400861144065857, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 644, "step_time": 56.08535035420209 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 114.396484375, "completions/mean_terminated_length": 113.61839294433594, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.2111971080303192, "epoch": 0.7354618015963512, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.17310090363025665, "kl": 0.4051952586742118, "learning_rate": 1.0009004340263778e-06, "loss": 0.002029441064223647, "num_tokens": 106864720.0, "reward": 2.2997071743011475, "reward_std": 0.5347214937210083, "rewards/code_complexity_reward/mean": 0.9072265625, "rewards/code_complexity_reward/std": 0.13427533209323883, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09902344644069672, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.49169921875, "rewards/xmlcount_reward_func/std": 0.05410672351717949, "step": 645, "step_time": 61.90518994163722 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 123.625, "completions/mean_terminated_length": 122.10196685791016, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21616630745120347, "epoch": 0.7366020524515393, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.0427250862121582, "kl": 0.186589524615556, "learning_rate": 9.929461714243166e-07, "loss": 0.0009328627493232489, "num_tokens": 106994552.0, "reward": 2.216552734375, "reward_std": 0.48313355445861816, "rewards/code_complexity_reward/mean": 0.8936523199081421, "rewards/code_complexity_reward/std": 0.13764776289463043, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.021246958523988724, "step": 646, "step_time": 59.07160801999271 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 118.310546875, "completions/mean_terminated_length": 117.54011535644531, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21111346315592527, "epoch": 0.7377423033067275, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.04351595416665077, "kl": 0.18843431712593883, "learning_rate": 9.850158020008757e-07, "loss": 0.0009422763250768185, "num_tokens": 107123151.0, "reward": 2.2947754859924316, "reward_std": 0.5623835325241089, "rewards/code_complexity_reward/mean": 0.8844726085662842, "rewards/code_complexity_reward/std": 0.16730894148349762, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.01981583796441555, "step": 647, "step_time": 50.26320345606655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 113.72265625, "completions/mean_terminated_length": 113.72265625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21026787883602083, "epoch": 0.7388825541619156, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04602813720703125, "kl": 0.18925504852086306, "learning_rate": 9.771094514859587e-07, "loss": 0.0009460902074351907, "num_tokens": 107247773.0, "reward": 2.2640626430511475, "reward_std": 0.4815022647380829, "rewards/code_complexity_reward/mean": 0.913378894329071, "rewards/code_complexity_reward/std": 0.1058545857667923, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49462890625, "rewards/xmlcount_reward_func/std": 0.04320749267935753, "step": 648, "step_time": 46.71594808995724 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 424.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 114.482421875, "completions/mean_terminated_length": 114.482421875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.213428903138265, "epoch": 0.7400228050171037, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04262743145227432, "kl": 0.20696093374863267, "learning_rate": 9.692272452286686e-07, "loss": 0.0010349315125495195, "num_tokens": 107375548.0, "reward": 2.2159180641174316, "reward_std": 0.503114640712738, "rewards/code_complexity_reward/mean": 0.8924804925918579, "rewards/code_complexity_reward/std": 0.15918424725532532, "rewards/code_execution_reward/mean": 0.23828125, "rewards/code_execution_reward/std": 0.42644867300987244, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.025862684473395348, "step": 649, "step_time": 43.5037845056504 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 428.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 116.984375, "completions/mean_terminated_length": 116.984375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21888003638014197, "epoch": 0.7411630558722919, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.055451471358537674, "kl": 0.20001960429362953, "learning_rate": 9.613693081953195e-07, "loss": 0.0010000059846788645, "num_tokens": 107504208.0, "reward": 2.2696290016174316, "reward_std": 0.508752167224884, "rewards/code_complexity_reward/mean": 0.899609386920929, "rewards/code_complexity_reward/std": 0.1353759765625, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02333279326558113, "step": 650, "step_time": 53.81225309334695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 438.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 114.734375, "completions/mean_terminated_length": 114.734375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21080623962916434, "epoch": 0.74230330672748, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.04022161662578583, "kl": 0.19228309276513755, "learning_rate": 9.535357649674554e-07, "loss": 0.0009615116869099438, "num_tokens": 107630196.0, "reward": 2.2550294399261475, "reward_std": 0.5461821556091309, "rewards/code_complexity_reward/mean": 0.8871093392372131, "rewards/code_complexity_reward/std": 0.16331882774829865, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.04358936473727226, "step": 651, "step_time": 63.23565281461924 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 113.18359375, "completions/mean_terminated_length": 112.40312957763672, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.20578471361659467, "epoch": 0.7434435575826682, "frac_reward_zero_std": 0.53125, "grad_norm": 0.04469587653875351, "kl": 0.18666076532099396, "learning_rate": 9.457267397398756e-07, "loss": 0.0009333117632195354, "num_tokens": 107756058.0, "reward": 2.2684082984924316, "reward_std": 0.5525773763656616, "rewards/code_complexity_reward/mean": 0.8919922113418579, "rewards/code_complexity_reward/std": 0.1740063726902008, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.019755469635128975, "step": 652, "step_time": 66.60980140790343 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 465.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 117.66796875, "completions/mean_terminated_length": 117.66796875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21451112302020192, "epoch": 0.7445838084378563, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.045496825128793716, "kl": 0.182030807598494, "learning_rate": 9.379423563186652e-07, "loss": 0.0009100954048335552, "num_tokens": 107884564.0, "reward": 2.3436524868011475, "reward_std": 0.5510719418525696, "rewards/code_complexity_reward/mean": 0.8896484375, "rewards/code_complexity_reward/std": 0.15536798536777496, "rewards/code_execution_reward/mean": 0.365234375, "rewards/code_execution_reward/std": 0.4819667339324951, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 653, "step_time": 48.71080093923956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 119.67578125, "completions/mean_terminated_length": 118.90802001953125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21902832738123834, "epoch": 0.7457240592930444, "frac_reward_zero_std": 0.390625, "grad_norm": 0.04971427470445633, "kl": 0.21154572092927992, "learning_rate": 9.301827381192321e-07, "loss": 0.0010578951332718134, "num_tokens": 108012366.0, "reward": 2.317138910293579, "reward_std": 0.5636063814163208, "rewards/code_complexity_reward/mean": 0.8878905773162842, "rewards/code_complexity_reward/std": 0.16412563621997833, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03516262024641037, "step": 654, "step_time": 50.18397238012403 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 484.0, "completions/mean_length": 116.455078125, "completions/mean_terminated_length": 115.68101501464844, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2290351812262088, "epoch": 0.7468643101482326, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.042487166821956635, "kl": 0.19412879145238549, "learning_rate": 9.224480081643516e-07, "loss": 0.0009706277633085847, "num_tokens": 108139935.0, "reward": 2.295459032058716, "reward_std": 0.5088359713554382, "rewards/code_complexity_reward/mean": 0.9029296636581421, "rewards/code_complexity_reward/std": 0.12339462339878082, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03686080500483513, "step": 655, "step_time": 60.406727441586554 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 118.34375, "completions/mean_terminated_length": 116.80001068115234, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21617771917954087, "epoch": 0.7480045610034207, "frac_reward_zero_std": 0.5, "grad_norm": 0.055640481412410736, "kl": 0.25437955337110907, "learning_rate": 9.147382890822143e-07, "loss": 0.0012721531093120575, "num_tokens": 108271091.0, "reward": 2.293017864227295, "reward_std": 0.5662139058113098, "rewards/code_complexity_reward/mean": 0.8868163824081421, "rewards/code_complexity_reward/std": 0.17752103507518768, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.012304205447435379, "step": 656, "step_time": 60.922236274927855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 112.3984375, "completions/mean_terminated_length": 112.3984375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2107440682593733, "epoch": 0.7491448118586089, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04529442638158798, "kl": 0.19662223244085908, "learning_rate": 9.070537031044829e-07, "loss": 0.0009828832698985934, "num_tokens": 108396775.0, "reward": 2.2826170921325684, "reward_std": 0.5300261378288269, "rewards/code_complexity_reward/mean": 0.901074230670929, "rewards/code_complexity_reward/std": 0.14809228479862213, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.013465282507240772, "step": 657, "step_time": 50.039923558942974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 445.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 111.712890625, "completions/mean_terminated_length": 111.712890625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20539070409722626, "epoch": 0.750285062713797, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04242321103811264, "kl": 0.1986009394749999, "learning_rate": 8.993943720643536e-07, "loss": 0.0009930443484336138, "num_tokens": 108521776.0, "reward": 2.344970941543579, "reward_std": 0.5514838695526123, "rewards/code_complexity_reward/mean": 0.8991210460662842, "rewards/code_complexity_reward/std": 0.14461706578731537, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.492919921875, "rewards/xmlcount_reward_func/std": 0.05399789288640022, "step": 658, "step_time": 45.72740554437041 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 497.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 119.552734375, "completions/mean_terminated_length": 119.552734375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21257653250359, "epoch": 0.7514253135689852, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.03976653143763542, "kl": 0.1882807785877958, "learning_rate": 8.917604173946268e-07, "loss": 0.0009414787054993212, "num_tokens": 108650579.0, "reward": 2.3238770961761475, "reward_std": 0.5078155994415283, "rewards/code_complexity_reward/mean": 0.9015624523162842, "rewards/code_complexity_reward/std": 0.11819541454315186, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.03260336071252823, "step": 659, "step_time": 62.148391062393785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 115.759765625, "completions/mean_terminated_length": 114.98434448242188, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2063054316677153, "epoch": 0.7525655644241733, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.04206797480583191, "kl": 0.20123572670854628, "learning_rate": 8.841519601257756e-07, "loss": 0.0010060227941721678, "num_tokens": 108778888.0, "reward": 2.2708497047424316, "reward_std": 0.55119389295578, "rewards/code_complexity_reward/mean": 0.8918944597244263, "rewards/code_complexity_reward/std": 0.16937123239040375, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.042219679802656174, "step": 660, "step_time": 57.806551018729806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 381.0, "completions/max_terminated_length": 381.0, "completions/mean_length": 108.716796875, "completions/mean_terminated_length": 108.716796875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.213139965897426, "epoch": 0.7537058152793614, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.04438921809196472, "kl": 0.19520708685740829, "learning_rate": 8.765691208840374e-07, "loss": 0.0009761472465470433, "num_tokens": 108903295.0, "reward": 2.2923831939697266, "reward_std": 0.5293703079223633, "rewards/code_complexity_reward/mean": 0.9040038585662842, "rewards/code_complexity_reward/std": 0.1441575288772583, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03206123411655426, "step": 661, "step_time": 47.24963322095573 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 500.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 121.171875, "completions/mean_terminated_length": 121.171875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21880270494148135, "epoch": 0.7548460661345496, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04104836657643318, "kl": 0.18322690308559686, "learning_rate": 8.690120198894919e-07, "loss": 0.000916220829822123, "num_tokens": 109034803.0, "reward": 2.2096681594848633, "reward_std": 0.4920222759246826, "rewards/code_complexity_reward/mean": 0.8889647722244263, "rewards/code_complexity_reward/std": 0.15527357161045074, "rewards/code_execution_reward/mean": 0.232421875, "rewards/code_execution_reward/std": 0.42278963327407837, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 662, "step_time": 57.22671654820442 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 489.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 120.5859375, "completions/mean_terminated_length": 120.5859375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21730291959829628, "epoch": 0.7559863169897377, "frac_reward_zero_std": 0.515625, "grad_norm": 0.0435837060213089, "kl": 0.1772701342124492, "learning_rate": 8.614807769541589e-07, "loss": 0.0008862497052177787, "num_tokens": 109164575.0, "reward": 2.3180665969848633, "reward_std": 0.5096603631973267, "rewards/code_complexity_reward/mean": 0.9072265625, "rewards/code_complexity_reward/std": 0.11631374806165695, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 663, "step_time": 48.402408053167164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 113.734375, "completions/mean_terminated_length": 112.95498657226562, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21882636472582817, "epoch": 0.7571265678449259, "frac_reward_zero_std": 0.515625, "grad_norm": 0.04672553762793541, "kl": 0.19939367659389973, "learning_rate": 8.539755114800996e-07, "loss": 0.0009969323873519897, "num_tokens": 109289631.0, "reward": 2.2619142532348633, "reward_std": 0.5429495573043823, "rewards/code_complexity_reward/mean": 0.8942382335662842, "rewards/code_complexity_reward/std": 0.16584895551204681, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.031092895194888115, "step": 664, "step_time": 53.00670532695949 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 452.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 118.60546875, "completions/mean_terminated_length": 118.60546875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.20408618613146245, "epoch": 0.758266818700114, "frac_reward_zero_std": 0.484375, "grad_norm": 0.046100594103336334, "kl": 0.1840546578168869, "learning_rate": 8.464963424575215e-07, "loss": 0.0009203385561704636, "num_tokens": 109419853.0, "reward": 2.27392578125, "reward_std": 0.5347890257835388, "rewards/code_complexity_reward/mean": 0.8873046636581421, "rewards/code_complexity_reward/std": 0.15395936369895935, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 665, "step_time": 54.99929126165807 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 118.513671875, "completions/mean_terminated_length": 118.513671875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.22008713753893971, "epoch": 0.7594070695553021, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.042294133454561234, "kl": 0.20139397913590074, "learning_rate": 8.390433884628948e-07, "loss": 0.0010070072021335363, "num_tokens": 109548396.0, "reward": 2.239501953125, "reward_std": 0.5854384899139404, "rewards/code_complexity_reward/mean": 0.874316394329071, "rewards/code_complexity_reward/std": 0.1987227201461792, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03686080500483513, "step": 666, "step_time": 40.598663362674415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 397.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 115.927734375, "completions/mean_terminated_length": 115.927734375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22218274464830756, "epoch": 0.7605473204104903, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04700295627117157, "kl": 0.18906623171642423, "learning_rate": 8.316167676570666e-07, "loss": 0.0009451808291487396, "num_tokens": 109674499.0, "reward": 2.325488567352295, "reward_std": 0.5777912735939026, "rewards/code_complexity_reward/mean": 0.8913085460662842, "rewards/code_complexity_reward/std": 0.1712382733821869, "rewards/code_execution_reward/mean": 0.35546875, "rewards/code_execution_reward/std": 0.47912323474884033, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.043265506625175476, "step": 667, "step_time": 41.98792881797999 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 112.962890625, "completions/mean_terminated_length": 112.962890625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2176629714667797, "epoch": 0.7616875712656784, "frac_reward_zero_std": 0.421875, "grad_norm": 0.04642092436552048, "kl": 0.19124648300930858, "learning_rate": 8.242165977833974e-07, "loss": 0.000955997034907341, "num_tokens": 109800064.0, "reward": 2.3296875953674316, "reward_std": 0.5313346982002258, "rewards/code_complexity_reward/mean": 0.9032226800918579, "rewards/code_complexity_reward/std": 0.1317664384841919, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.031092895194888115, "step": 668, "step_time": 52.51577632967383 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 110.916015625, "completions/mean_terminated_length": 110.13111114501953, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.20570916286669672, "epoch": 0.7628278221208666, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.046864401549100876, "kl": 0.19060588453430682, "learning_rate": 8.168429961658822e-07, "loss": 0.000953114649746567, "num_tokens": 109924861.0, "reward": 2.320556879043579, "reward_std": 0.532319962978363, "rewards/code_complexity_reward/mean": 0.8995116949081421, "rewards/code_complexity_reward/std": 0.13994252681732178, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.03160629794001579, "step": 669, "step_time": 60.547566692344844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 486.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 121.8984375, "completions/mean_terminated_length": 121.8984375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.220805469667539, "epoch": 0.7639680729760547, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.05142267420887947, "kl": 0.19163693743757904, "learning_rate": 8.094960797073023e-07, "loss": 0.0009581368649378419, "num_tokens": 110057213.0, "reward": 2.2325685024261475, "reward_std": 0.5268222689628601, "rewards/code_complexity_reward/mean": 0.887499988079071, "rewards/code_complexity_reward/std": 0.15876233577728271, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09921874850988388, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.04358936473727226, "step": 670, "step_time": 56.45463081821799 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 120.025390625, "completions/mean_terminated_length": 117.71513366699219, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.21145996288396418, "epoch": 0.7651083238312428, "frac_reward_zero_std": 0.421875, "grad_norm": 0.045424796640872955, "kl": 0.19539793115109205, "learning_rate": 8.021759648873642e-07, "loss": 0.0009771683253347874, "num_tokens": 110187186.0, "reward": 2.366748094558716, "reward_std": 0.5612216591835022, "rewards/code_complexity_reward/mean": 0.8943359851837158, "rewards/code_complexity_reward/std": 0.15072281658649445, "rewards/code_execution_reward/mean": 0.38671875, "rewards/code_execution_reward/std": 0.48747459053993225, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02746524289250374, "step": 671, "step_time": 58.67320317402482 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 477.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 117.53515625, "completions/mean_terminated_length": 117.53515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21795076597481966, "epoch": 0.766248574686431, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.05079904943704605, "kl": 0.2058538420824334, "learning_rate": 7.94882767760852e-07, "loss": 0.0010291689541190863, "num_tokens": 110317396.0, "reward": 2.2425293922424316, "reward_std": 0.5468010306358337, "rewards/code_complexity_reward/mean": 0.881054699420929, "rewards/code_complexity_reward/std": 0.18333841860294342, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.01981583796441555, "step": 672, "step_time": 48.55769520904869 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 511.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 118.74609375, "completions/mean_terminated_length": 118.74609375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2121827839873731, "epoch": 0.7673888255416191, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.046324051916599274, "kl": 0.19740496901795268, "learning_rate": 7.876166039557967e-07, "loss": 0.0009872299851849675, "num_tokens": 110444882.0, "reward": 2.3072266578674316, "reward_std": 0.5399014949798584, "rewards/code_complexity_reward/mean": 0.889843761920929, "rewards/code_complexity_reward/std": 0.14706647396087646, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.0347534641623497, "step": 673, "step_time": 55.86690554860979 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 453.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 114.140625, "completions/mean_terminated_length": 114.140625, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.2159816175699234, "epoch": 0.7685290763968073, "frac_reward_zero_std": 0.5, "grad_norm": 0.0384770892560482, "kl": 0.20299849449656904, "learning_rate": 7.8037758867163e-07, "loss": 0.0010149157606065273, "num_tokens": 110572830.0, "reward": 2.3219728469848633, "reward_std": 0.5120233297348022, "rewards/code_complexity_reward/mean": 0.9039062261581421, "rewards/code_complexity_reward/std": 0.12007163465023041, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49365234375, "rewards/xmlcount_reward_func/std": 0.05148233473300934, "step": 674, "step_time": 48.8107139095664 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 120.345703125, "completions/mean_terminated_length": 119.57925415039062, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.218550339108333, "epoch": 0.7696693272519954, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04425109177827835, "kl": 0.19336381438188255, "learning_rate": 7.731658366773717e-07, "loss": 0.0009669105638749897, "num_tokens": 110703023.0, "reward": 2.2113282680511475, "reward_std": 0.48738348484039307, "rewards/code_complexity_reward/mean": 0.8883788585662842, "rewards/code_complexity_reward/std": 0.1453341841697693, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 675, "step_time": 50.855736701749265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 116.611328125, "completions/mean_terminated_length": 115.060791015625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21226555947214365, "epoch": 0.7708095781071835, "frac_reward_zero_std": 0.421875, "grad_norm": 0.044480692595243454, "kl": 0.19121135224122554, "learning_rate": 7.659814623097958e-07, "loss": 0.0009561283513903618, "num_tokens": 110830860.0, "reward": 2.325927734375, "reward_std": 0.5383737087249756, "rewards/code_complexity_reward/mean": 0.9011718034744263, "rewards/code_complexity_reward/std": 0.14299902319908142, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.025244520977139473, "step": 676, "step_time": 61.83711375948042 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 501.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 116.896484375, "completions/mean_terminated_length": 116.896484375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21588699286803603, "epoch": 0.7719498289623717, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.05767342448234558, "kl": 0.19147798069752753, "learning_rate": 7.588245794716315e-07, "loss": 0.0009573273127898574, "num_tokens": 110958215.0, "reward": 2.2772462368011475, "reward_std": 0.5498886704444885, "rewards/code_complexity_reward/mean": 0.8883788585662842, "rewards/code_complexity_reward/std": 0.15985777974128723, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 677, "step_time": 59.77116129826754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 117.130859375, "completions/mean_terminated_length": 115.58235931396484, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.21739960624836385, "epoch": 0.7730900798175598, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04259096831083298, "kl": 0.20884517743252218, "learning_rate": 7.516953016297479e-07, "loss": 0.0010437463643029332, "num_tokens": 111087690.0, "reward": 2.28466796875, "reward_std": 0.5820784568786621, "rewards/code_complexity_reward/mean": 0.884570300579071, "rewards/code_complexity_reward/std": 0.1789735108613968, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49365234375, "rewards/xmlcount_reward_func/std": 0.04515400901436806, "step": 678, "step_time": 51.07103706058115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 120.953125, "completions/mean_terminated_length": 120.1878662109375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.22262453008443117, "epoch": 0.774230330672748, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.03965511918067932, "kl": 0.18509554234333336, "learning_rate": 7.445937418133564e-07, "loss": 0.0009253112366423011, "num_tokens": 111219538.0, "reward": 2.246337890625, "reward_std": 0.5128915905952454, "rewards/code_complexity_reward/mean": 0.8935546875, "rewards/code_complexity_reward/std": 0.14540401101112366, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.042219679802656174, "step": 679, "step_time": 70.2373418789357 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 122.259765625, "completions/mean_terminated_length": 121.49706268310547, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21932453685440123, "epoch": 0.7753705815279361, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.046155232936143875, "kl": 0.1819938935805112, "learning_rate": 7.375200126122256e-07, "loss": 0.0009100485476665199, "num_tokens": 111351195.0, "reward": 2.253955125808716, "reward_std": 0.5253973007202148, "rewards/code_complexity_reward/mean": 0.8896484375, "rewards/code_complexity_reward/std": 0.15786699950695038, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03521692752838135, "step": 680, "step_time": 58.14070367347449 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 118.130859375, "completions/mean_terminated_length": 116.5862808227539, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2065985668450594, "epoch": 0.7765108323831242, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.039603400975465775, "kl": 0.19663848693016917, "learning_rate": 7.304742261748848e-07, "loss": 0.0009832798969000578, "num_tokens": 111477774.0, "reward": 2.351123332977295, "reward_std": 0.5791255831718445, "rewards/code_complexity_reward/mean": 0.8829101324081421, "rewards/code_complexity_reward/std": 0.1714332550764084, "rewards/code_execution_reward/mean": 0.384765625, "rewards/code_execution_reward/std": 0.4870156943798065, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.027517380192875862, "step": 681, "step_time": 60.47953243739903 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 373.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 117.49609375, "completions/mean_terminated_length": 117.49609375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.21551625081337988, "epoch": 0.7776510832383124, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.04294244199991226, "kl": 0.21217594807967544, "learning_rate": 7.23456494206859e-07, "loss": 0.001060741487890482, "num_tokens": 111606448.0, "reward": 2.2550294399261475, "reward_std": 0.5224021077156067, "rewards/code_complexity_reward/mean": 0.8951171636581421, "rewards/code_complexity_reward/std": 0.15036024153232574, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.03255936875939369, "step": 682, "step_time": 76.93890109378844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 479.0, "completions/max_terminated_length": 479.0, "completions/mean_length": 119.45703125, "completions/mean_terminated_length": 119.45703125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21605279599316418, "epoch": 0.7787913340935005, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.041950523853302, "kl": 0.1964288033777848, "learning_rate": 7.164669279688846e-07, "loss": 0.0009820954874157906, "num_tokens": 111735666.0, "reward": 2.227978467941284, "reward_std": 0.5819758772850037, "rewards/code_complexity_reward/mean": 0.870800793170929, "rewards/code_complexity_reward/std": 0.20419269800186157, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.03438635915517807, "step": 683, "step_time": 49.55642463453114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 417.0, "completions/mean_length": 120.408203125, "completions/mean_terminated_length": 118.87255859375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2136964334640652, "epoch": 0.7799315849486887, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04091335088014603, "kl": 0.1979217806365341, "learning_rate": 7.095056382751559e-07, "loss": 0.0009894839022308588, "num_tokens": 111865643.0, "reward": 2.270751953125, "reward_std": 0.5321354269981384, "rewards/code_complexity_reward/mean": 0.8916991949081421, "rewards/code_complexity_reward/std": 0.156947523355484, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 684, "step_time": 51.28299034200609 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 123.697265625, "completions/mean_terminated_length": 120.63976287841797, "completions/min_length": 9.0, "completions/min_terminated_length": 9.0, "entropy": 0.2251453474164009, "epoch": 0.7810718358038768, "frac_reward_zero_std": 0.5, "grad_norm": 0.06864529103040695, "kl": 0.18597204226534814, "learning_rate": 7.025727354915655e-07, "loss": 0.0009298598160967231, "num_tokens": 111998984.0, "reward": 2.2879395484924316, "reward_std": 0.5664544701576233, "rewards/code_complexity_reward/mean": 0.8780273199081421, "rewards/code_complexity_reward/std": 0.18357349932193756, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.028431102633476257, "step": 685, "step_time": 64.45653882157058 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01171875, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 120.794921875, "completions/mean_terminated_length": 116.1561279296875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.20874618156813085, "epoch": 0.7822120866590649, "frac_reward_zero_std": 0.421875, "grad_norm": 0.03827380761504173, "kl": 0.2024929482722655, "learning_rate": 6.956683295339483e-07, "loss": 0.0010123095707967877, "num_tokens": 112128087.0, "reward": 2.28662109375, "reward_std": 0.5761479139328003, "rewards/code_complexity_reward/mean": 0.88134765625, "rewards/code_complexity_reward/std": 0.17365288734436035, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.03879402577877045, "step": 686, "step_time": 69.05544034112245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 416.0, "completions/max_terminated_length": 416.0, "completions/mean_length": 115.759765625, "completions/mean_terminated_length": 115.759765625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21055837837047875, "epoch": 0.7833523375142531, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.03683967515826225, "kl": 0.19611000025179237, "learning_rate": 6.887925298663506e-07, "loss": 0.0009804443689063191, "num_tokens": 112254684.0, "reward": 2.320068359375, "reward_std": 0.5162277221679688, "rewards/code_complexity_reward/mean": 0.90625, "rewards/code_complexity_reward/std": 0.11649646610021591, "rewards/code_execution_reward/mean": 0.32421875, "rewards/code_execution_reward/std": 0.4685399830341339, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03686080500483513, "step": 687, "step_time": 52.47664982825518 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 443.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 120.361328125, "completions/mean_terminated_length": 120.361328125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.20670745661482215, "epoch": 0.7844925883694412, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.048011600971221924, "kl": 0.17366108566056937, "learning_rate": 6.81945445499281e-07, "loss": 0.000868182280100882, "num_tokens": 112386565.0, "reward": 2.301074266433716, "reward_std": 0.546252429485321, "rewards/code_complexity_reward/mean": 0.8877929449081421, "rewards/code_complexity_reward/std": 0.1588313728570938, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02059757336974144, "step": 688, "step_time": 45.04976528324187 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 116.9765625, "completions/mean_terminated_length": 116.20352172851562, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21046009636484087, "epoch": 0.7856328392246295, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.045193739235401154, "kl": 0.19438843405805528, "learning_rate": 6.751271849879959e-07, "loss": 0.0009720007656142116, "num_tokens": 112515849.0, "reward": 2.3070313930511475, "reward_std": 0.5158916711807251, "rewards/code_complexity_reward/mean": 0.9073241949081421, "rewards/code_complexity_reward/std": 0.12748533487319946, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.04114582762122154, "step": 689, "step_time": 56.452981003560126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 453.0, "completions/mean_length": 115.455078125, "completions/mean_terminated_length": 114.67906188964844, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.22023326298221946, "epoch": 0.7867730900798175, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04605240747332573, "kl": 0.19960611942224205, "learning_rate": 6.68337856430766e-07, "loss": 0.0009980890899896622, "num_tokens": 112642710.0, "reward": 2.259033203125, "reward_std": 0.5241316556930542, "rewards/code_complexity_reward/mean": 0.8946288824081421, "rewards/code_complexity_reward/std": 0.15213888883590698, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 690, "step_time": 57.61370627209544 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 111.326171875, "completions/mean_terminated_length": 111.326171875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.20916478266008198, "epoch": 0.7879133409350056, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.0423860065639019, "kl": 0.18877590540796518, "learning_rate": 6.615775674671706e-07, "loss": 0.000943889026530087, "num_tokens": 112767741.0, "reward": 2.292041301727295, "reward_std": 0.5673210024833679, "rewards/code_complexity_reward/mean": 0.89208984375, "rewards/code_complexity_reward/std": 0.17472630739212036, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.03438635915517807, "step": 691, "step_time": 43.61197255551815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 351.0, "completions/max_terminated_length": 351.0, "completions/mean_length": 111.490234375, "completions/mean_terminated_length": 111.490234375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21623058058321476, "epoch": 0.7890535917901939, "frac_reward_zero_std": 0.53125, "grad_norm": 0.043169088661670685, "kl": 0.18552083999384195, "learning_rate": 6.548464252763906e-07, "loss": 0.0009274498443119228, "num_tokens": 112894668.0, "reward": 2.296630859375, "reward_std": 0.5012456178665161, "rewards/code_complexity_reward/mean": 0.9053710699081421, "rewards/code_complexity_reward/std": 0.11979050189256668, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 692, "step_time": 47.71586604882032 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 114.21484375, "completions/mean_terminated_length": 113.4364013671875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22171921376138926, "epoch": 0.790193842645382, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.03914031758904457, "kl": 0.18858751805964857, "learning_rate": 6.481445365755021e-07, "loss": 0.0009428664343431592, "num_tokens": 113020394.0, "reward": 2.2770509719848633, "reward_std": 0.5229920744895935, "rewards/code_complexity_reward/mean": 0.9030272960662842, "rewards/code_complexity_reward/std": 0.14653737843036652, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.04044608399271965, "step": 693, "step_time": 48.50120793376118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 116.638671875, "completions/mean_terminated_length": 116.638671875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21936990902759135, "epoch": 0.7913340935005702, "frac_reward_zero_std": 0.390625, "grad_norm": 0.04422120004892349, "kl": 0.18685835134238005, "learning_rate": 6.414720076177958e-07, "loss": 0.0009341153199784458, "num_tokens": 113148817.0, "reward": 2.260742425918579, "reward_std": 0.5458205938339233, "rewards/code_complexity_reward/mean": 0.8936523199081421, "rewards/code_complexity_reward/std": 0.16979248821735382, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.020545316860079765, "step": 694, "step_time": 46.7420039344579 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 434.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 120.208984375, "completions/mean_terminated_length": 120.208984375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20886015240103006, "epoch": 0.7924743443557583, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04130272939801216, "kl": 0.19739727722480893, "learning_rate": 6.348289441910807e-07, "loss": 0.0009867995977401733, "num_tokens": 113278892.0, "reward": 2.2386231422424316, "reward_std": 0.5331581234931946, "rewards/code_complexity_reward/mean": 0.88671875, "rewards/code_complexity_reward/std": 0.1659621000289917, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.04358936473727226, "step": 695, "step_time": 51.78097239136696 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 468.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 123.08984375, "completions/mean_terminated_length": 123.08984375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2215735036879778, "epoch": 0.7936145952109465, "frac_reward_zero_std": 0.3984375, "grad_norm": 0.052773527801036835, "kl": 0.20735999487806112, "learning_rate": 6.282154516160157e-07, "loss": 0.0010369222145527601, "num_tokens": 113410602.0, "reward": 2.246875047683716, "reward_std": 0.5769271850585938, "rewards/code_complexity_reward/mean": 0.87646484375, "rewards/code_complexity_reward/std": 0.19494040310382843, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.030093414708971977, "step": 696, "step_time": 58.04324149619788 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 118.1171875, "completions/mean_terminated_length": 117.34638214111328, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21720154560171068, "epoch": 0.7947548460661346, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.04377533867955208, "kl": 0.18270824733190238, "learning_rate": 6.216316347444362e-07, "loss": 0.0009134472929872572, "num_tokens": 113539630.0, "reward": 2.1996097564697266, "reward_std": 0.467427134513855, "rewards/code_complexity_reward/mean": 0.8984375, "rewards/code_complexity_reward/std": 0.1351504921913147, "rewards/code_execution_reward/mean": 0.2109375, "rewards/code_execution_reward/std": 0.4083731174468994, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.019055325537919998, "step": 697, "step_time": 61.91372948139906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 445.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 118.578125, "completions/mean_terminated_length": 118.578125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2142471878323704, "epoch": 0.7958950969213227, "frac_reward_zero_std": 0.46875, "grad_norm": 0.0428793765604496, "kl": 0.184582578134723, "learning_rate": 6.150775979576906e-07, "loss": 0.0009227320551872253, "num_tokens": 113671186.0, "reward": 2.324267625808716, "reward_std": 0.5455124974250793, "rewards/code_complexity_reward/mean": 0.8965820074081421, "rewards/code_complexity_reward/std": 0.14868317544460297, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 698, "step_time": 47.27435071673244 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 467.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 119.1953125, "completions/mean_terminated_length": 119.1953125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21849942859262228, "epoch": 0.7970353477765109, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05083238705992699, "kl": 0.1978339608758688, "learning_rate": 6.085534451649905e-07, "loss": 0.0009892748203128576, "num_tokens": 113800174.0, "reward": 2.2411623001098633, "reward_std": 0.5210690498352051, "rewards/code_complexity_reward/mean": 0.891406238079071, "rewards/code_complexity_reward/std": 0.16219015419483185, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.016543962061405182, "step": 699, "step_time": 54.098046951927245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 487.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 113.484375, "completions/mean_terminated_length": 113.484375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.20552135934121907, "epoch": 0.798175598631699, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.03818579763174057, "kl": 0.19743484782520682, "learning_rate": 6.020592798017554e-07, "loss": 0.000986973405815661, "num_tokens": 113926514.0, "reward": 2.2655763626098633, "reward_std": 0.5441292524337769, "rewards/code_complexity_reward/mean": 0.8910156488418579, "rewards/code_complexity_reward/std": 0.16518785059452057, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.0376812107861042, "step": 700, "step_time": 58.41267904546112 }, { "epoch": 0.798175598631699, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0025, "eval_completions/max_length": 203.78, "eval_completions/max_terminated_length": 201.16, "eval_completions/mean_length": 120.69, "eval_completions/mean_terminated_length": 119.86678588867187, "eval_completions/min_length": 72.5, "eval_completions/min_terminated_length": 72.5, "eval_entropy": 0.22333887755870818, "eval_frac_reward_zero_std": 0.47, "eval_kl": 0.1889938907325268, "eval_loss": 0.000942988961469382, "eval_num_tokens": 113926514.0, "eval_reward": 2.2364376044273375, "eval_reward_std": 0.3691736926138401, "eval_rewards/code_complexity_reward/mean": 0.8962499809265136, "eval_rewards/code_complexity_reward/std": 0.08837413407862187, "eval_rewards/code_execution_reward/mean": 0.255, "eval_rewards/code_execution_reward/std": 0.28970741987228393, "eval_rewards/code_syntax_reward/mean": 0.49, "eval_rewards/code_syntax_reward/std": 0.028284270763397217, "eval_rewards/reasoning_present_reward_func/mean": 0.09925000175833702, "eval_rewards/reasoning_present_reward_func/std": 0.002121320441365242, "eval_rewards/xmlcount_reward_func/mean": 0.4959375, "eval_rewards/xmlcount_reward_func/std": 0.011490484923124314, "eval_runtime": 440.5846, "eval_samples_per_second": 0.227, "eval_steps_per_second": 0.03, "step": 700 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 123.1875, "completions/mean_terminated_length": 120.89588165283203, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2130933734588325, "epoch": 0.7993158494868872, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04341607540845871, "kl": 0.19713449431583285, "learning_rate": 5.955952048279795e-07, "loss": 0.0009857559343799949, "num_tokens": 114060710.0, "reward": 2.28369140625, "reward_std": 0.5612989068031311, "rewards/code_complexity_reward/mean": 0.880664050579071, "rewards/code_complexity_reward/std": 0.18131166696548462, "rewards/code_execution_reward/mean": 0.322265625, "rewards/code_execution_reward/std": 0.46780112385749817, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.036414988338947296, "step": 701, "step_time": 56.52325738128275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 120.35546875, "completions/mean_terminated_length": 120.35546875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2244148044846952, "epoch": 0.8004561003420753, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04431646317243576, "kl": 0.19246847135946155, "learning_rate": 5.891613227265971e-07, "loss": 0.0009623367222957313, "num_tokens": 114190716.0, "reward": 2.290820598602295, "reward_std": 0.49890899658203125, "rewards/code_complexity_reward/mean": 0.9056640267372131, "rewards/code_complexity_reward/std": 0.12452314794063568, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 702, "step_time": 55.5694587379694 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 394.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 114.224609375, "completions/mean_terminated_length": 114.224609375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21350903389975429, "epoch": 0.8015963511972634, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04155283421278, "kl": 0.20246175001375377, "learning_rate": 5.827577355018577e-07, "loss": 0.0010123989777639508, "num_tokens": 114316595.0, "reward": 2.2597169876098633, "reward_std": 0.5188657641410828, "rewards/code_complexity_reward/mean": 0.8966796398162842, "rewards/code_complexity_reward/std": 0.14810845255851746, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.035264380276203156, "step": 703, "step_time": 42.41332990769297 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 442.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 118.353515625, "completions/mean_terminated_length": 118.353515625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21503327786922455, "epoch": 0.8027366020524516, "frac_reward_zero_std": 0.390625, "grad_norm": 0.04606826603412628, "kl": 0.20100850239396095, "learning_rate": 5.763845446777077e-07, "loss": 0.0010050202254205942, "num_tokens": 114444632.0, "reward": 2.19970703125, "reward_std": 0.5574284195899963, "rewards/code_complexity_reward/mean": 0.8720703125, "rewards/code_complexity_reward/std": 0.19333617389202118, "rewards/code_execution_reward/mean": 0.251953125, "rewards/code_execution_reward/std": 0.43455907702445984, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03567269444465637, "step": 704, "step_time": 45.32332541234791 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 481.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 118.107421875, "completions/mean_terminated_length": 118.107421875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21875869412906468, "epoch": 0.8038768529076397, "frac_reward_zero_std": 0.3828125, "grad_norm": 0.046464379876852036, "kl": 0.18896957067772746, "learning_rate": 5.700418512961825e-07, "loss": 0.0009447732591070235, "num_tokens": 114572187.0, "reward": 2.2943360805511475, "reward_std": 0.5233113765716553, "rewards/code_complexity_reward/mean": 0.89501953125, "rewards/code_complexity_reward/std": 0.14207574725151062, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 705, "step_time": 51.969965633936226 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 336.0, "completions/mean_length": 114.748046875, "completions/mean_terminated_length": 113.97064208984375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21456078020855784, "epoch": 0.8050171037628279, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04875292256474495, "kl": 0.19901781785301864, "learning_rate": 5.637297559158067e-07, "loss": 0.0009952562395483255, "num_tokens": 114698550.0, "reward": 2.321045160293579, "reward_std": 0.5510266423225403, "rewards/code_complexity_reward/mean": 0.896777331829071, "rewards/code_complexity_reward/std": 0.15071377158164978, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03250797092914581, "step": 706, "step_time": 50.225641324184835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 507.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 121.046875, "completions/mean_terminated_length": 121.046875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2274817847646773, "epoch": 0.806157354618016, "frac_reward_zero_std": 0.5, "grad_norm": 0.05080932378768921, "kl": 0.19203021982684731, "learning_rate": 5.574483586099924e-07, "loss": 0.0009599705226719379, "num_tokens": 114829658.0, "reward": 2.3014163970947266, "reward_std": 0.5283468961715698, "rewards/code_complexity_reward/mean": 0.901074230670929, "rewards/code_complexity_reward/std": 0.1382509469985962, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.030623575672507286, "step": 707, "step_time": 58.354377215728164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 113.853515625, "completions/mean_terminated_length": 113.853515625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21573433163575828, "epoch": 0.8072976054732041, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.0493302047252655, "kl": 0.19247134239412844, "learning_rate": 5.511977589654601e-07, "loss": 0.0009624327067285776, "num_tokens": 114956311.0, "reward": 2.214404344558716, "reward_std": 0.5212920904159546, "rewards/code_complexity_reward/mean": 0.8875976204872131, "rewards/code_complexity_reward/std": 0.1723484843969345, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01650058664381504, "step": 708, "step_time": 53.14596063364297 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 497.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 112.904296875, "completions/mean_terminated_length": 112.904296875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2184273435268551, "epoch": 0.8084378563283923, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.04053648188710213, "kl": 0.18321789614856243, "learning_rate": 5.449780560806572e-07, "loss": 0.0009160230401903391, "num_tokens": 115082506.0, "reward": 2.346972703933716, "reward_std": 0.5653983354568481, "rewards/code_complexity_reward/mean": 0.8943359851837158, "rewards/code_complexity_reward/std": 0.15933799743652344, "rewards/code_execution_reward/mean": 0.3671875, "rewards/code_execution_reward/std": 0.48250964283943176, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 709, "step_time": 57.8986167954281 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 462.0, "completions/mean_length": 119.3203125, "completions/mean_terminated_length": 118.5518569946289, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22442189790308475, "epoch": 0.8095781071835804, "frac_reward_zero_std": 0.46875, "grad_norm": 0.041805144399404526, "kl": 0.20034891332034022, "learning_rate": 5.387893485641862e-07, "loss": 0.001001827884465456, "num_tokens": 115209558.0, "reward": 2.3099610805511475, "reward_std": 0.5409624576568604, "rewards/code_complexity_reward/mean": 0.89306640625, "rewards/code_complexity_reward/std": 0.15271762013435364, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.021983280777931213, "step": 710, "step_time": 58.63638108596206 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 431.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 120.908203125, "completions/mean_terminated_length": 120.908203125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2197300896514207, "epoch": 0.8107183580387686, "frac_reward_zero_std": 0.4375, "grad_norm": 0.041274044662714005, "kl": 0.20052504551131278, "learning_rate": 5.326317345332416e-07, "loss": 0.0010025326628237963, "num_tokens": 115337583.0, "reward": 2.2450196743011475, "reward_std": 0.5273601412773132, "rewards/code_complexity_reward/mean": 0.8924804329872131, "rewards/code_complexity_reward/std": 0.15748481452465057, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.04044608399271965, "step": 711, "step_time": 51.85751205217093 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 424.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 109.50390625, "completions/mean_terminated_length": 109.50390625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2147213325370103, "epoch": 0.8118586088939567, "frac_reward_zero_std": 0.5, "grad_norm": 0.0420551598072052, "kl": 0.20080209127627313, "learning_rate": 5.26505311612055e-07, "loss": 0.001003804849460721, "num_tokens": 115461865.0, "reward": 2.3021485805511475, "reward_std": 0.5369017720222473, "rewards/code_complexity_reward/mean": 0.9025390148162842, "rewards/code_complexity_reward/std": 0.15071064233779907, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 712, "step_time": 50.68605125788599 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 481.0, "completions/mean_length": 119.74609375, "completions/mean_terminated_length": 118.97846984863281, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.2214569270145148, "epoch": 0.8129988597491448, "frac_reward_zero_std": 0.40625, "grad_norm": 0.0451953262090683, "kl": 0.18874667095951736, "learning_rate": 5.204101769303474e-07, "loss": 0.0009436379768885672, "num_tokens": 115590175.0, "reward": 2.280224561691284, "reward_std": 0.5380401015281677, "rewards/code_complexity_reward/mean": 0.89697265625, "rewards/code_complexity_reward/std": 0.16019073128700256, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.027517380192875862, "step": 713, "step_time": 49.235590517520905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 109.6328125, "completions/mean_terminated_length": 109.6328125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21776370122097433, "epoch": 0.814139110604333, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.04023529589176178, "kl": 0.17865597119089216, "learning_rate": 5.143464271217876e-07, "loss": 0.0008934216457419097, "num_tokens": 115712455.0, "reward": 2.3416996002197266, "reward_std": 0.5074609518051147, "rewards/code_complexity_reward/mean": 0.91796875, "rewards/code_complexity_reward/std": 0.10468186438083649, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 714, "step_time": 44.72147117275745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 430.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 112.998046875, "completions/mean_terminated_length": 112.998046875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.21311058150604367, "epoch": 0.8152793614595211, "frac_reward_zero_std": 0.515625, "grad_norm": 0.03972504660487175, "kl": 0.20167189557105303, "learning_rate": 5.083141583224627e-07, "loss": 0.0010083435336127877, "num_tokens": 115837422.0, "reward": 2.304492473602295, "reward_std": 0.5201745629310608, "rewards/code_complexity_reward/mean": 0.9048827886581421, "rewards/code_complexity_reward/std": 0.1302398145198822, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 715, "step_time": 51.07515705190599 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 422.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 123.3828125, "completions/mean_terminated_length": 123.3828125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.22012234292924404, "epoch": 0.8164196123147093, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.044310323894023895, "kl": 0.21478101378306746, "learning_rate": 5.023134661693518e-07, "loss": 0.0010737497359514236, "num_tokens": 115972570.0, "reward": 2.198242425918579, "reward_std": 0.5356441736221313, "rewards/code_complexity_reward/mean": 0.8773437738418579, "rewards/code_complexity_reward/std": 0.18578225374221802, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.04666558653116226, "step": 716, "step_time": 63.46185419615358 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 383.0, "completions/mean_length": 117.23046875, "completions/mean_terminated_length": 116.45792388916016, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.21272786846384406, "epoch": 0.8175598631698974, "frac_reward_zero_std": 0.53125, "grad_norm": 0.04007934778928757, "kl": 0.21317939017899334, "learning_rate": 4.963444457988109e-07, "loss": 0.0010660444386303425, "num_tokens": 116101148.0, "reward": 2.268798828125, "reward_std": 0.5163111090660095, "rewards/code_complexity_reward/mean": 0.902148425579071, "rewards/code_complexity_reward/std": 0.14563201367855072, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 717, "step_time": 61.43659026827663 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 386.0, "completions/max_terminated_length": 386.0, "completions/mean_length": 112.779296875, "completions/mean_terminated_length": 112.779296875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21675262972712517, "epoch": 0.8187001140250855, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.05279812961816788, "kl": 0.19697551592253149, "learning_rate": 4.904071918450643e-07, "loss": 0.0009848900372162461, "num_tokens": 116227219.0, "reward": 2.3182129859924316, "reward_std": 0.5499280095100403, "rewards/code_complexity_reward/mean": 0.898632824420929, "rewards/code_complexity_reward/std": 0.15441349148750305, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 718, "step_time": 50.43138645682484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 458.0, "completions/mean_length": 122.166015625, "completions/mean_terminated_length": 121.40312957763672, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.20559568237513304, "epoch": 0.8198403648802737, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.04096808284521103, "kl": 0.18822891684249043, "learning_rate": 4.84501798438704e-07, "loss": 0.0009411206701770425, "num_tokens": 116358068.0, "reward": 2.286914348602295, "reward_std": 0.5223873853683472, "rewards/code_complexity_reward/mean": 0.890429675579071, "rewards/code_complexity_reward/std": 0.14026187360286713, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.033960822969675064, "step": 719, "step_time": 57.875511947087944 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 431.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 117.015625, "completions/mean_terminated_length": 117.015625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2133118100464344, "epoch": 0.8209806157354618, "frac_reward_zero_std": 0.40625, "grad_norm": 0.04506140202283859, "kl": 0.17964422737713903, "learning_rate": 4.78628359205198e-07, "loss": 0.0008981946157291532, "num_tokens": 116489152.0, "reward": 2.225342035293579, "reward_std": 0.529597282409668, "rewards/code_complexity_reward/mean": 0.8892577886581421, "rewards/code_complexity_reward/std": 0.16638152301311493, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.033241886645555496, "step": 720, "step_time": 53.189533228985965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 461.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 114.33984375, "completions/mean_terminated_length": 114.33984375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21275903680361807, "epoch": 0.82212086659065, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04365622624754906, "kl": 0.19464367837645113, "learning_rate": 4.727869672634044e-07, "loss": 0.0009733362239785492, "num_tokens": 116615554.0, "reward": 2.2665038108825684, "reward_std": 0.5380380153656006, "rewards/code_complexity_reward/mean": 0.8908202648162842, "rewards/code_complexity_reward/std": 0.1611289381980896, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.023414555937051773, "step": 721, "step_time": 45.68447150848806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 428.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 116.0546875, "completions/mean_terminated_length": 116.0546875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.21945184911601245, "epoch": 0.8232611174458381, "frac_reward_zero_std": 0.515625, "grad_norm": 0.044199805706739426, "kl": 0.18112984870094806, "learning_rate": 4.669777152240976e-07, "loss": 0.0009055115515366197, "num_tokens": 116743458.0, "reward": 2.255127191543579, "reward_std": 0.5285969376564026, "rewards/code_complexity_reward/mean": 0.8943359851837158, "rewards/code_complexity_reward/std": 0.16083548963069916, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 722, "step_time": 46.50669476110488 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 117.267578125, "completions/mean_terminated_length": 116.49510955810547, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21776136104017496, "epoch": 0.8244013683010262, "frac_reward_zero_std": 0.484375, "grad_norm": 0.045164529234170914, "kl": 0.1885308501077816, "learning_rate": 4.612006951884973e-07, "loss": 0.0009425851167179644, "num_tokens": 116873039.0, "reward": 2.28564453125, "reward_std": 0.5097618103027344, "rewards/code_complexity_reward/mean": 0.8997069597244263, "rewards/code_complexity_reward/std": 0.12673419713974, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02465205453336239, "step": 723, "step_time": 52.09449317958206 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 451.0, "completions/mean_length": 116.80078125, "completions/mean_terminated_length": 116.02739715576172, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22457348322495818, "epoch": 0.8255416191562144, "frac_reward_zero_std": 0.484375, "grad_norm": 0.038766104727983475, "kl": 0.20024896075483412, "learning_rate": 4.5545599874681076e-07, "loss": 0.0010009363759309053, "num_tokens": 117001497.0, "reward": 2.3131346702575684, "reward_std": 0.5060563087463379, "rewards/code_complexity_reward/mean": 0.90771484375, "rewards/code_complexity_reward/std": 0.11537420004606247, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.026328405365347862, "step": 724, "step_time": 58.29461889527738 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 118.310546875, "completions/mean_terminated_length": 118.310546875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2174702121410519, "epoch": 0.8266818700114025, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.042613424360752106, "kl": 0.21500171686057, "learning_rate": 4.4974371697677847e-07, "loss": 0.0010748808272182941, "num_tokens": 117128692.0, "reward": 2.2718262672424316, "reward_std": 0.5266907215118408, "rewards/code_complexity_reward/mean": 0.8995116949081421, "rewards/code_complexity_reward/std": 0.15248994529247284, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 725, "step_time": 47.04365489911288 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 351.0, "completions/mean_length": 114.673828125, "completions/mean_terminated_length": 113.89627838134766, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2168662187177688, "epoch": 0.8278221208665907, "frac_reward_zero_std": 0.421875, "grad_norm": 0.04367070272564888, "kl": 0.20404913881793618, "learning_rate": 4.4406394044223174e-07, "loss": 0.001020264346152544, "num_tokens": 117255769.0, "reward": 2.2670412063598633, "reward_std": 0.5360409617424011, "rewards/code_complexity_reward/mean": 0.8949218988418579, "rewards/code_complexity_reward/std": 0.1535732001066208, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660499989986, "rewards/xmlcount_reward_func/mean": 0.494384765625, "rewards/xmlcount_reward_func/std": 0.04625359922647476, "step": 726, "step_time": 61.1579035371542 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 118.25390625, "completions/mean_terminated_length": 117.48336791992188, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21964135207235813, "epoch": 0.8289623717217788, "frac_reward_zero_std": 0.40625, "grad_norm": 0.047192733734846115, "kl": 0.18475011980626732, "learning_rate": 4.384167591916566e-07, "loss": 0.0009235728066414595, "num_tokens": 117382947.0, "reward": 2.3067383766174316, "reward_std": 0.5485758781433105, "rewards/code_complexity_reward/mean": 0.890332043170929, "rewards/code_complexity_reward/std": 0.1539413183927536, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03562243655323982, "step": 727, "step_time": 61.01566889323294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 387.0, "completions/max_terminated_length": 387.0, "completions/mean_length": 117.23046875, "completions/mean_terminated_length": 117.23046875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21486163744702935, "epoch": 0.830102622576967, "frac_reward_zero_std": 0.453125, "grad_norm": 0.03865674138069153, "kl": 0.18710880761500448, "learning_rate": 4.328022627567657e-07, "loss": 0.0009354889625683427, "num_tokens": 117510941.0, "reward": 2.2499022483825684, "reward_std": 0.5249592661857605, "rewards/code_complexity_reward/mean": 0.88818359375, "rewards/code_complexity_reward/std": 0.15889176726341248, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.03304819017648697, "step": 728, "step_time": 48.438202461227775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 115.43359375, "completions/mean_terminated_length": 114.65753173828125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.20705574587918818, "epoch": 0.8312428734321551, "frac_reward_zero_std": 0.46875, "grad_norm": 0.042393915355205536, "kl": 0.20936957478988916, "learning_rate": 4.2722054015107874e-07, "loss": 0.0010469621047377586, "num_tokens": 117638191.0, "reward": 2.2745118141174316, "reward_std": 0.5815302133560181, "rewards/code_complexity_reward/mean": 0.8772460222244263, "rewards/code_complexity_reward/std": 0.18628311157226562, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.04044608399271965, "step": 729, "step_time": 67.70748670212924 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 459.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 118.8984375, "completions/mean_terminated_length": 118.8984375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21523063606582582, "epoch": 0.8323831242873432, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04311360791325569, "kl": 0.20893269404768944, "learning_rate": 4.216716798685125e-07, "loss": 0.0010446406668052077, "num_tokens": 117766503.0, "reward": 2.3099610805511475, "reward_std": 0.5474061965942383, "rewards/code_complexity_reward/mean": 0.889843761920929, "rewards/code_complexity_reward/std": 0.15760056674480438, "rewards/code_execution_reward/mean": 0.333984375, "rewards/code_execution_reward/std": 0.47209542989730835, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.025862684473395348, "step": 730, "step_time": 54.948318992741406 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 117.64453125, "completions/mean_terminated_length": 116.87279510498047, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21036746306344867, "epoch": 0.8335233751425314, "frac_reward_zero_std": 0.46875, "grad_norm": 0.047659892588853836, "kl": 0.1993050198070705, "learning_rate": 4.161557698819757e-07, "loss": 0.0009966512443497777, "num_tokens": 117894101.0, "reward": 2.32861328125, "reward_std": 0.5208017826080322, "rewards/code_complexity_reward/mean": 0.897753894329071, "rewards/code_complexity_reward/std": 0.13245326280593872, "rewards/code_execution_reward/mean": 0.341796875, "rewards/code_execution_reward/std": 0.4747757613658905, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 731, "step_time": 50.621001795865595 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 357.0, "completions/mean_length": 111.10546875, "completions/mean_terminated_length": 110.32093811035156, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.20992708508856595, "epoch": 0.8346636259977195, "frac_reward_zero_std": 0.5, "grad_norm": 0.043122127652168274, "kl": 0.19903102866373956, "learning_rate": 4.106728976419763e-07, "loss": 0.0009949463419616222, "num_tokens": 118016803.0, "reward": 2.328418254852295, "reward_std": 0.5354169011116028, "rewards/code_complexity_reward/mean": 0.903613269329071, "rewards/code_complexity_reward/std": 0.13033762574195862, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 732, "step_time": 48.37757377978414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 121.580078125, "completions/mean_terminated_length": 120.04902648925781, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.21967541240155697, "epoch": 0.8358038768529077, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04785420373082161, "kl": 0.18279821728356183, "learning_rate": 4.0522315007523486e-07, "loss": 0.0009139457833953202, "num_tokens": 118148716.0, "reward": 2.3136231899261475, "reward_std": 0.5753849148750305, "rewards/code_complexity_reward/mean": 0.8791015148162842, "rewards/code_complexity_reward/std": 0.17475947737693787, "rewards/code_execution_reward/mean": 0.353515625, "rewards/code_execution_reward/std": 0.47852855920791626, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02746524289250374, "step": 733, "step_time": 57.15299557801336 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 121.419921875, "completions/mean_terminated_length": 119.11788177490234, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2176288808695972, "epoch": 0.8369441277080958, "frac_reward_zero_std": 0.421875, "grad_norm": 0.05229008197784424, "kl": 0.190035380423069, "learning_rate": 3.998066135833031e-07, "loss": 0.0009498444269411266, "num_tokens": 118278583.0, "reward": 2.2723145484924316, "reward_std": 0.5175599455833435, "rewards/code_complexity_reward/mean": 0.8963867425918579, "rewards/code_complexity_reward/std": 0.1419793963432312, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.025244520977139473, "step": 734, "step_time": 59.13801111187786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 429.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 114.9296875, "completions/mean_terminated_length": 114.9296875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21550137316808105, "epoch": 0.8380843785632839, "frac_reward_zero_std": 0.484375, "grad_norm": 0.040183305740356445, "kl": 0.2148742601275444, "learning_rate": 3.944233740412001e-07, "loss": 0.0010742744198068976, "num_tokens": 118405487.0, "reward": 2.2829103469848633, "reward_std": 0.5274077653884888, "rewards/code_complexity_reward/mean": 0.8973633050918579, "rewards/code_complexity_reward/std": 0.1412409245967865, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.04600567743182182, "step": 735, "step_time": 45.388730220496655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 112.619140625, "completions/mean_terminated_length": 112.619140625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22081576939672232, "epoch": 0.8392246294184721, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04541017860174179, "kl": 0.1966289789415896, "learning_rate": 3.890735167960455e-07, "loss": 0.0009833329822868109, "num_tokens": 118530352.0, "reward": 2.2576661109924316, "reward_std": 0.5027941465377808, "rewards/code_complexity_reward/mean": 0.9027343392372131, "rewards/code_complexity_reward/std": 0.14014369249343872, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 736, "step_time": 63.55258823465556 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 110.9765625, "completions/mean_terminated_length": 110.19178009033203, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2116806749254465, "epoch": 0.8403648802736602, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.0411132387816906, "kl": 0.1939416320528835, "learning_rate": 3.8375712666570865e-07, "loss": 0.0009696747874841094, "num_tokens": 118657436.0, "reward": 2.3638672828674316, "reward_std": 0.5470259785652161, "rewards/code_complexity_reward/mean": 0.9051758050918579, "rewards/code_complexity_reward/std": 0.13655520975589752, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03206123411655426, "step": 737, "step_time": 58.58079747390002 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 119.189453125, "completions/mean_terminated_length": 118.42074584960938, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.22060592635534704, "epoch": 0.8415051311288484, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.05250760540366173, "kl": 0.20992374001070857, "learning_rate": 3.784742879374631e-07, "loss": 0.0010494228918105364, "num_tokens": 118785945.0, "reward": 2.229248046875, "reward_std": 0.5408849716186523, "rewards/code_complexity_reward/mean": 0.8811523914337158, "rewards/code_complexity_reward/std": 0.17653779685497284, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.021246958523988724, "step": 738, "step_time": 51.18255949392915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 118.953125, "completions/mean_terminated_length": 118.18395233154297, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22550944704562426, "epoch": 0.8426453819840365, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04289526492357254, "kl": 0.20333593618124723, "learning_rate": 3.7322508436665184e-07, "loss": 0.0010168332373723388, "num_tokens": 118913701.0, "reward": 2.2375001907348633, "reward_std": 0.5374479293823242, "rewards/code_complexity_reward/mean": 0.885937511920929, "rewards/code_complexity_reward/std": 0.1675117164850235, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.042552899569272995, "step": 739, "step_time": 50.294265046715736 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 435.0, "completions/max_terminated_length": 435.0, "completions/mean_length": 119.423828125, "completions/mean_terminated_length": 119.423828125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.21648086817003787, "epoch": 0.8437856328392246, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04396629333496094, "kl": 0.19732447434216738, "learning_rate": 3.680095991753577e-07, "loss": 0.0009866695618256927, "num_tokens": 119042862.0, "reward": 2.24462890625, "reward_std": 0.5103795528411865, "rewards/code_complexity_reward/mean": 0.8935546278953552, "rewards/code_complexity_reward/std": 0.1411701887845993, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03567269444465637, "step": 740, "step_time": 50.100990042090416 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 115.109375, "completions/mean_terminated_length": 114.33267974853516, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21951419720426202, "epoch": 0.8449258836944128, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.04268374294042587, "kl": 0.20705637452192605, "learning_rate": 3.6282791505108327e-07, "loss": 0.001035273540765047, "num_tokens": 119169862.0, "reward": 2.230273485183716, "reward_std": 0.5378474593162537, "rewards/code_complexity_reward/mean": 0.8843749761581421, "rewards/code_complexity_reward/std": 0.1724974811077118, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03562243655323982, "step": 741, "step_time": 49.38975086901337 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 432.0, "completions/mean_length": 118.767578125, "completions/mean_terminated_length": 117.99803924560547, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21613432210870087, "epoch": 0.8460661345496009, "frac_reward_zero_std": 0.4375, "grad_norm": 0.041286639869213104, "kl": 0.18980358634144068, "learning_rate": 3.576801141454439e-07, "loss": 0.000948803499341011, "num_tokens": 119300047.0, "reward": 2.2538087368011475, "reward_std": 0.5703538060188293, "rewards/code_complexity_reward/mean": 0.8751952648162842, "rewards/code_complexity_reward/std": 0.18484658002853394, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03480498120188713, "step": 742, "step_time": 59.57568064145744 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 113.177734375, "completions/mean_terminated_length": 112.39726257324219, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.20954701234586537, "epoch": 0.8472063854047891, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04658324643969536, "kl": 0.19839194300584495, "learning_rate": 3.5256627807286086e-07, "loss": 0.0009920853190124035, "num_tokens": 119428454.0, "reward": 2.310839891433716, "reward_std": 0.5201066732406616, "rewards/code_complexity_reward/mean": 0.9041992425918579, "rewards/code_complexity_reward/std": 0.1282004863023758, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02059757336974144, "step": 743, "step_time": 59.04317466262728 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 485.0, "completions/mean_length": 115.548828125, "completions/mean_terminated_length": 114.77299499511719, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.20950994896702468, "epoch": 0.8483466362599772, "frac_reward_zero_std": 0.484375, "grad_norm": 0.044612567871809006, "kl": 0.19168867880944163, "learning_rate": 3.474864879092693e-07, "loss": 0.0009583963546901941, "num_tokens": 119554691.0, "reward": 2.308349609375, "reward_std": 0.6093670129776001, "rewards/code_complexity_reward/mean": 0.87744140625, "rewards/code_complexity_reward/std": 0.19475483894348145, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.493408203125, "rewards/xmlcount_reward_func/std": 0.04994375631213188, "step": 744, "step_time": 52.75476545561105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 406.0, "completions/mean_length": 109.587890625, "completions/mean_terminated_length": 108.8003921508789, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21337275905534625, "epoch": 0.8494868871151653, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.04333147034049034, "kl": 0.21043863613158464, "learning_rate": 3.424408241908336e-07, "loss": 0.0010522911325097084, "num_tokens": 119679824.0, "reward": 2.260058879852295, "reward_std": 0.48133090138435364, "rewards/code_complexity_reward/mean": 0.9071289300918579, "rewards/code_complexity_reward/std": 0.11973943561315536, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02465205453336239, "step": 745, "step_time": 69.26120687369257 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 466.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 117.580078125, "completions/mean_terminated_length": 117.580078125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21167009905911982, "epoch": 0.8506271379703535, "frac_reward_zero_std": 0.453125, "grad_norm": 0.045924168080091476, "kl": 0.1963560211006552, "learning_rate": 3.374293669126669e-07, "loss": 0.0009815674275159836, "num_tokens": 119807213.0, "reward": 2.2840332984924316, "reward_std": 0.5201353430747986, "rewards/code_complexity_reward/mean": 0.8941406011581421, "rewards/code_complexity_reward/std": 0.13903090357780457, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.024042511358857155, "step": 746, "step_time": 47.77646171953529 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 114.6015625, "completions/mean_terminated_length": 113.8238754272461, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2150196535512805, "epoch": 0.8517673888255416, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.049486804753541946, "kl": 0.20303237717598677, "learning_rate": 3.324521955275697e-07, "loss": 0.0010151129681617022, "num_tokens": 119933745.0, "reward": 2.288867235183716, "reward_std": 0.5423688292503357, "rewards/code_complexity_reward/mean": 0.8882812261581421, "rewards/code_complexity_reward/std": 0.15829749405384064, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.02327641472220421, "step": 747, "step_time": 58.02485660649836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 452.0, "completions/mean_length": 115.541015625, "completions/mean_terminated_length": 114.76516723632812, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.20232220296747983, "epoch": 0.8529076396807298, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.0410701259970665, "kl": 0.19232337619177997, "learning_rate": 3.2750938894476223e-07, "loss": 0.0009614718146622181, "num_tokens": 120061138.0, "reward": 2.296630859375, "reward_std": 0.5380821228027344, "rewards/code_complexity_reward/mean": 0.896777331829071, "rewards/code_complexity_reward/std": 0.14881911873817444, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 748, "step_time": 51.498969284817576 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 373.0, "completions/max_terminated_length": 373.0, "completions/mean_length": 115.544921875, "completions/mean_terminated_length": 115.544921875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21272826381027699, "epoch": 0.8540478905359179, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.0436876006424427, "kl": 0.23230692348442972, "learning_rate": 3.2260102552863994e-07, "loss": 0.0011617809068411589, "num_tokens": 120189565.0, "reward": 2.2592287063598633, "reward_std": 0.5262893438339233, "rewards/code_complexity_reward/mean": 0.8994140625, "rewards/code_complexity_reward/std": 0.15359248220920563, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03521692752838135, "step": 749, "step_time": 40.39211235847324 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 120.578125, "completions/mean_terminated_length": 119.04314422607422, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2109571814071387, "epoch": 0.855188141391106, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04506834223866463, "kl": 0.1793093680171296, "learning_rate": 3.177271830975276e-07, "loss": 0.0008967370958998799, "num_tokens": 120318097.0, "reward": 2.2503418922424316, "reward_std": 0.5485246181488037, "rewards/code_complexity_reward/mean": 0.886914074420929, "rewards/code_complexity_reward/std": 0.17654725909233093, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 750, "step_time": 67.52273076586425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 115.875, "completions/mean_terminated_length": 115.875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21980785648338497, "epoch": 0.8563283922462942, "frac_reward_zero_std": 0.515625, "grad_norm": 0.043211109936237335, "kl": 0.20201573602389544, "learning_rate": 3.1288793892244427e-07, "loss": 0.0010100877843797207, "num_tokens": 120446129.0, "reward": 2.2496094703674316, "reward_std": 0.510525107383728, "rewards/code_complexity_reward/mean": 0.8932616710662842, "rewards/code_complexity_reward/std": 0.15371620655059814, "rewards/code_execution_reward/mean": 0.267578125, "rewards/code_execution_reward/std": 0.4431293308734894, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.011048543266952038, "step": 751, "step_time": 48.84624019637704 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 113.0, "completions/mean_terminated_length": 113.0, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.21581411501392722, "epoch": 0.8574686431014823, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04443197697401047, "kl": 0.22547970665618777, "learning_rate": 3.080833697258842e-07, "loss": 0.0011272013653069735, "num_tokens": 120572805.0, "reward": 2.218554973602295, "reward_std": 0.5468956828117371, "rewards/code_complexity_reward/mean": 0.8877929449081421, "rewards/code_complexity_reward/std": 0.18911786377429962, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.030093414708971977, "step": 752, "step_time": 40.00688813533634 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 511.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 116.708984375, "completions/mean_terminated_length": 116.708984375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21902802004478872, "epoch": 0.8586088939566705, "frac_reward_zero_std": 0.46875, "grad_norm": 114.8189697265625, "kl": 43.3257318851538, "learning_rate": 3.0331355168059214e-07, "loss": 0.21619458496570587, "num_tokens": 120701240.0, "reward": 2.214648485183716, "reward_std": 0.5402310490608215, "rewards/code_complexity_reward/mean": 0.880664050579071, "rewards/code_complexity_reward/std": 0.1870230883359909, "rewards/code_execution_reward/mean": 0.255859375, "rewards/code_execution_reward/std": 0.43676990270614624, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.029112961143255234, "step": 753, "step_time": 49.96182425413281 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 490.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 115.458984375, "completions/mean_terminated_length": 115.458984375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.21534068742766976, "epoch": 0.8597491448118586, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.03942900151014328, "kl": 0.19421991286799312, "learning_rate": 2.985785604083649e-07, "loss": 0.000970932946074754, "num_tokens": 120829647.0, "reward": 2.2870118618011475, "reward_std": 0.5053163766860962, "rewards/code_complexity_reward/mean": 0.902050793170929, "rewards/code_complexity_reward/std": 0.12995818257331848, "rewards/code_execution_reward/mean": 0.294921875, "rewards/code_execution_reward/std": 0.4564536213874817, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.019055325537919998, "step": 754, "step_time": 57.12909788265824 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 446.0, "completions/max_terminated_length": 446.0, "completions/mean_length": 115.833984375, "completions/mean_terminated_length": 115.833984375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2064382415264845, "epoch": 0.8608893956670467, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.04089687764644623, "kl": 0.21024989674333483, "learning_rate": 2.9387847097884254e-07, "loss": 0.0010513439774513245, "num_tokens": 120956086.0, "reward": 2.2846193313598633, "reward_std": 0.5282123684883118, "rewards/code_complexity_reward/mean": 0.89697265625, "rewards/code_complexity_reward/std": 0.15871798992156982, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 755, "step_time": 49.652969061397016 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 112.341796875, "completions/mean_terminated_length": 112.341796875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21258824435062706, "epoch": 0.8620296465222349, "frac_reward_zero_std": 0.46875, "grad_norm": 0.044567666947841644, "kl": 0.22449997626245022, "learning_rate": 2.8921335790832757e-07, "loss": 0.0011228682706132531, "num_tokens": 121082309.0, "reward": 2.276172161102295, "reward_std": 0.5405170321464539, "rewards/code_complexity_reward/mean": 0.8905273675918579, "rewards/code_complexity_reward/std": 0.1602136343717575, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 756, "step_time": 41.81529042497277 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 503.0, "completions/max_terminated_length": 503.0, "completions/mean_length": 111.537109375, "completions/mean_terminated_length": 111.537109375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21752678649500012, "epoch": 0.863169897377423, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.05356040224432945, "kl": 0.22611759137362242, "learning_rate": 2.845832951585969e-07, "loss": 0.0011302826460450888, "num_tokens": 121207308.0, "reward": 2.348437547683716, "reward_std": 0.5553291440010071, "rewards/code_complexity_reward/mean": 0.8975585699081421, "rewards/code_complexity_reward/std": 0.1525353193283081, "rewards/code_execution_reward/mean": 0.36328125, "rewards/code_execution_reward/std": 0.4814152419567108, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 757, "step_time": 50.09807265922427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 445.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 115.005859375, "completions/mean_terminated_length": 115.005859375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.20781439170241356, "epoch": 0.8643101482326112, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.043142903596162796, "kl": 0.18581429601181298, "learning_rate": 2.799883561357314e-07, "loss": 0.0009290609741583467, "num_tokens": 121335447.0, "reward": 2.292773485183716, "reward_std": 0.5479538440704346, "rewards/code_complexity_reward/mean": 0.8924804925918579, "rewards/code_complexity_reward/std": 0.15992017090320587, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 758, "step_time": 53.410900495015085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 471.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 116.228515625, "completions/mean_terminated_length": 116.228515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21213341620750725, "epoch": 0.8654503990877993, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.03868380934000015, "kl": 0.19760041346307844, "learning_rate": 2.7542861368895444e-07, "loss": 0.0009878437267616391, "num_tokens": 121462492.0, "reward": 2.26611328125, "reward_std": 0.5179665088653564, "rewards/code_complexity_reward/mean": 0.9022460579872131, "rewards/code_complexity_reward/std": 0.14466992020606995, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03890470787882805, "step": 759, "step_time": 48.120254427194595 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 116.6328125, "completions/mean_terminated_length": 116.6328125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21786275506019592, "epoch": 0.8665906499429875, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.042424898594617844, "kl": 0.19778672163374722, "learning_rate": 2.709041401094717e-07, "loss": 0.0009888762142509222, "num_tokens": 121592632.0, "reward": 2.265625, "reward_std": 0.5226008296012878, "rewards/code_complexity_reward/mean": 0.897265613079071, "rewards/code_complexity_reward/std": 0.15376025438308716, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.022097086533904076, "step": 760, "step_time": 48.49499293230474 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 116.94921875, "completions/mean_terminated_length": 116.1761245727539, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.22042955574579537, "epoch": 0.8677309007981756, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04126614332199097, "kl": 0.19364071334712207, "learning_rate": 2.664150071293314e-07, "loss": 0.0009682658710516989, "num_tokens": 121718590.0, "reward": 2.2633302211761475, "reward_std": 0.5220826268196106, "rewards/code_complexity_reward/mean": 0.89599609375, "rewards/code_complexity_reward/std": 0.1514398604631424, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 761, "step_time": 49.165006808936596 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 507.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 117.96875, "completions/mean_terminated_length": 117.96875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2122872401960194, "epoch": 0.8688711516533637, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05057762563228607, "kl": 0.2161692613735795, "learning_rate": 2.619612859202808e-07, "loss": 0.0010804469930008054, "num_tokens": 121845882.0, "reward": 2.2913575172424316, "reward_std": 0.5545773506164551, "rewards/code_complexity_reward/mean": 0.8878905773162842, "rewards/code_complexity_reward/std": 0.16634626686573029, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.024002734571695328, "step": 762, "step_time": 49.25319650955498 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 470.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 117.828125, "completions/mean_terminated_length": 117.828125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21835078834556043, "epoch": 0.8700114025085519, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.047088805586099625, "kl": 0.21298021846450865, "learning_rate": 2.575430470926421e-07, "loss": 0.0010648812167346478, "num_tokens": 121973042.0, "reward": 2.278076171875, "reward_std": 0.565528929233551, "rewards/code_complexity_reward/mean": 0.882128894329071, "rewards/code_complexity_reward/std": 0.18313950300216675, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 763, "step_time": 45.61383614875376 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 111.462890625, "completions/mean_terminated_length": 111.462890625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2255581037607044, "epoch": 0.87115165336374, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.03946434333920479, "kl": 0.216792248073034, "learning_rate": 2.531603606941929e-07, "loss": 0.001084119314327836, "num_tokens": 122098067.0, "reward": 2.2571778297424316, "reward_std": 0.5159050226211548, "rewards/code_complexity_reward/mean": 0.9029296636581421, "rewards/code_complexity_reward/std": 0.14242520928382874, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03686080500483513, "step": 764, "step_time": 48.46408719290048 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 438.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 115.87109375, "completions/mean_terminated_length": 115.87109375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21154077188111842, "epoch": 0.8722919042189282, "frac_reward_zero_std": 0.5703125, "grad_norm": 0.04390939697623253, "kl": 0.1957313499879092, "learning_rate": 2.4881329620905144e-07, "loss": 0.0009785837028175592, "num_tokens": 122225877.0, "reward": 2.2132325172424316, "reward_std": 0.5190165042877197, "rewards/code_complexity_reward/mean": 0.8871093988418579, "rewards/code_complexity_reward/std": 0.16684550046920776, "rewards/code_execution_reward/mean": 0.244140625, "rewards/code_execution_reward/std": 0.42999663949012756, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.030623575672507286, "step": 765, "step_time": 43.8139785528183 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 118.525390625, "completions/mean_terminated_length": 117.75537872314453, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21206878568045795, "epoch": 0.8734321550741163, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04298677295446396, "kl": 0.21220060368068516, "learning_rate": 2.4450192255658115e-07, "loss": 0.0010609703604131937, "num_tokens": 122355842.0, "reward": 2.238330125808716, "reward_std": 0.5084263682365417, "rewards/code_complexity_reward/mean": 0.8994140625, "rewards/code_complexity_reward/std": 0.14385537803173065, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.04502354562282562, "step": 766, "step_time": 74.5010131392628 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 114.744140625, "completions/mean_terminated_length": 114.744140625, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.21268913778476417, "epoch": 0.8745724059293044, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04768912494182587, "kl": 0.21691558801103383, "learning_rate": 2.402263080902917e-07, "loss": 0.0010845153592526913, "num_tokens": 122482063.0, "reward": 2.2347657680511475, "reward_std": 0.4782600998878479, "rewards/code_complexity_reward/mean": 0.9063476324081421, "rewards/code_complexity_reward/std": 0.12599405646324158, "rewards/code_execution_reward/mean": 0.240234375, "rewards/code_execution_reward/std": 0.4276435375213623, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.038950733840465546, "step": 767, "step_time": 38.210683471523225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 116.31640625, "completions/mean_terminated_length": 116.31640625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2208328174892813, "epoch": 0.8757126567844926, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.041325461119413376, "kl": 0.19393406563904136, "learning_rate": 2.359865205967618e-07, "loss": 0.0009695308981463313, "num_tokens": 122608993.0, "reward": 2.2982423305511475, "reward_std": 0.5174776315689087, "rewards/code_complexity_reward/mean": 0.89990234375, "rewards/code_complexity_reward/std": 0.1347062885761261, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 768, "step_time": 48.72921137884259 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 466.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 118.20703125, "completions/mean_terminated_length": 118.20703125, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.2104203801136464, "epoch": 0.8768529076396807, "frac_reward_zero_std": 0.515625, "grad_norm": 0.04475165158510208, "kl": 0.18692770693451166, "learning_rate": 2.317826272945578e-07, "loss": 0.0009346536826342344, "num_tokens": 122738459.0, "reward": 2.2975587844848633, "reward_std": 0.5504258275032043, "rewards/code_complexity_reward/mean": 0.89453125, "rewards/code_complexity_reward/std": 0.1542590856552124, "rewards/code_execution_reward/mean": 0.3203125, "rewards/code_execution_reward/std": 0.4670529365539551, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.03805733472108841, "step": 769, "step_time": 45.94161521177739 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 376.0, "completions/mean_length": 119.23828125, "completions/mean_terminated_length": 116.92338562011719, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.21719333995133638, "epoch": 0.8779931584948689, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.03942590951919556, "kl": 0.19050398492254317, "learning_rate": 2.2761469483317257e-07, "loss": 0.0009524331544525921, "num_tokens": 122868953.0, "reward": 2.281494140625, "reward_std": 0.5369817614555359, "rewards/code_complexity_reward/mean": 0.893359363079071, "rewards/code_complexity_reward/std": 0.15641796588897705, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03607473522424698, "step": 770, "step_time": 59.03536934964359 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 120.943359375, "completions/mean_terminated_length": 119.4098129272461, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21810190193355083, "epoch": 0.879133409350057, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.040216606110334396, "kl": 0.20403253962285817, "learning_rate": 2.2348278929196886e-07, "loss": 0.0010202830890193582, "num_tokens": 122997348.0, "reward": 2.2439942359924316, "reward_std": 0.5259501934051514, "rewards/code_complexity_reward/mean": 0.887890636920929, "rewards/code_complexity_reward/std": 0.15832984447479248, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09902343899011612, "rewards/reasoning_present_reward_func/std": 0.009843364357948303, "rewards/xmlcount_reward_func/mean": 0.494384765625, "rewards/xmlcount_reward_func/std": 0.0413680262863636, "step": 771, "step_time": 58.399030366912484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 414.0, "completions/mean_length": 113.318359375, "completions/mean_terminated_length": 112.53816223144531, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2254500286653638, "epoch": 0.8802736602052451, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04636260122060776, "kl": 0.20832840399816632, "learning_rate": 2.1938697617912759e-07, "loss": 0.0010415586875751615, "num_tokens": 123123855.0, "reward": 2.2608399391174316, "reward_std": 0.5253363251686096, "rewards/code_complexity_reward/mean": 0.8995116949081421, "rewards/code_complexity_reward/std": 0.15408577024936676, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.032946839928627014, "step": 772, "step_time": 58.67137873265892 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 418.0, "completions/max_terminated_length": 418.0, "completions/mean_length": 118.044921875, "completions/mean_terminated_length": 118.044921875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.222000434063375, "epoch": 0.8814139110604333, "frac_reward_zero_std": 0.4375, "grad_norm": 0.04367734491825104, "kl": 0.20684762846212834, "learning_rate": 2.1532732043061527e-07, "loss": 0.0010341042652726173, "num_tokens": 123253402.0, "reward": 2.167236328125, "reward_std": 0.5365775227546692, "rewards/code_complexity_reward/mean": 0.871874988079071, "rewards/code_complexity_reward/std": 0.1977366954088211, "rewards/code_execution_reward/mean": 0.21875, "rewards/code_execution_reward/std": 0.41380295157432556, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 773, "step_time": 46.88944824412465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 431.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 120.60546875, "completions/mean_terminated_length": 120.60546875, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.2185527803376317, "epoch": 0.8825541619156214, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.0441366471350193, "kl": 0.19440712744835764, "learning_rate": 2.1130388640914794e-07, "loss": 0.0009716692147776484, "num_tokens": 123383444.0, "reward": 2.2370119094848633, "reward_std": 0.49876293540000916, "rewards/code_complexity_reward/mean": 0.891796886920929, "rewards/code_complexity_reward/std": 0.14090529084205627, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03651979938149452, "step": 774, "step_time": 45.058152372948825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 465.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 109.765625, "completions/mean_terminated_length": 109.765625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21762084448710084, "epoch": 0.8836944127708096, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.04497579485177994, "kl": 0.2157475792337209, "learning_rate": 2.0731673790317596e-07, "loss": 0.001079107285477221, "num_tokens": 123506064.0, "reward": 2.3661134243011475, "reward_std": 0.5563362240791321, "rewards/code_complexity_reward/mean": 0.900390625, "rewards/code_complexity_reward/std": 0.15115290880203247, "rewards/code_execution_reward/mean": 0.37890625, "rewards/code_execution_reward/std": 0.4855891764163971, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 775, "step_time": 54.714161553420126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 432.0, "completions/max_terminated_length": 432.0, "completions/mean_length": 115.73828125, "completions/mean_terminated_length": 115.73828125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.20919834566302598, "epoch": 0.8848346636259977, "frac_reward_zero_std": 0.46875, "grad_norm": 0.041407935321331024, "kl": 0.18343709188047796, "learning_rate": 2.0336593812587096e-07, "loss": 0.0009171862620860338, "num_tokens": 123632650.0, "reward": 2.3702149391174316, "reward_std": 0.5265635848045349, "rewards/code_complexity_reward/mean": 0.9054687023162842, "rewards/code_complexity_reward/std": 0.11682934314012527, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 776, "step_time": 53.798917825333774 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 490.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 119.80859375, "completions/mean_terminated_length": 119.80859375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.21468158601783216, "epoch": 0.8859749144811858, "frac_reward_zero_std": 0.515625, "grad_norm": 0.037401121109724045, "kl": 0.18451015569735318, "learning_rate": 1.9945154971412168e-07, "loss": 0.0009224032401107252, "num_tokens": 123761560.0, "reward": 2.2586426734924316, "reward_std": 0.5502644181251526, "rewards/code_complexity_reward/mean": 0.8882812261581421, "rewards/code_complexity_reward/std": 0.17334449291229248, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.039215847849845886, "step": 777, "step_time": 50.84787834342569 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 404.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 115.349609375, "completions/mean_terminated_length": 115.349609375, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.21678764722310007, "epoch": 0.887115165336374, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04812830686569214, "kl": 0.20837289839982986, "learning_rate": 1.9557363472754582e-07, "loss": 0.0010417886078357697, "num_tokens": 123888599.0, "reward": 2.245410203933716, "reward_std": 0.5091923475265503, "rewards/code_complexity_reward/mean": 0.8974609375, "rewards/code_complexity_reward/std": 0.14865146577358246, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.032150521874427795, "step": 778, "step_time": 52.05088079441339 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 436.0, "completions/max_terminated_length": 436.0, "completions/mean_length": 111.986328125, "completions/mean_terminated_length": 111.986328125, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.22123950510285795, "epoch": 0.8882554161915621, "frac_reward_zero_std": 0.421875, "grad_norm": 0.042469073086977005, "kl": 0.20475109457038343, "learning_rate": 1.9173225464749867e-07, "loss": 0.0010234835790470243, "num_tokens": 124013012.0, "reward": 2.2630372047424316, "reward_std": 0.5787704586982727, "rewards/code_complexity_reward/mean": 0.8811522722244263, "rewards/code_complexity_reward/std": 0.19035254418849945, "rewards/code_execution_reward/mean": 0.30859375, "rewards/code_execution_reward/std": 0.4623647928237915, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.493408203125, "rewards/xmlcount_reward_func/std": 0.046124301850795746, "step": 779, "step_time": 52.426566747017205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 112.0234375, "completions/mean_terminated_length": 111.24070739746094, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.21853644168004394, "epoch": 0.8893956670467503, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.040782373398542404, "kl": 0.20170468045398593, "learning_rate": 1.879274703761072e-07, "loss": 0.0010085252579301596, "num_tokens": 124139148.0, "reward": 2.2474608421325684, "reward_std": 0.539549708366394, "rewards/code_complexity_reward/mean": 0.8915039300918579, "rewards/code_complexity_reward/std": 0.17047499120235443, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.029059575870633125, "step": 780, "step_time": 50.99517859145999 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 345.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 111.59375, "completions/mean_terminated_length": 111.59375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.21300466661341488, "epoch": 0.8905359179019384, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.040565475821495056, "kl": 0.19784568075556308, "learning_rate": 1.8415934223529665e-07, "loss": 0.0009891495574265718, "num_tokens": 124263660.0, "reward": 2.2464356422424316, "reward_std": 0.5635371804237366, "rewards/code_complexity_reward/mean": 0.8825194835662842, "rewards/code_complexity_reward/std": 0.18516956269741058, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 781, "step_time": 38.49750126898289 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 488.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 118.6640625, "completions/mean_terminated_length": 118.6640625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21784874610602856, "epoch": 0.8916761687571265, "frac_reward_zero_std": 0.578125, "grad_norm": 0.03881094232201576, "kl": 0.1908741262741387, "learning_rate": 1.8042792996583902e-07, "loss": 0.0009543477790430188, "num_tokens": 124392428.0, "reward": 2.2516114711761475, "reward_std": 0.4785860776901245, "rewards/code_complexity_reward/mean": 0.8976562023162842, "rewards/code_complexity_reward/std": 0.1250220090150833, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.033528104424476624, "step": 782, "step_time": 49.615790996700525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 109.5078125, "completions/mean_terminated_length": 109.5078125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2081959145143628, "epoch": 0.8928164196123147, "frac_reward_zero_std": 0.578125, "grad_norm": 0.03975078836083412, "kl": 0.2265079969074577, "learning_rate": 1.76733292726404e-07, "loss": 0.0011325395898893476, "num_tokens": 124516424.0, "reward": 2.280566453933716, "reward_std": 0.5155467391014099, "rewards/code_complexity_reward/mean": 0.9043945074081421, "rewards/code_complexity_reward/std": 0.13896185159683228, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.01350956130772829, "step": 783, "step_time": 38.55918738339096 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 507.0, "completions/mean_length": 117.908203125, "completions/mean_terminated_length": 116.36275482177734, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.21453402540646493, "epoch": 0.8939566704675028, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.04687684401869774, "kl": 0.1846365159144625, "learning_rate": 1.7307548909262118e-07, "loss": 0.0009231336880475283, "num_tokens": 124645273.0, "reward": 2.2313477993011475, "reward_std": 0.550676703453064, "rewards/code_complexity_reward/mean": 0.8807617425918579, "rewards/code_complexity_reward/std": 0.17214544117450714, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.030144967138767242, "step": 784, "step_time": 64.34970055706799 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 112.138671875, "completions/mean_terminated_length": 111.35616302490234, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2100252821110189, "epoch": 0.895096921322691, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.038228850811719894, "kl": 0.20442377543076873, "learning_rate": 1.694545770561537e-07, "loss": 0.001022137119434774, "num_tokens": 124771212.0, "reward": 2.2426271438598633, "reward_std": 0.537539541721344, "rewards/code_complexity_reward/mean": 0.895312488079071, "rewards/code_complexity_reward/std": 0.16934169828891754, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.012342973612248898, "step": 785, "step_time": 49.07928215432912 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 364.0, "completions/max_terminated_length": 364.0, "completions/mean_length": 108.28515625, "completions/mean_terminated_length": 108.28515625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2069990891031921, "epoch": 0.8962371721778791, "frac_reward_zero_std": 0.53125, "grad_norm": 0.04385004937648773, "kl": 0.2080312727484852, "learning_rate": 1.658706140237737e-07, "loss": 0.001040260074660182, "num_tokens": 124893886.0, "reward": 2.3250489234924316, "reward_std": 0.5243285894393921, "rewards/code_complexity_reward/mean": 0.9061523675918579, "rewards/code_complexity_reward/std": 0.12914834916591644, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.022766664624214172, "step": 786, "step_time": 47.07541485503316 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 498.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 114.75, "completions/mean_terminated_length": 114.75, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.20450246008113027, "epoch": 0.8973774230330672, "frac_reward_zero_std": 0.40625, "grad_norm": 0.04412291571497917, "kl": 0.20053558191284537, "learning_rate": 1.623236568164574e-07, "loss": 0.0010027586249634624, "num_tokens": 125019462.0, "reward": 2.2565431594848633, "reward_std": 0.5355933904647827, "rewards/code_complexity_reward/mean": 0.89306640625, "rewards/code_complexity_reward/std": 0.16128692030906677, "rewards/code_execution_reward/mean": 0.283203125, "rewards/code_execution_reward/std": 0.4509948492050171, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.494140625, "rewards/xmlcount_reward_func/std": 0.04784845933318138, "step": 787, "step_time": 58.04436765797436 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 114.96484375, "completions/mean_terminated_length": 114.1878662109375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2125592848751694, "epoch": 0.8985176738882554, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.04445023089647293, "kl": 0.19685897091403604, "learning_rate": 1.5881376166848151e-07, "loss": 0.0009842792060226202, "num_tokens": 125145868.0, "reward": 2.256152629852295, "reward_std": 0.5867284536361694, "rewards/code_complexity_reward/mean": 0.8763672113418579, "rewards/code_complexity_reward/std": 0.2038717269897461, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4775390625, "rewards/code_syntax_reward/std": 0.10366757214069366, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 788, "step_time": 59.249347490258515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 117.83984375, "completions/mean_terminated_length": 117.06848907470703, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2263654707930982, "epoch": 0.8996579247434435, "frac_reward_zero_std": 0.453125, "grad_norm": 0.05321469530463219, "kl": 0.21609585965052247, "learning_rate": 1.5534098422653243e-07, "loss": 0.001080546178855002, "num_tokens": 125275234.0, "reward": 2.212451219558716, "reward_std": 0.47182610630989075, "rewards/code_complexity_reward/mean": 0.9048827886581421, "rewards/code_complexity_reward/std": 0.13419902324676514, "rewards/code_execution_reward/mean": 0.220703125, "rewards/code_execution_reward/std": 0.4151262938976288, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.03686080500483513, "step": 789, "step_time": 58.47614361625165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 110.78125, "completions/mean_terminated_length": 110.78125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2186935825739056, "epoch": 0.9007981755986317, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.0470275916159153, "kl": 0.21038230252452195, "learning_rate": 1.5190537954882373e-07, "loss": 0.0010517413029447198, "num_tokens": 125400334.0, "reward": 2.30419921875, "reward_std": 0.5574740767478943, "rewards/code_complexity_reward/mean": 0.8914061784744263, "rewards/code_complexity_reward/std": 0.16425831615924835, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.028089813888072968, "step": 790, "step_time": 39.71652852278203 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 501.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 118.23046875, "completions/mean_terminated_length": 118.23046875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22461070539429784, "epoch": 0.9019384264538198, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04518669471144676, "kl": 0.19142539077438414, "learning_rate": 1.485070021042237e-07, "loss": 0.0009571927366778255, "num_tokens": 125529264.0, "reward": 2.2312989234924316, "reward_std": 0.5068253874778748, "rewards/code_complexity_reward/mean": 0.8912109136581421, "rewards/code_complexity_reward/std": 0.14705489575862885, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03343535214662552, "step": 791, "step_time": 55.523878030478954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 397.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 112.513671875, "completions/mean_terminated_length": 112.513671875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21138334670104086, "epoch": 0.9030786773090079, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04049926996231079, "kl": 0.20370589895173907, "learning_rate": 1.4514590577139136e-07, "loss": 0.001018546987324953, "num_tokens": 125656575.0, "reward": 2.2837891578674316, "reward_std": 0.5329834222793579, "rewards/code_complexity_reward/mean": 0.8993164300918579, "rewards/code_complexity_reward/std": 0.1485893279314041, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03480498120188713, "step": 792, "step_time": 45.54128070920706 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 115.5546875, "completions/mean_terminated_length": 114.77886199951172, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.217420922126621, "epoch": 0.9042189281641961, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04215313866734505, "kl": 0.19338584085926414, "learning_rate": 1.4182214383792192e-07, "loss": 0.0009666542755439878, "num_tokens": 125784831.0, "reward": 2.2254395484924316, "reward_std": 0.5166892409324646, "rewards/code_complexity_reward/mean": 0.8949218392372131, "rewards/code_complexity_reward/std": 0.162759929895401, "rewards/code_execution_reward/mean": 0.24609375, "rewards/code_execution_reward/std": 0.4311550557613373, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 793, "step_time": 59.795641362667084 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 113.833984375, "completions/mean_terminated_length": 113.833984375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22458418388850987, "epoch": 0.9053591790193842, "frac_reward_zero_std": 0.515625, "grad_norm": 0.046478766947984695, "kl": 0.2084858367452398, "learning_rate": 1.3853576899950343e-07, "loss": 0.0010429247049614787, "num_tokens": 125910790.0, "reward": 2.1958985328674316, "reward_std": 0.46153703331947327, "rewards/code_complexity_reward/mean": 0.8996093273162842, "rewards/code_complexity_reward/std": 0.1317867636680603, "rewards/code_execution_reward/mean": 0.205078125, "rewards/code_execution_reward/std": 0.4041535556316376, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.011016063392162323, "step": 794, "step_time": 37.423754767514765 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 352.0, "completions/max_terminated_length": 352.0, "completions/mean_length": 108.677734375, "completions/mean_terminated_length": 108.677734375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21567135374061763, "epoch": 0.9064994298745724, "frac_reward_zero_std": 0.546875, "grad_norm": 0.0408976674079895, "kl": 0.20090107014402747, "learning_rate": 1.3528683335907928e-07, "loss": 0.0010044695809483528, "num_tokens": 126036149.0, "reward": 2.292529344558716, "reward_std": 0.4862232208251953, "rewards/code_complexity_reward/mean": 0.911425769329071, "rewards/code_complexity_reward/std": 0.10492434352636337, "rewards/code_execution_reward/mean": 0.2890625, "rewards/code_execution_reward/std": 0.45377036929130554, "rewards/code_syntax_reward/mean": 0.4951171875, "rewards/code_syntax_reward/std": 0.049216821789741516, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.027517380192875862, "step": 795, "step_time": 55.947700562886894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 111.68359375, "completions/mean_terminated_length": 111.68359375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2189141195267439, "epoch": 0.9076396807297605, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.039945099502801895, "kl": 0.18472794780973345, "learning_rate": 1.3207538842602396e-07, "loss": 0.0009237727499566972, "num_tokens": 126160539.0, "reward": 2.3336915969848633, "reward_std": 0.5500779747962952, "rewards/code_complexity_reward/mean": 0.9006836414337158, "rewards/code_complexity_reward/std": 0.14492228627204895, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02465205453336239, "step": 796, "step_time": 36.02621084544808 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 480.0, "completions/mean_length": 118.689453125, "completions/mean_terminated_length": 116.37132263183594, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22088947845622897, "epoch": 0.9087799315849487, "frac_reward_zero_std": 0.46875, "grad_norm": 0.05059666186571121, "kl": 0.2202872873749584, "learning_rate": 1.289014851153253e-07, "loss": 0.0011011086171492934, "num_tokens": 126288264.0, "reward": 2.314941644668579, "reward_std": 0.5725871920585632, "rewards/code_complexity_reward/mean": 0.885058581829071, "rewards/code_complexity_reward/std": 0.1740683764219284, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.01737673208117485, "step": 797, "step_time": 56.95119221974164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 110.98828125, "completions/mean_terminated_length": 110.98828125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2137386854737997, "epoch": 0.9099201824401368, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04774702712893486, "kl": 0.21959953452460468, "learning_rate": 1.2576517374677745e-07, "loss": 0.0010981005616486073, "num_tokens": 126411362.0, "reward": 2.29296875, "reward_std": 0.5405176877975464, "rewards/code_complexity_reward/mean": 0.89892578125, "rewards/code_complexity_reward/std": 0.16076450049877167, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.028089813888072968, "step": 798, "step_time": 47.11840798147023 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 482.0, "completions/max_terminated_length": 482.0, "completions/mean_length": 116.552734375, "completions/mean_terminated_length": 116.552734375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21483815414831042, "epoch": 0.9110604332953249, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.042184017598629, "kl": 0.19649435149040073, "learning_rate": 1.2266650404418378e-07, "loss": 0.000982570694759488, "num_tokens": 126539409.0, "reward": 2.270556688308716, "reward_std": 0.5842570066452026, "rewards/code_complexity_reward/mean": 0.8744140863418579, "rewards/code_complexity_reward/std": 0.19297762215137482, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 799, "step_time": 64.31286925263703 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 488.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 113.8671875, "completions/mean_terminated_length": 113.8671875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21350824064575136, "epoch": 0.9122006841505131, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.0393916517496109, "kl": 0.20011721341870725, "learning_rate": 1.1960552513456764e-07, "loss": 0.0010001660557463765, "num_tokens": 126665609.0, "reward": 2.2609376907348633, "reward_std": 0.5210233330726624, "rewards/code_complexity_reward/mean": 0.899609386920929, "rewards/code_complexity_reward/std": 0.14665161073207855, "rewards/code_execution_reward/mean": 0.275390625, "rewards/code_execution_reward/std": 0.44714778661727905, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.03484956547617912, "step": 800, "step_time": 59.508081802167 }, { "epoch": 0.9122006841505131, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0025, "eval_completions/max_length": 186.36, "eval_completions/max_terminated_length": 184.34, "eval_completions/mean_length": 114.79, "eval_completions/mean_terminated_length": 114.00107147216796, "eval_completions/min_length": 75.16, "eval_completions/min_terminated_length": 75.16, "eval_entropy": 0.2183621370792389, "eval_frac_reward_zero_std": 0.5, "eval_kl": 0.19977654427289962, "eval_loss": 0.0010011015692725778, "eval_num_tokens": 126665609.0, "eval_reward": 2.231500120162964, "eval_reward_std": 0.35087743535637855, "eval_rewards/code_complexity_reward/mean": 0.8967499780654907, "eval_rewards/code_complexity_reward/std": 0.08326835036277772, "eval_rewards/code_execution_reward/mean": 0.2475, "eval_rewards/code_execution_reward/std": 0.27894101560115814, "eval_rewards/code_syntax_reward/mean": 0.49, "eval_rewards/code_syntax_reward/std": 0.02584230363368988, "eval_rewards/reasoning_present_reward_func/mean": 0.09975000157952309, "eval_rewards/reasoning_present_reward_func/std": 0.000707106813788414, "eval_rewards/xmlcount_reward_func/mean": 0.4975, "eval_rewards/xmlcount_reward_func/std": 0.007071067690849304, "eval_runtime": 395.3069, "eval_samples_per_second": 0.253, "eval_steps_per_second": 0.033, "step": 800 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 419.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 113.74609375, "completions/mean_terminated_length": 113.74609375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21292177215218544, "epoch": 0.9133409350057012, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04486369341611862, "kl": 0.192747387685813, "learning_rate": 1.1658228554739359e-07, "loss": 0.00096387870144099, "num_tokens": 126792619.0, "reward": 2.307373046875, "reward_std": 0.5374143719673157, "rewards/code_complexity_reward/mean": 0.893261730670929, "rewards/code_complexity_reward/std": 0.145272895693779, "rewards/code_execution_reward/mean": 0.328125, "rewards/code_execution_reward/std": 0.4699897766113281, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 801, "step_time": 42.87398476805538 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 329.0, "completions/mean_length": 113.203125, "completions/mean_terminated_length": 111.63922119140625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2184194321744144, "epoch": 0.9144811858608894, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.04143727198243141, "kl": 0.2083915607072413, "learning_rate": 1.1359683321379878e-07, "loss": 0.0010420015314593911, "num_tokens": 126918339.0, "reward": 2.2601075172424316, "reward_std": 0.4952061176300049, "rewards/code_complexity_reward/mean": 0.9085937738418579, "rewards/code_complexity_reward/std": 0.13049796223640442, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.014579027891159058, "step": 802, "step_time": 51.510687762871385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 111.455078125, "completions/mean_terminated_length": 111.455078125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21246732911095023, "epoch": 0.9156214367160775, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04558968171477318, "kl": 0.21352698863483965, "learning_rate": 1.106492154658323e-07, "loss": 0.0010676190722733736, "num_tokens": 127041412.0, "reward": 2.3175783157348633, "reward_std": 0.5698611736297607, "rewards/code_complexity_reward/mean": 0.8874022960662842, "rewards/code_complexity_reward/std": 0.16670659184455872, "rewards/code_execution_reward/mean": 0.34765625, "rewards/code_execution_reward/std": 0.47669193148612976, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.032150521874427795, "step": 803, "step_time": 47.354887251742184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 454.0, "completions/max_terminated_length": 454.0, "completions/mean_length": 117.025390625, "completions/mean_terminated_length": 117.025390625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21962533053010702, "epoch": 0.9167616875712656, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.04576835408806801, "kl": 0.19087058084551245, "learning_rate": 1.0773947903570503e-07, "loss": 0.0009545038919895887, "num_tokens": 127169377.0, "reward": 2.272265672683716, "reward_std": 0.48768579959869385, "rewards/code_complexity_reward/mean": 0.9039062261581421, "rewards/code_complexity_reward/std": 0.120843306183815, "rewards/code_execution_reward/mean": 0.27734375, "rewards/code_execution_reward/std": 0.4481254518032074, "rewards/code_syntax_reward/mean": 0.494140625, "rewards/code_syntax_reward/std": 0.05386113002896309, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.033960822969675064, "step": 804, "step_time": 62.54892620164901 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 111.64453125, "completions/mean_terminated_length": 110.86105346679688, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21635859343223274, "epoch": 0.9179019384264538, "frac_reward_zero_std": 0.5625, "grad_norm": 0.04148300737142563, "kl": 0.24593737116083503, "learning_rate": 1.0486767005504911e-07, "loss": 0.0012295714113861322, "num_tokens": 127294399.0, "reward": 2.2835938930511475, "reward_std": 0.5295794606208801, "rewards/code_complexity_reward/mean": 0.8983398675918579, "rewards/code_complexity_reward/std": 0.1453862488269806, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.04114582762122154, "step": 805, "step_time": 50.12252782750875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 114.521484375, "completions/mean_terminated_length": 114.521484375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22128788218833506, "epoch": 0.9190421892816419, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04916040971875191, "kl": 0.19607493677176535, "learning_rate": 1.0203383405418515e-07, "loss": 0.000980221200734377, "num_tokens": 127421982.0, "reward": 2.2408692836761475, "reward_std": 0.5715091824531555, "rewards/code_complexity_reward/mean": 0.8833984136581421, "rewards/code_complexity_reward/std": 0.19418597221374512, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.478515625, "rewards/code_syntax_reward/std": 0.1014925017952919, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.01825989969074726, "step": 806, "step_time": 59.45089708082378 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 114.642578125, "completions/mean_terminated_length": 113.8649673461914, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.21993986750021577, "epoch": 0.9201824401368301, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.0436062291264534, "kl": 0.20709582406561822, "learning_rate": 9.923801596140258e-08, "loss": 0.0010355215054005384, "num_tokens": 127548391.0, "reward": 2.2654786109924316, "reward_std": 0.5306375622749329, "rewards/code_complexity_reward/mean": 0.898242175579071, "rewards/code_complexity_reward/std": 0.15336038172245026, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.039987966418266296, "step": 807, "step_time": 51.04188730940223 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 442.0, "completions/mean_length": 114.62890625, "completions/mean_terminated_length": 113.85127258300781, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21015164349228144, "epoch": 0.9213226909920182, "frac_reward_zero_std": 0.515625, "grad_norm": 0.037061724811792374, "kl": 0.19574393914081156, "learning_rate": 9.64802601022452e-08, "loss": 0.0009786745067685843, "num_tokens": 127675321.0, "reward": 2.35791015625, "reward_std": 0.5558032393455505, "rewards/code_complexity_reward/mean": 0.900585949420929, "rewards/code_complexity_reward/std": 0.14741626381874084, "rewards/code_execution_reward/mean": 0.37109375, "rewards/code_execution_reward/std": 0.4835699498653412, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.038950733840465546, "step": 808, "step_time": 60.10820126533508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 498.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 112.0625, "completions/mean_terminated_length": 112.0625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2127607346046716, "epoch": 0.9224629418472063, "frac_reward_zero_std": 0.4140625, "grad_norm": 0.04378420114517212, "kl": 0.1910053533501923, "learning_rate": 9.376061019881006e-08, "loss": 0.0009550874237902462, "num_tokens": 127800737.0, "reward": 2.2945313453674316, "reward_std": 0.5830758213996887, "rewards/code_complexity_reward/mean": 0.8815429210662842, "rewards/code_complexity_reward/std": 0.18591855466365814, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4814453125, "rewards/code_syntax_reward/std": 0.09460734575986862, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 809, "step_time": 58.346940516494215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 445.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 119.236328125, "completions/mean_terminated_length": 119.236328125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.20861529000103474, "epoch": 0.9236031927023945, "frac_reward_zero_std": 0.421875, "grad_norm": 0.04605037719011307, "kl": 0.2046043227892369, "learning_rate": 9.107910936905301e-08, "loss": 0.00102278683334589, "num_tokens": 127929974.0, "reward": 2.2229981422424316, "reward_std": 0.5308530926704407, "rewards/code_complexity_reward/mean": 0.8845702409744263, "rewards/code_complexity_reward/std": 0.1679454743862152, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.041540902107954025, "step": 810, "step_time": 56.411430140957236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 114.58984375, "completions/mean_terminated_length": 114.58984375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21016489318571985, "epoch": 0.9247434435575826, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04516269266605377, "kl": 0.21799322485458106, "learning_rate": 8.843580012610625e-08, "loss": 0.0010899801272898912, "num_tokens": 128056588.0, "reward": 2.2711427211761475, "reward_std": 0.4987984597682953, "rewards/code_complexity_reward/mean": 0.905566394329071, "rewards/code_complexity_reward/std": 0.11998561024665833, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09921874850988388, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.493896484375, "rewards/xmlcount_reward_func/std": 0.04813653603196144, "step": 811, "step_time": 49.24558804463595 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 112.771484375, "completions/mean_terminated_length": 112.771484375, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21891216933727264, "epoch": 0.9258836944127709, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.040008239448070526, "kl": 0.2000774551415816, "learning_rate": 8.583072437760381e-08, "loss": 0.0010004076175391674, "num_tokens": 128185003.0, "reward": 2.2330079078674316, "reward_std": 0.47346606850624084, "rewards/code_complexity_reward/mean": 0.9083983898162842, "rewards/code_complexity_reward/std": 0.12662938237190247, "rewards/code_execution_reward/mean": 0.234375, "rewards/code_execution_reward/std": 0.42402184009552, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.025862684473395348, "step": 812, "step_time": 46.29089165478945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 469.0, "completions/max_terminated_length": 469.0, "completions/mean_length": 114.36328125, "completions/mean_terminated_length": 114.36328125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.21886793081648648, "epoch": 0.927023945267959, "frac_reward_zero_std": 0.421875, "grad_norm": 0.04507589340209961, "kl": 0.20019790914375335, "learning_rate": 8.3263923425016e-08, "loss": 0.001000968157313764, "num_tokens": 128313941.0, "reward": 2.2731447219848633, "reward_std": 0.5796313881874084, "rewards/code_complexity_reward/mean": 0.878710925579071, "rewards/code_complexity_reward/std": 0.1898498237133026, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.48046875, "rewards/code_syntax_reward/std": 0.09696658700704575, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.028089813888072968, "step": 813, "step_time": 53.78419906273484 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 501.0, "completions/mean_length": 121.859375, "completions/mean_terminated_length": 119.55992889404297, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.20803004642948508, "epoch": 0.928164196123147, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.04425416141748428, "kl": 0.18820834311190993, "learning_rate": 8.07354379629971e-08, "loss": 0.0009413448278792202, "num_tokens": 128443365.0, "reward": 2.293212890625, "reward_std": 0.5506348013877869, "rewards/code_complexity_reward/mean": 0.89013671875, "rewards/code_complexity_reward/std": 0.1587480902671814, "rewards/code_execution_reward/mean": 0.318359375, "rewards/code_execution_reward/std": 0.46629536151885986, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.033485326915979385, "step": 814, "step_time": 49.26558659784496 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 430.0, "completions/mean_length": 119.1015625, "completions/mean_terminated_length": 118.33267974853516, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21575527335517108, "epoch": 0.9293044469783353, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.03945532068610191, "kl": 0.2058213505661115, "learning_rate": 7.82453080787382e-08, "loss": 0.0010291491635143757, "num_tokens": 128569809.0, "reward": 2.2709474563598633, "reward_std": 0.5047518014907837, "rewards/code_complexity_reward/mean": 0.8994140625, "rewards/code_complexity_reward/std": 0.1304427534341812, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.024042511358857155, "step": 815, "step_time": 79.2996659995988 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 112.498046875, "completions/mean_terminated_length": 112.498046875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21443129517138004, "epoch": 0.9304446978335233, "frac_reward_zero_std": 0.53125, "grad_norm": 0.044263437390327454, "kl": 0.20247807016130537, "learning_rate": 7.579357325133208e-08, "loss": 0.001012386055663228, "num_tokens": 128696200.0, "reward": 2.28369140625, "reward_std": 0.5395830869674683, "rewards/code_complexity_reward/mean": 0.8941406011581421, "rewards/code_complexity_reward/std": 0.1616174429655075, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 816, "step_time": 52.16953933611512 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 475.0, "completions/mean_length": 119.55859375, "completions/mean_terminated_length": 118.79060363769531, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21703825378790498, "epoch": 0.9315849486887116, "frac_reward_zero_std": 0.4375, "grad_norm": 0.0474821962416172, "kl": 0.20639786985702813, "learning_rate": 7.338027235114759e-08, "loss": 0.0010316893458366394, "num_tokens": 128824034.0, "reward": 2.313720941543579, "reward_std": 0.5698933005332947, "rewards/code_complexity_reward/mean": 0.885546863079071, "rewards/code_complexity_reward/std": 0.17518816888332367, "rewards/code_execution_reward/mean": 0.345703125, "rewards/code_execution_reward/std": 0.4760620892047882, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.025282343849539757, "step": 817, "step_time": 50.49760962277651 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 120.4765625, "completions/mean_terminated_length": 119.71037292480469, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.208693124121055, "epoch": 0.9327251995438997, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04335222393274307, "kl": 0.1895402893424034, "learning_rate": 7.100544363921324e-08, "loss": 0.0009477338753640652, "num_tokens": 128954890.0, "reward": 2.230517625808716, "reward_std": 0.5309433341026306, "rewards/code_complexity_reward/mean": 0.8851562142372131, "rewards/code_complexity_reward/std": 0.16999588906764984, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.033528104424476624, "step": 818, "step_time": 58.965051799081266 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 114.51171875, "completions/mean_terminated_length": 113.73385620117188, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.21395123470574617, "epoch": 0.9338654503990877, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.27056264877319336, "kl": 0.74976360949222, "learning_rate": 6.866912476661075e-08, "loss": 0.0037483866326510906, "num_tokens": 129084032.0, "reward": 2.3023927211761475, "reward_std": 0.5659473538398743, "rewards/code_complexity_reward/mean": 0.8916015625, "rewards/code_complexity_reward/std": 0.171438530087471, "rewards/code_execution_reward/mean": 0.330078125, "rewards/code_execution_reward/std": 0.47070086002349854, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.028498241677880287, "step": 819, "step_time": 64.79336120840162 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 441.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 112.302734375, "completions/mean_terminated_length": 112.302734375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21585111366584897, "epoch": 0.935005701254276, "frac_reward_zero_std": 0.578125, "grad_norm": 0.03729502484202385, "kl": 0.1973567612003535, "learning_rate": 6.637135277387713e-08, "loss": 0.0009869406931102276, "num_tokens": 129207859.0, "reward": 2.3057618141174316, "reward_std": 0.5215427875518799, "rewards/code_complexity_reward/mean": 0.9032226800918579, "rewards/code_complexity_reward/std": 0.13213719427585602, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.03304819017648697, "step": 820, "step_time": 44.37103540543467 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 420.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 115.546875, "completions/mean_terminated_length": 115.546875, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.20809203758835793, "epoch": 0.936145952109464, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.0448678620159626, "kl": 0.21034562145359814, "learning_rate": 6.411216409041965e-08, "loss": 0.0010518691269680858, "num_tokens": 129336879.0, "reward": 2.3113770484924316, "reward_std": 0.5364217162132263, "rewards/code_complexity_reward/mean": 0.8995116949081421, "rewards/code_complexity_reward/std": 0.1419212371110916, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.04288214072585106, "step": 821, "step_time": 47.33945519011468 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 327.0, "completions/mean_length": 114.6015625, "completions/mean_terminated_length": 113.8238754272461, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.21928279008716345, "epoch": 0.9372862029646523, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.04188251867890358, "kl": 0.21783165214583278, "learning_rate": 6.189159453393573e-08, "loss": 0.0010891328565776348, "num_tokens": 129464395.0, "reward": 2.276660442352295, "reward_std": 0.5396551489830017, "rewards/code_complexity_reward/mean": 0.8960937261581421, "rewards/code_complexity_reward/std": 0.15899135172367096, "rewards/code_execution_reward/mean": 0.298828125, "rewards/code_execution_reward/std": 0.45819199085235596, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.037245213985443115, "step": 822, "step_time": 49.7325205700472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 119.611328125, "completions/mean_terminated_length": 119.611328125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.21985059441067278, "epoch": 0.9384264538198404, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.04441162943840027, "kl": 0.1918999047484249, "learning_rate": 5.970967930984728e-08, "loss": 0.0009595233132131398, "num_tokens": 129594052.0, "reward": 2.20751953125, "reward_std": 0.48925548791885376, "rewards/code_complexity_reward/mean": 0.89697265625, "rewards/code_complexity_reward/std": 0.1546589434146881, "rewards/code_execution_reward/mean": 0.22265625, "rewards/code_execution_reward/std": 0.41643625497817993, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.03121940791606903, "step": 823, "step_time": 61.271230657584965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 500.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 115.6953125, "completions/mean_terminated_length": 115.6953125, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22241086862049997, "epoch": 0.9395667046750285, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04690289497375488, "kl": 0.18858550442382693, "learning_rate": 5.756645301074088e-08, "loss": 0.0009426990291103721, "num_tokens": 129722836.0, "reward": 2.2433595657348633, "reward_std": 0.514386773109436, "rewards/code_complexity_reward/mean": 0.8957030773162842, "rewards/code_complexity_reward/std": 0.1466224193572998, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.03304819017648697, "step": 824, "step_time": 60.29175863135606 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 477.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 119.638671875, "completions/mean_terminated_length": 119.638671875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21182535821571946, "epoch": 0.9407069555302167, "frac_reward_zero_std": 0.5, "grad_norm": 0.03816008195281029, "kl": 0.18762191815767437, "learning_rate": 5.546194961581958e-08, "loss": 0.0009380167466588318, "num_tokens": 129852903.0, "reward": 2.2758302688598633, "reward_std": 0.5006850361824036, "rewards/code_complexity_reward/mean": 0.9013671875, "rewards/code_complexity_reward/std": 0.1278226673603058, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03250797092914581, "step": 825, "step_time": 54.82069675065577 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 113.873046875, "completions/mean_terminated_length": 113.873046875, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.2119212921243161, "epoch": 0.9418472063854048, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04462358355522156, "kl": 0.2085663704201579, "learning_rate": 5.3396202490365586e-08, "loss": 0.0010429518297314644, "num_tokens": 129980846.0, "reward": 2.2647461891174316, "reward_std": 0.5250336527824402, "rewards/code_complexity_reward/mean": 0.89404296875, "rewards/code_complexity_reward/std": 0.1454416960477829, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.03879402577877045, "step": 826, "step_time": 41.77487935870886 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 429.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 117.04296875, "completions/mean_terminated_length": 117.04296875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.20249626855365932, "epoch": 0.942987457240593, "frac_reward_zero_std": 0.484375, "grad_norm": 0.041226793080568314, "kl": 0.21137657179497182, "learning_rate": 5.136924438520902e-08, "loss": 0.0010566046694293618, "num_tokens": 130107544.0, "reward": 2.2870607376098633, "reward_std": 0.5426455140113831, "rewards/code_complexity_reward/mean": 0.8831053972244263, "rewards/code_complexity_reward/std": 0.15892472863197327, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499267578125, "rewards/xmlcount_reward_func/std": 0.009549576789140701, "step": 827, "step_time": 45.07807275373489 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 467.0, "completions/mean_length": 122.103515625, "completions/mean_terminated_length": 119.80550384521484, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21754176635295153, "epoch": 0.9441277080957811, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04006800800561905, "kl": 0.18695024761836976, "learning_rate": 4.9381107436211607e-08, "loss": 0.0009348994353786111, "num_tokens": 130237365.0, "reward": 2.3290529251098633, "reward_std": 0.5320473313331604, "rewards/code_complexity_reward/mean": 0.90576171875, "rewards/code_complexity_reward/std": 0.1314193308353424, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.03842822089791298, "step": 828, "step_time": 49.74653651099652 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 443.0, "completions/mean_length": 115.5625, "completions/mean_terminated_length": 114.78668975830078, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2124454965814948, "epoch": 0.9452679589509693, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04147333279252052, "kl": 0.20011256425641477, "learning_rate": 4.743182316375439e-08, "loss": 0.0010006376542150974, "num_tokens": 130365281.0, "reward": 2.2791504859924316, "reward_std": 0.5402747988700867, "rewards/code_complexity_reward/mean": 0.8907226324081421, "rewards/code_complexity_reward/std": 0.16159328818321228, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.02642802521586418, "step": 829, "step_time": 60.009542260318995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 118.166015625, "completions/mean_terminated_length": 115.06495666503906, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.22023609722964466, "epoch": 0.9464082098061574, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.04695567861199379, "kl": 0.1977868340909481, "learning_rate": 4.55214224722389e-08, "loss": 0.0009888706263154745, "num_tokens": 130494170.0, "reward": 2.259277582168579, "reward_std": 0.5602469444274902, "rewards/code_complexity_reward/mean": 0.8888671398162842, "rewards/code_complexity_reward/std": 0.1764320284128189, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03651979938149452, "step": 830, "step_time": 130.050562588498 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 119.263671875, "completions/mean_terminated_length": 117.7235336303711, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.20727725536562502, "epoch": 0.9475484606613455, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.043378058820962906, "kl": 0.20506394398398697, "learning_rate": 4.364993564959841e-08, "loss": 0.0010255139786750078, "num_tokens": 130621453.0, "reward": 2.2550294399261475, "reward_std": 0.5566163659095764, "rewards/code_complexity_reward/mean": 0.8841797113418579, "rewards/code_complexity_reward/std": 0.17913061380386353, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.03438635915517807, "step": 831, "step_time": 57.80455730203539 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 477.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 122.41796875, "completions/mean_terminated_length": 122.41796875, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.21161719784140587, "epoch": 0.9486887115165337, "frac_reward_zero_std": 0.4375, "grad_norm": 0.045965030789375305, "kl": 0.20860415394417942, "learning_rate": 4.1817392366815535e-08, "loss": 0.001043135765939951, "num_tokens": 130753591.0, "reward": 2.2332520484924316, "reward_std": 0.5467516779899597, "rewards/code_complexity_reward/mean": 0.8824218511581421, "rewards/code_complexity_reward/std": 0.17770440876483917, "rewards/code_execution_reward/mean": 0.271484375, "rewards/code_execution_reward/std": 0.44516023993492126, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.04148911312222481, "step": 832, "step_time": 49.650673107244074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 397.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 115.173828125, "completions/mean_terminated_length": 115.173828125, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21713033225387335, "epoch": 0.9498289623717218, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.041540756821632385, "kl": 0.19731544703245163, "learning_rate": 4.002382167745428e-08, "loss": 0.0009865115862339735, "num_tokens": 130879808.0, "reward": 2.2669923305511475, "reward_std": 0.5102322101593018, "rewards/code_complexity_reward/mean": 0.8969725966453552, "rewards/code_complexity_reward/std": 0.138751819729805, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.015571397729218006, "step": 833, "step_time": 48.77968459483236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 461.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 118.51953125, "completions/mean_terminated_length": 118.51953125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.21748150209896266, "epoch": 0.95096921322691, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04417231306433678, "kl": 0.1930065369233489, "learning_rate": 3.8269252017197056e-08, "loss": 0.0009650542051531374, "num_tokens": 131007606.0, "reward": 2.244678020477295, "reward_std": 0.5224413871765137, "rewards/code_complexity_reward/mean": 0.8953124284744263, "rewards/code_complexity_reward/std": 0.1565200686454773, "rewards/code_execution_reward/mean": 0.265625, "rewards/code_execution_reward/std": 0.44209739565849304, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.030623575672507286, "step": 834, "step_time": 46.0270975753665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 113.044921875, "completions/mean_terminated_length": 111.48040008544922, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21840074728243053, "epoch": 0.9521094640820981, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.04381146281957626, "kl": 0.1994408038444817, "learning_rate": 3.6553711203395906e-08, "loss": 0.000997209339402616, "num_tokens": 131133773.0, "reward": 2.2584962844848633, "reward_std": 0.5140597820281982, "rewards/code_complexity_reward/mean": 0.9032226204872131, "rewards/code_complexity_reward/std": 0.14077769219875336, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03968288004398346, "step": 835, "step_time": 51.750558748841286 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 466.0, "completions/mean_length": 111.06640625, "completions/mean_terminated_length": 110.28179931640625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21344944299198687, "epoch": 0.9532497149372862, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04550051689147949, "kl": 0.2083749327575788, "learning_rate": 3.487722643463032e-08, "loss": 0.0010421820916235447, "num_tokens": 131259079.0, "reward": 2.266650676727295, "reward_std": 0.5048200488090515, "rewards/code_complexity_reward/mean": 0.9076172113418579, "rewards/code_complexity_reward/std": 0.13410784304141998, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4912109375, "rewards/code_syntax_reward/std": 0.06577029824256897, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498291015625, "rewards/xmlcount_reward_func/std": 0.024042511358857155, "step": 836, "step_time": 59.09153588116169 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 111.93359375, "completions/mean_terminated_length": 111.15068054199219, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2110265197698027, "epoch": 0.9543899657924744, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.045263830572366714, "kl": 0.19898773485329002, "learning_rate": 3.323982429027567e-08, "loss": 0.000994884641841054, "num_tokens": 131383897.0, "reward": 2.3666017055511475, "reward_std": 0.5422783493995667, "rewards/code_complexity_reward/mean": 0.90234375, "rewards/code_complexity_reward/std": 0.1335001289844513, "rewards/code_execution_reward/mean": 0.373046875, "rewards/code_execution_reward/std": 0.48408737778663635, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02059757336974144, "step": 837, "step_time": 52.50911260023713 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 509.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 114.890625, "completions/mean_terminated_length": 114.890625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.2183029786683619, "epoch": 0.9555302166476625, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.040942542254924774, "kl": 0.1991866584867239, "learning_rate": 3.164153073008297e-08, "loss": 0.0009959996677935123, "num_tokens": 131511293.0, "reward": 2.273730754852295, "reward_std": 0.5310891270637512, "rewards/code_complexity_reward/mean": 0.8958984017372131, "rewards/code_complexity_reward/std": 0.15254774689674377, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 838, "step_time": 58.06075564958155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 498.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 120.00390625, "completions/mean_terminated_length": 120.00390625, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.21193064702674747, "epoch": 0.9566704675028507, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04262998700141907, "kl": 0.18331592285539955, "learning_rate": 3.0082371093766435e-08, "loss": 0.0009166357922367752, "num_tokens": 131640943.0, "reward": 2.2616212368011475, "reward_std": 0.521003782749176, "rewards/code_complexity_reward/mean": 0.891406238079071, "rewards/code_complexity_reward/std": 0.14802804589271545, "rewards/code_execution_reward/mean": 0.28515625, "rewards/code_execution_reward/std": 0.45193037390708923, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09921874850988388, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03480498120188713, "step": 839, "step_time": 55.418136943131685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 317.0, "completions/max_terminated_length": 317.0, "completions/mean_length": 110.64453125, "completions/mean_terminated_length": 110.64453125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.21591261913999915, "epoch": 0.9578107183580388, "frac_reward_zero_std": 0.40625, "grad_norm": 0.04986517131328583, "kl": 0.22112437966279685, "learning_rate": 2.856237010060242e-08, "loss": 0.0011056829243898392, "num_tokens": 131766281.0, "reward": 2.3351075649261475, "reward_std": 0.5763367414474487, "rewards/code_complexity_reward/mean": 0.895214855670929, "rewards/code_complexity_reward/std": 0.16854259371757507, "rewards/code_execution_reward/mean": 0.359375, "rewards/code_execution_reward/std": 0.48028653860092163, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03250797092914581, "step": 840, "step_time": 63.413751201704144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 114.04296875, "completions/mean_terminated_length": 114.04296875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2250844796653837, "epoch": 0.9589509692132269, "frac_reward_zero_std": 0.421875, "grad_norm": 0.05202499404549599, "kl": 0.20183859497774392, "learning_rate": 2.708155184903666e-08, "loss": 0.001009168103337288, "num_tokens": 131892667.0, "reward": 2.233691453933716, "reward_std": 0.5266737341880798, "rewards/code_complexity_reward/mean": 0.8919922113418579, "rewards/code_complexity_reward/std": 0.16297243535518646, "rewards/code_execution_reward/mean": 0.2578125, "rewards/code_execution_reward/std": 0.43785804510116577, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.031092895194888115, "step": 841, "step_time": 41.87927342765033 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 445.0, "completions/max_terminated_length": 445.0, "completions/mean_length": 110.42578125, "completions/mean_terminated_length": 110.42578125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2141903480514884, "epoch": 0.9600912200684151, "frac_reward_zero_std": 0.5625, "grad_norm": 0.0382932648062706, "kl": 0.19997890340164304, "learning_rate": 2.5639939816302917e-08, "loss": 0.0010001000482589006, "num_tokens": 132016745.0, "reward": 2.253467082977295, "reward_std": 0.504431426525116, "rewards/code_complexity_reward/mean": 0.9029296636581421, "rewards/code_complexity_reward/std": 0.1435883790254593, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.027560751885175705, "step": 842, "step_time": 53.76438216678798 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 431.0, "completions/mean_length": 117.26953125, "completions/mean_terminated_length": 116.49706268310547, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.21725277556106448, "epoch": 0.9612314709236032, "frac_reward_zero_std": 0.484375, "grad_norm": 0.044982921332120895, "kl": 0.19409097079187632, "learning_rate": 2.4237556858050794e-08, "loss": 0.000970620836596936, "num_tokens": 132143871.0, "reward": 2.2779297828674316, "reward_std": 0.5631141662597656, "rewards/code_complexity_reward/mean": 0.8833984136581421, "rewards/code_complexity_reward/std": 0.17998632788658142, "rewards/code_execution_reward/mean": 0.314453125, "rewards/code_execution_reward/std": 0.4647517800331116, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.028043000027537346, "step": 843, "step_time": 57.77219749800861 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 119.154296875, "completions/mean_terminated_length": 118.3855209350586, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2161537220235914, "epoch": 0.9623717217787914, "frac_reward_zero_std": 0.40625, "grad_norm": 0.05072968453168869, "kl": 0.20220713946036994, "learning_rate": 2.2874425207982386e-08, "loss": 0.0010108675342053175, "num_tokens": 132273990.0, "reward": 2.263427734375, "reward_std": 0.5568108558654785, "rewards/code_complexity_reward/mean": 0.89208984375, "rewards/code_complexity_reward/std": 0.17753179371356964, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.021246958523988724, "step": 844, "step_time": 52.77197929285467 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 477.0, "completions/mean_length": 115.5234375, "completions/mean_terminated_length": 114.74755096435547, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21993868052959442, "epoch": 0.9635119726339795, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.05121326819062233, "kl": 0.19388316106051207, "learning_rate": 2.15505664775012e-08, "loss": 0.0009692635503597558, "num_tokens": 132400886.0, "reward": 2.286376953125, "reward_std": 0.5604035258293152, "rewards/code_complexity_reward/mean": 0.8928711414337158, "rewards/code_complexity_reward/std": 0.1749558448791504, "rewards/code_execution_reward/mean": 0.3125, "rewards/code_execution_reward/std": 0.4639657139778137, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.025244520977139473, "step": 845, "step_time": 54.20384260173887 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 409.0, "completions/max_terminated_length": 409.0, "completions/mean_length": 115.8359375, "completions/mean_terminated_length": 115.8359375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2237892218399793, "epoch": 0.9646522234891676, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.047599174082279205, "kl": 0.20076306769624352, "learning_rate": 2.026600165536824e-08, "loss": 0.0010035361628979445, "num_tokens": 132528050.0, "reward": 2.291796922683716, "reward_std": 0.5117897391319275, "rewards/code_complexity_reward/mean": 0.9056640267372131, "rewards/code_complexity_reward/std": 0.12953004240989685, "rewards/code_execution_reward/mean": 0.296875, "rewards/code_execution_reward/std": 0.45732781291007996, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.03484956547617912, "step": 846, "step_time": 52.167921563610435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 353.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 110.599609375, "completions/mean_terminated_length": 110.599609375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.21085519320331514, "epoch": 0.9657924743443558, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.04354257136583328, "kl": 0.21503835311159492, "learning_rate": 1.9020751107370894e-08, "loss": 0.0010750304209068418, "num_tokens": 132651233.0, "reward": 2.2860350608825684, "reward_std": 0.5061343908309937, "rewards/code_complexity_reward/mean": 0.9054687023162842, "rewards/code_complexity_reward/std": 0.1280188262462616, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.028089813888072968, "step": 847, "step_time": 37.50479530822486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 444.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 117.306640625, "completions/mean_terminated_length": 117.306640625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21317138709127903, "epoch": 0.9669327251995439, "frac_reward_zero_std": 0.46875, "grad_norm": 0.04208116605877876, "kl": 0.21487501449882984, "learning_rate": 1.7814834575997364e-08, "loss": 0.0010747069027274847, "num_tokens": 132780362.0, "reward": 2.241748332977295, "reward_std": 0.5249396562576294, "rewards/code_complexity_reward/mean": 0.889453113079071, "rewards/code_complexity_reward/std": 0.15255840122699738, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.494873046875, "rewards/xmlcount_reward_func/std": 0.04358936473727226, "step": 848, "step_time": 58.384569615125656 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 387.0, "completions/max_terminated_length": 387.0, "completions/mean_length": 111.09375, "completions/mean_terminated_length": 111.09375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.21358874742873013, "epoch": 0.9680729760547321, "frac_reward_zero_std": 0.46875, "grad_norm": 0.09370254725217819, "kl": 0.34073584631551057, "learning_rate": 1.664827118012663e-08, "loss": 0.0017055298667401075, "num_tokens": 132905434.0, "reward": 2.3063478469848633, "reward_std": 0.517406165599823, "rewards/code_complexity_reward/mean": 0.9064452648162842, "rewards/code_complexity_reward/std": 0.12503407895565033, "rewards/code_execution_reward/mean": 0.310546875, "rewards/code_execution_reward/std": 0.46317005157470703, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.02697930857539177, "step": 849, "step_time": 48.49731491971761 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 118.125, "completions/mean_terminated_length": 115.02362060546875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2225821795873344, "epoch": 0.9692132269099202, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.04526621475815773, "kl": 0.20284870197065175, "learning_rate": 1.5521079414723695e-08, "loss": 0.001014315290376544, "num_tokens": 133034378.0, "reward": 2.226806640625, "reward_std": 0.5422424077987671, "rewards/code_complexity_reward/mean": 0.8824218511581421, "rewards/code_complexity_reward/std": 0.18165254592895508, "rewards/code_execution_reward/mean": 0.263671875, "rewards/code_execution_reward/std": 0.4410543739795685, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.027517380192875862, "step": 850, "step_time": 56.51227837614715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 116.236328125, "completions/mean_terminated_length": 115.46183776855469, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.21715749241411686, "epoch": 0.9703534777651083, "frac_reward_zero_std": 0.5234375, "grad_norm": 0.045916587114334106, "kl": 0.1886747629614547, "learning_rate": 1.4433277150545932e-08, "loss": 0.0009436316322535276, "num_tokens": 133163651.0, "reward": 2.235351800918579, "reward_std": 0.515434741973877, "rewards/code_complexity_reward/mean": 0.8958008289337158, "rewards/code_complexity_reward/std": 0.15828996896743774, "rewards/code_execution_reward/mean": 0.25390625, "rewards/code_execution_reward/std": 0.43567025661468506, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 851, "step_time": 50.51276112999767 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 316.0, "completions/mean_length": 109.0234375, "completions/mean_terminated_length": 108.23483276367188, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.20451017213054, "epoch": 0.9714937286202965, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.038792144507169724, "kl": 0.22877675923518836, "learning_rate": 1.3384881633861923e-08, "loss": 0.0011438047513365746, "num_tokens": 133287467.0, "reward": 2.3270998001098633, "reward_std": 0.5403817892074585, "rewards/code_complexity_reward/mean": 0.9049804210662842, "rewards/code_complexity_reward/std": 0.13834215700626373, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.03607473522424698, "step": 852, "step_time": 50.17305930983275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 474.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 118.708984375, "completions/mean_terminated_length": 118.708984375, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.22754912613891065, "epoch": 0.9726339794754846, "frac_reward_zero_std": 0.421875, "grad_norm": 0.04196017608046532, "kl": 0.18907271861098707, "learning_rate": 1.2375909486175008e-08, "loss": 0.0009452051017433405, "num_tokens": 133417694.0, "reward": 2.249756097793579, "reward_std": 0.527574896812439, "rewards/code_complexity_reward/mean": 0.893847644329071, "rewards/code_complexity_reward/std": 0.1589047610759735, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.495361328125, "rewards/xmlcount_reward_func/std": 0.039215847849845886, "step": 853, "step_time": 53.947559972293675 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 511.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 111.21875, "completions/mean_terminated_length": 111.21875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21064179460518062, "epoch": 0.9737742303306728, "frac_reward_zero_std": 0.515625, "grad_norm": 0.047649040818214417, "kl": 0.20129295997321606, "learning_rate": 1.1406376703962384e-08, "loss": 0.0010063945082947612, "num_tokens": 133542210.0, "reward": 2.256103515625, "reward_std": 0.5032654404640198, "rewards/code_complexity_reward/mean": 0.907519519329071, "rewards/code_complexity_reward/std": 0.13910919427871704, "rewards/code_execution_reward/mean": 0.259765625, "rewards/code_execution_reward/std": 0.4389347732067108, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498779296875, "rewards/xmlcount_reward_func/std": 0.019900046288967133, "step": 854, "step_time": 48.145288683474064 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 117.4140625, "completions/mean_terminated_length": 116.64187622070312, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.21884385985322297, "epoch": 0.9749144811858609, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.048524171113967896, "kl": 0.25599179056007415, "learning_rate": 1.0476298658420036e-08, "loss": 0.0012796975206583738, "num_tokens": 133669998.0, "reward": 2.2699217796325684, "reward_std": 0.4993499517440796, "rewards/code_complexity_reward/mean": 0.9010741710662842, "rewards/code_complexity_reward/std": 0.13228319585323334, "rewards/code_execution_reward/mean": 0.279296875, "rewards/code_execution_reward/std": 0.44909247756004333, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49658203125, "rewards/xmlcount_reward_func/std": 0.03567269444465637, "step": 855, "step_time": 50.553934891708195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 415.0, "completions/mean_length": 115.166015625, "completions/mean_terminated_length": 112.82711791992188, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.21475373348221183, "epoch": 0.976054732041049, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.045696936547756195, "kl": 0.19360765046440065, "learning_rate": 9.58569009521959e-09, "loss": 0.0009681619703769684, "num_tokens": 133800263.0, "reward": 2.2862305641174316, "reward_std": 0.5306099057197571, "rewards/code_complexity_reward/mean": 0.9002929925918579, "rewards/code_complexity_reward/std": 0.14716801047325134, "rewards/code_execution_reward/mean": 0.30078125, "rewards/code_execution_reward/std": 0.45904624462127686, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.031035220250487328, "step": 856, "step_time": 69.81647194363177 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 493.0, "completions/max_terminated_length": 493.0, "completions/mean_length": 114.654296875, "completions/mean_terminated_length": 114.654296875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.22699987050145864, "epoch": 0.9771949828962372, "frac_reward_zero_std": 0.4609375, "grad_norm": 0.04479582980275154, "kl": 0.1877583267632872, "learning_rate": 8.73456513427462e-09, "loss": 0.0009387535974383354, "num_tokens": 133928622.0, "reward": 2.212451457977295, "reward_std": 0.5233328342437744, "rewards/code_complexity_reward/mean": 0.8929687738418579, "rewards/code_complexity_reward/std": 0.1751135289669037, "rewards/code_execution_reward/mean": 0.236328125, "rewards/code_execution_reward/std": 0.42524150013923645, "rewards/code_syntax_reward/mean": 0.4833984375, "rewards/code_syntax_reward/std": 0.08967091888189316, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.499755859375, "rewards/xmlcount_reward_func/std": 0.005524271633476019, "step": 857, "step_time": 49.807878320105374 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 490.0, "completions/max_terminated_length": 490.0, "completions/mean_length": 116.806640625, "completions/mean_terminated_length": 116.806640625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2224892999511212, "epoch": 0.9783352337514253, "frac_reward_zero_std": 0.390625, "grad_norm": 0.04882029816508293, "kl": 0.20581728836987168, "learning_rate": 7.922937269516096e-09, "loss": 0.001029058126732707, "num_tokens": 134054599.0, "reward": 2.2936525344848633, "reward_std": 0.5616592764854431, "rewards/code_complexity_reward/mean": 0.8888671398162842, "rewards/code_complexity_reward/std": 0.16641460359096527, "rewards/code_execution_reward/mean": 0.326171875, "rewards/code_execution_reward/std": 0.4692695140838623, "rewards/code_syntax_reward/mean": 0.4853515625, "rewards/code_syntax_reward/std": 0.08440115302801132, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.49365234375, "rewards/xmlcount_reward_func/std": 0.0496685691177845, "step": 858, "step_time": 50.03644905332476 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.005859375, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 117.642578125, "completions/mean_terminated_length": 115.31827545166016, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21480377321131527, "epoch": 0.9794754846066135, "frac_reward_zero_std": 0.5390625, "grad_norm": 0.03679094836115837, "kl": 0.23072400665841997, "learning_rate": 7.150819368679229e-09, "loss": 0.0011537820100784302, "num_tokens": 134182844.0, "reward": 2.2956056594848633, "reward_std": 0.5573036670684814, "rewards/code_complexity_reward/mean": 0.8968750238418579, "rewards/code_complexity_reward/std": 0.1564028114080429, "rewards/code_execution_reward/mean": 0.31640625, "rewards/code_execution_reward/std": 0.46552830934524536, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639660965651274, "rewards/xmlcount_reward_func/mean": 0.49560546875, "rewards/xmlcount_reward_func/std": 0.04114582762122154, "step": 859, "step_time": 55.14065580628812 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 419.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 113.84765625, "completions/mean_terminated_length": 113.84765625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2120169708505273, "epoch": 0.9806157354618016, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04322896897792816, "kl": 0.1968265169998631, "learning_rate": 6.418223673099466e-09, "loss": 0.000983987469226122, "num_tokens": 134308682.0, "reward": 2.302295207977295, "reward_std": 0.5940123796463013, "rewards/code_complexity_reward/mean": 0.8851562142372131, "rewards/code_complexity_reward/std": 0.1898716539144516, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4794921875, "rewards/code_syntax_reward/std": 0.09926015883684158, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.018207494169473648, "step": 860, "step_time": 48.79411215428263 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 117.05078125, "completions/mean_terminated_length": 116.27788543701172, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.2132799623068422, "epoch": 0.9817559863169898, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04577140510082245, "kl": 0.2047085758531466, "learning_rate": 5.725161797517087e-09, "loss": 0.0010235033696517348, "num_tokens": 134436368.0, "reward": 2.2945313453674316, "reward_std": 0.510342538356781, "rewards/code_complexity_reward/mean": 0.8999999761581421, "rewards/code_complexity_reward/std": 0.129465252161026, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4990234375, "rewards/xmlcount_reward_func/std": 0.015609703958034515, "step": 861, "step_time": 50.64652017876506 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 506.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 122.92578125, "completions/mean_terminated_length": 122.92578125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.21308421972207725, "epoch": 0.9828962371721779, "frac_reward_zero_std": 0.453125, "grad_norm": 0.04639558866620064, "kl": 0.19089061638806015, "learning_rate": 5.071644729895408e-09, "loss": 0.0009543277556076646, "num_tokens": 134568366.0, "reward": 2.262988328933716, "reward_std": 0.5540809035301208, "rewards/code_complexity_reward/mean": 0.886035144329071, "rewards/code_complexity_reward/std": 0.16849541664123535, "rewards/code_execution_reward/mean": 0.29296875, "rewards/code_execution_reward/std": 0.455569326877594, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.02327641472220421, "step": 862, "step_time": 67.55095878150314 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 407.0, "completions/mean_length": 115.82421875, "completions/mean_terminated_length": 114.27059173583984, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2061112669762224, "epoch": 0.984036488027366, "frac_reward_zero_std": 0.5546875, "grad_norm": 0.03889443725347519, "kl": 0.1903388889040798, "learning_rate": 4.4576828312442586e-09, "loss": 0.0009516997961327434, "num_tokens": 134695772.0, "reward": 2.3394532203674316, "reward_std": 0.5129092931747437, "rewards/code_complexity_reward/mean": 0.909863293170929, "rewards/code_complexity_reward/std": 0.10628663748502731, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.49609375, "rewards/code_syntax_reward/std": 0.04406425356864929, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49755859375, "rewards/xmlcount_reward_func/std": 0.028089813888072968, "step": 863, "step_time": 52.15768493898213 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 510.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 121.16796875, "completions/mean_terminated_length": 121.16796875, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.21476097265258431, "epoch": 0.9851767388825542, "frac_reward_zero_std": 0.4296875, "grad_norm": 0.04715282469987869, "kl": 0.21085852477699518, "learning_rate": 3.8832858354567736e-09, "loss": 0.0010544309625402093, "num_tokens": 134826074.0, "reward": 2.217041015625, "reward_std": 0.5454345941543579, "rewards/code_complexity_reward/mean": 0.8792968988418579, "rewards/code_complexity_reward/std": 0.18048393726348877, "rewards/code_execution_reward/mean": 0.26171875, "rewards/code_execution_reward/std": 0.44000017642974854, "rewards/code_syntax_reward/mean": 0.482421875, "rewards/code_syntax_reward/std": 0.0921773687005043, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812851272523403, "rewards/xmlcount_reward_func/mean": 0.494384765625, "rewards/xmlcount_reward_func/std": 0.045587729662656784, "step": 864, "step_time": 49.68822178989649 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 116.8671875, "completions/mean_terminated_length": 116.09393310546875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21834377688355744, "epoch": 0.9863169897377423, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04607750475406647, "kl": 0.19288754765875638, "learning_rate": 3.348462849155909e-09, "loss": 0.0009642998338676989, "num_tokens": 134954622.0, "reward": 2.2582521438598633, "reward_std": 0.5195484161376953, "rewards/code_complexity_reward/mean": 0.8987303972244263, "rewards/code_complexity_reward/std": 0.1499946117401123, "rewards/code_execution_reward/mean": 0.2734375, "rewards/code_execution_reward/std": 0.4461594223976135, "rewards/code_syntax_reward/mean": 0.48828125, "rewards/code_syntax_reward/std": 0.07571818679571152, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497802734375, "rewards/xmlcount_reward_func/std": 0.021303100511431694, "step": 865, "step_time": 58.93940037954599 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 495.0, "completions/max_terminated_length": 495.0, "completions/mean_length": 116.826171875, "completions/mean_terminated_length": 116.826171875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.2203833742532879, "epoch": 0.9874572405929305, "frac_reward_zero_std": 0.4765625, "grad_norm": 0.0436788871884346, "kl": 0.1987122108694166, "learning_rate": 2.8532223515481683e-09, "loss": 0.0009936585556715727, "num_tokens": 135082865.0, "reward": 2.2640624046325684, "reward_std": 0.5361856818199158, "rewards/code_complexity_reward/mean": 0.8919922113418579, "rewards/code_complexity_reward/std": 0.16544529795646667, "rewards/code_execution_reward/mean": 0.287109375, "rewards/code_execution_reward/std": 0.45285552740097046, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.025862684473395348, "step": 866, "step_time": 52.309539534151554 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 303.0, "completions/max_terminated_length": 303.0, "completions/mean_length": 111.513671875, "completions/mean_terminated_length": 111.513671875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.21739643439650536, "epoch": 0.9885974914481186, "frac_reward_zero_std": 0.5078125, "grad_norm": 0.04077056422829628, "kl": 0.21516881370916963, "learning_rate": 2.3975721942903762e-09, "loss": 0.0010757026029750705, "num_tokens": 135209236.0, "reward": 2.250683546066284, "reward_std": 0.5200665593147278, "rewards/code_complexity_reward/mean": 0.8985351324081421, "rewards/code_complexity_reward/std": 0.1522267907857895, "rewards/code_execution_reward/mean": 0.26953125, "rewards/code_execution_reward/std": 0.44415023922920227, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09921875596046448, "rewards/reasoning_present_reward_func/std": 0.008812850341200829, "rewards/xmlcount_reward_func/mean": 0.49609375, "rewards/xmlcount_reward_func/std": 0.03647070750594139, "step": 867, "step_time": 34.10790188424289 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 400.0, "completions/mean_length": 112.359375, "completions/mean_terminated_length": 111.57730102539062, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.21389862522482872, "epoch": 0.9897377423033067, "frac_reward_zero_std": 0.484375, "grad_norm": 0.05911806598305702, "kl": 0.2688353101257235, "learning_rate": 1.9815196013650563e-09, "loss": 0.001344506163150072, "num_tokens": 135333688.0, "reward": 2.2804200649261475, "reward_std": 0.5142287611961365, "rewards/code_complexity_reward/mean": 0.9025390148162842, "rewards/code_complexity_reward/std": 0.13895541429519653, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.490234375, "rewards/code_syntax_reward/std": 0.06925903260707855, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496826171875, "rewards/xmlcount_reward_func/std": 0.02960825525224209, "step": 868, "step_time": 50.716291734948754 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 419.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 115.822265625, "completions/mean_terminated_length": 115.822265625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21037229103967547, "epoch": 0.9908779931584949, "frac_reward_zero_std": 0.484375, "grad_norm": 0.04051118344068527, "kl": 0.20270752301439643, "learning_rate": 1.6050711689663544e-09, "loss": 0.0010136780329048634, "num_tokens": 135460133.0, "reward": 2.281054973602295, "reward_std": 0.5475639700889587, "rewards/code_complexity_reward/mean": 0.8931640386581421, "rewards/code_complexity_reward/std": 0.15852224826812744, "rewards/code_execution_reward/mean": 0.302734375, "rewards/code_execution_reward/std": 0.45989060401916504, "rewards/code_syntax_reward/mean": 0.4873046875, "rewards/code_syntax_reward/std": 0.07873113453388214, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.498046875, "rewards/xmlcount_reward_func/std": 0.02701912261545658, "step": 869, "step_time": 42.015810766257346 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 117.228515625, "completions/mean_terminated_length": 116.45597076416016, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.21514453343115747, "epoch": 0.992018244013683, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.04121711105108261, "kl": 0.19469071528874338, "learning_rate": 1.2682328653940146e-09, "loss": 0.0009732992039062083, "num_tokens": 135585682.0, "reward": 2.26611328125, "reward_std": 0.5401870608329773, "rewards/code_complexity_reward/mean": 0.8902343511581421, "rewards/code_complexity_reward/std": 0.16224434971809387, "rewards/code_execution_reward/mean": 0.291015625, "rewards/code_execution_reward/std": 0.45467492938041687, "rewards/code_syntax_reward/mean": 0.486328125, "rewards/code_syntax_reward/std": 0.08162125200033188, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49853515625, "rewards/xmlcount_reward_func/std": 0.01742478273808956, "step": 870, "step_time": 49.54158889967948 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 404.0, "completions/mean_length": 110.712890625, "completions/mean_terminated_length": 109.9275894165039, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.21640713443048298, "epoch": 0.9931584948688712, "frac_reward_zero_std": 0.578125, "grad_norm": 0.03911013528704643, "kl": 0.21855483506806195, "learning_rate": 9.710100309603954e-10, "loss": 0.0010925468523055315, "num_tokens": 135710919.0, "reward": 2.3333985805511475, "reward_std": 0.5384849309921265, "rewards/code_complexity_reward/mean": 0.9078124761581421, "rewards/code_complexity_reward/std": 0.14452669024467468, "rewards/code_execution_reward/mean": 0.33984375, "rewards/code_execution_reward/std": 0.4741191864013672, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.4970703125, "rewards/xmlcount_reward_func/std": 0.028043000027537346, "step": 871, "step_time": 50.11259215604514 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 427.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 117.7890625, "completions/mean_terminated_length": 117.7890625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2150321628432721, "epoch": 0.9942987457240593, "frac_reward_zero_std": 0.40625, "grad_norm": 0.046198517084121704, "kl": 0.19415406370535493, "learning_rate": 7.134073779044293e-10, "loss": 0.0009709315490908921, "num_tokens": 135838555.0, "reward": 2.3036134243011475, "reward_std": 0.5042473077774048, "rewards/code_complexity_reward/mean": 0.90625, "rewards/code_complexity_reward/std": 0.11911282688379288, "rewards/code_execution_reward/mean": 0.3046875, "rewards/code_execution_reward/std": 0.4607250988483429, "rewards/code_syntax_reward/mean": 0.4931640625, "rewards/code_syntax_reward/std": 0.05811915174126625, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.49951171875, "rewards/xmlcount_reward_func/std": 0.007804851979017258, "step": 872, "step_time": 43.89454901497811 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 496.0, "completions/max_terminated_length": 496.0, "completions/mean_length": 118.5703125, "completions/mean_terminated_length": 118.5703125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.21323927189223468, "epoch": 0.9954389965792474, "frac_reward_zero_std": 0.53125, "grad_norm": 0.04069061949849129, "kl": 0.19403478188905865, "learning_rate": 4.954289903180698e-10, "loss": 0.000970054476056248, "num_tokens": 135967103.0, "reward": 2.3134279251098633, "reward_std": 0.5537270307540894, "rewards/code_complexity_reward/mean": 0.89111328125, "rewards/code_complexity_reward/std": 0.15563181042671204, "rewards/code_execution_reward/mean": 0.3359375, "rewards/code_execution_reward/std": 0.4727790653705597, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.02860700711607933, "step": 873, "step_time": 68.5761090433225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.0, "completions/max_terminated_length": 390.0, "completions/mean_length": 115.025390625, "completions/mean_terminated_length": 115.025390625, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.20336008653976023, "epoch": 0.9965792474344356, "frac_reward_zero_std": 0.4453125, "grad_norm": 0.0425407811999321, "kl": 0.21382639382500201, "learning_rate": 3.1707832408134354e-10, "loss": 0.0010687923058867455, "num_tokens": 136093476.0, "reward": 2.3217287063598633, "reward_std": 0.5222358107566833, "rewards/code_complexity_reward/mean": 0.9013671875, "rewards/code_complexity_reward/std": 0.129571333527565, "rewards/code_execution_reward/mean": 0.33203125, "rewards/code_execution_reward/std": 0.47140273451805115, "rewards/code_syntax_reward/mean": 0.4921875, "rewards/code_syntax_reward/std": 0.062070440500974655, "rewards/reasoning_present_reward_func/mean": 0.09980468451976776, "rewards/reasoning_present_reward_func/std": 0.0044194175861775875, "rewards/xmlcount_reward_func/mean": 0.496337890625, "rewards/xmlcount_reward_func/std": 0.039319269359111786, "step": 874, "step_time": 42.707073513418436 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00390625, "completions/max_length": 512.0, "completions/max_terminated_length": 458.0, "completions/mean_length": 115.49609375, "completions/mean_terminated_length": 113.9411849975586, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.199556466890499, "epoch": 0.9977194982896237, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04358580708503723, "kl": 0.19634421134833246, "learning_rate": 1.7835820680600634e-10, "loss": 0.000981607474386692, "num_tokens": 136220870.0, "reward": 2.3045411109924316, "reward_std": 0.5778931975364685, "rewards/code_complexity_reward/mean": 0.8849608898162842, "rewards/code_complexity_reward/std": 0.17410211265087128, "rewards/code_execution_reward/mean": 0.337890625, "rewards/code_execution_reward/std": 0.4734536409378052, "rewards/code_syntax_reward/mean": 0.484375, "rewards/code_syntax_reward/std": 0.08708140254020691, "rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "rewards/reasoning_present_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.497314453125, "rewards/xmlcount_reward_func/std": 0.03260336071252823, "step": 875, "step_time": 53.33960290905088 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.001953125, "completions/max_length": 512.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 114.27734375, "completions/mean_terminated_length": 113.4990234375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.2070969846099615, "epoch": 0.9988597491448119, "frac_reward_zero_std": 0.4921875, "grad_norm": 0.04219535365700722, "kl": 0.20320449664723128, "learning_rate": 7.927083779335487e-11, "loss": 0.001016006339341402, "num_tokens": 136347776.0, "reward": 2.2850587368011475, "reward_std": 0.5324879884719849, "rewards/code_complexity_reward/mean": 0.8944335579872131, "rewards/code_complexity_reward/std": 0.14782597124576569, "rewards/code_execution_reward/mean": 0.306640625, "rewards/code_execution_reward/std": 0.4615498185157776, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.099609375, "rewards/reasoning_present_reward_func/std": 0.006243881769478321, "rewards/xmlcount_reward_func/mean": 0.4951171875, "rewards/xmlcount_reward_func/std": 0.03957437351346016, "step": 876, "step_time": 49.541470520198345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 424.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 113.603515625, "completions/mean_terminated_length": 113.603515625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.20893901865929365, "epoch": 1.0, "frac_reward_zero_std": 0.5, "grad_norm": 0.043643876910209656, "kl": 0.22944669984281063, "learning_rate": 1.98177879973116e-11, "loss": 0.0011475281789898872, "num_tokens": 136473729.0, "reward": 2.2608888149261475, "reward_std": 0.5139729380607605, "rewards/code_complexity_reward/mean": 0.8951171636581421, "rewards/code_complexity_reward/std": 0.14539770781993866, "rewards/code_execution_reward/mean": 0.28125, "rewards/code_execution_reward/std": 0.45004892349243164, "rewards/code_syntax_reward/mean": 0.4892578125, "rewards/code_syntax_reward/std": 0.07256709784269333, "rewards/reasoning_present_reward_func/mean": 0.09941406548023224, "rewards/reasoning_present_reward_func/std": 0.007639661431312561, "rewards/xmlcount_reward_func/mean": 0.495849609375, "rewards/xmlcount_reward_func/std": 0.038484130054712296, "step": 877, "step_time": 44.80613525211811 }, { "epoch": 1.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.005, "eval_completions/max_length": 195.94, "eval_completions/max_terminated_length": 188.9, "eval_completions/mean_length": 118.3025, "eval_completions/mean_terminated_length": 116.56500045776367, "eval_completions/min_length": 74.36, "eval_completions/min_terminated_length": 74.36, "eval_entropy": 0.21839634716510772, "eval_frac_reward_zero_std": 0.49, "eval_kl": 0.2022037136554718, "eval_loss": 0.0010131248272955418, "eval_num_tokens": 136473729.0, "eval_reward": 2.215250108242035, "eval_reward_std": 0.35850373629480603, "eval_rewards/code_complexity_reward/mean": 0.894624981880188, "eval_rewards/code_complexity_reward/std": 0.08908241916447877, "eval_rewards/code_execution_reward/mean": 0.235, "eval_rewards/code_execution_reward/std": 0.2707787317037582, "eval_rewards/code_syntax_reward/mean": 0.48875, "eval_rewards/code_syntax_reward/std": 0.029377837479114533, "eval_rewards/reasoning_present_reward_func/mean": 0.10000000149011612, "eval_rewards/reasoning_present_reward_func/std": 0.0, "eval_rewards/xmlcount_reward_func/mean": 0.496875, "eval_rewards/xmlcount_reward_func/std": 0.00883883461356163, "eval_runtime": 415.0623, "eval_samples_per_second": 0.241, "eval_steps_per_second": 0.031, "step": 877 } ], "logging_steps": 1, "max_steps": 877, "num_input_tokens_seen": 136473729, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 1, "trial_name": null, "trial_params": null }