{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.37593984962406013, "eval_steps": 500, "global_step": 200, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 590.1875, "completions/clipped_ratio": 0.0, "completions/max_length": 757.0, "completions/max_terminated_length": 757.0, "completions/mean_length": 590.1875, "completions/mean_terminated_length": 590.1875, "completions/min_length": 359.0, "completions/min_terminated_length": 359.0, "epoch": 0.0018796992481203006, "frac_reward_zero_std": 0.0, "grad_norm": 0.04373836889863014, "kl": 0.0, "learning_rate": 0.0, "loss": 0.009791684336960316, "num_tokens": 16371.0, "reward": 4.735293388366699, "reward_std": 0.5943833589553833, "rewards/coordinate_accuracy_reward_func/mean": 0.7290432453155518, "rewards/coordinate_accuracy_reward_func/std": 0.08427873998880386, "rewards/graph_topology_reward_func/mean": 3.90625, "rewards/graph_topology_reward_func/std": 1.157853603363037, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 1 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 736.875, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1103.0, "completions/mean_length": 736.875, "completions/mean_terminated_length": 652.6666870117188, "completions/min_length": 422.0, "completions/min_terminated_length": 422.0, "epoch": 0.0037593984962406013, "frac_reward_zero_std": 0.0, "grad_norm": 0.09037133306264877, "kl": 0.0, "learning_rate": 2.0833333333333333e-07, "loss": 0.14494961500167847, "num_tokens": 35153.0, "reward": 4.714605808258057, "reward_std": 0.7861276865005493, "rewards/coordinate_accuracy_reward_func/mean": 0.7343171238899231, "rewards/coordinate_accuracy_reward_func/std": 0.1979934126138687, "rewards/graph_topology_reward_func/mean": 3.899038314819336, "rewards/graph_topology_reward_func/std": 1.3836090564727783, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 2 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1393.8125, "completions/clipped_ratio": 0.375, "completions/max_length": 2000.0, "completions/max_terminated_length": 1617.0, "completions/mean_length": 1393.8125, "completions/mean_terminated_length": 1030.0999755859375, "completions/min_length": 457.0, "completions/min_terminated_length": 457.0, "epoch": 0.005639097744360902, "frac_reward_zero_std": 0.0, "grad_norm": 0.2884621024131775, "kl": 0.0, "learning_rate": 4.1666666666666667e-07, "loss": 0.3705242872238159, "num_tokens": 65166.0, "reward": 3.840770721435547, "reward_std": 2.5260701179504395, "rewards/coordinate_accuracy_reward_func/mean": 0.544937252998352, "rewards/coordinate_accuracy_reward_func/std": 0.33048826456069946, "rewards/graph_topology_reward_func/mean": 3.2708334922790527, "rewards/graph_topology_reward_func/std": 1.9933918714523315, "rewards/xml_formatting_reward_func/mean": 0.02500000037252903, "rewards/xml_formatting_reward_func/std": 0.13416409492492676, "step": 3 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 914.3125, "completions/clipped_ratio": 0.25, "completions/max_length": 2000.0, "completions/max_terminated_length": 1139.0, "completions/mean_length": 914.3125, "completions/mean_terminated_length": 552.4166870117188, "completions/min_length": 299.0, "completions/min_terminated_length": 299.0, "epoch": 0.007518796992481203, "frac_reward_zero_std": 0.0, "grad_norm": 0.2184051126241684, "kl": 0.0, "learning_rate": 6.25e-07, "loss": 0.5314029455184937, "num_tokens": 86083.0, "reward": 3.120427131652832, "reward_std": 2.2966792583465576, "rewards/coordinate_accuracy_reward_func/mean": 0.5954271554946899, "rewards/coordinate_accuracy_reward_func/std": 0.355163037776947, "rewards/graph_topology_reward_func/mean": 2.5, "rewards/graph_topology_reward_func/std": 1.8348479270935059, "rewards/xml_formatting_reward_func/mean": 0.02500000037252903, "rewards/xml_formatting_reward_func/std": 0.13416409492492676, "step": 4 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 669.6875, "completions/clipped_ratio": 0.0, "completions/max_length": 1247.0, "completions/max_terminated_length": 1247.0, "completions/mean_length": 669.6875, "completions/mean_terminated_length": 669.6875, "completions/min_length": 302.0, "completions/min_terminated_length": 302.0, "epoch": 0.009398496240601503, "frac_reward_zero_std": 0.0, "grad_norm": 0.08891818672418594, "kl": 0.0, "learning_rate": 8.333333333333333e-07, "loss": 0.032497234642505646, "num_tokens": 103806.0, "reward": 5.364262580871582, "reward_std": 1.2523577213287354, "rewards/coordinate_accuracy_reward_func/mean": 0.739262580871582, "rewards/coordinate_accuracy_reward_func/std": 0.1981801688671112, "rewards/graph_topology_reward_func/mean": 4.525000095367432, "rewards/graph_topology_reward_func/std": 1.0853571891784668, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 5 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 604.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 982.0, "completions/mean_length": 604.0, "completions/mean_terminated_length": 510.933349609375, "completions/min_length": 305.0, "completions/min_terminated_length": 305.0, "epoch": 0.011278195488721804, "frac_reward_zero_std": 0.0, "grad_norm": 0.07929574698209763, "kl": 0.0, "learning_rate": 1.0416666666666667e-06, "loss": 0.19365747272968292, "num_tokens": 119054.0, "reward": 5.201510906219482, "reward_std": 0.9837310314178467, "rewards/coordinate_accuracy_reward_func/mean": 0.7452607750892639, "rewards/coordinate_accuracy_reward_func/std": 0.19906705617904663, "rewards/graph_topology_reward_func/mean": 4.375, "rewards/graph_topology_reward_func/std": 1.2583057880401611, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 6 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 991.8125, "completions/clipped_ratio": 0.3125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1462.0, "completions/mean_length": 991.8125, "completions/mean_terminated_length": 533.5454711914062, "completions/min_length": 240.0, "completions/min_terminated_length": 240.0, "epoch": 0.013157894736842105, "frac_reward_zero_std": 0.5, "grad_norm": 0.4627147316932678, "kl": 0.0, "learning_rate": 1.25e-06, "loss": 0.25529852509498596, "num_tokens": 141931.0, "reward": 3.627457618713379, "reward_std": 1.0764048099517822, "rewards/coordinate_accuracy_reward_func/mean": 0.527457594871521, "rewards/coordinate_accuracy_reward_func/std": 0.3703506588935852, "rewards/graph_topology_reward_func/mean": 3.09375, "rewards/graph_topology_reward_func/std": 2.2672946453094482, "rewards/xml_formatting_reward_func/mean": 0.006250001490116119, "rewards/xml_formatting_reward_func/std": 0.14361408352851868, "step": 7 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 572.375, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 653.0, "completions/mean_length": 572.375, "completions/mean_terminated_length": 477.20001220703125, "completions/min_length": 349.0, "completions/min_terminated_length": 349.0, "epoch": 0.015037593984962405, "frac_reward_zero_std": 0.0, "grad_norm": 0.00427014147862792, "kl": 0.0, "learning_rate": 1.4583333333333335e-06, "loss": -0.0018687100382521749, "num_tokens": 158657.0, "reward": 5.823794364929199, "reward_std": 0.050050895661115646, "rewards/coordinate_accuracy_reward_func/mean": 0.7237942218780518, "rewards/coordinate_accuracy_reward_func/std": 0.06562276929616928, "rewards/graph_topology_reward_func/mean": 5.0, "rewards/graph_topology_reward_func/std": 0.0, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 8 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 990.4375, "completions/clipped_ratio": 0.125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1370.0, "completions/mean_length": 990.4375, "completions/mean_terminated_length": 846.2142944335938, "completions/min_length": 537.0, "completions/min_terminated_length": 537.0, "epoch": 0.016917293233082706, "frac_reward_zero_std": 0.0, "grad_norm": 0.1499452441930771, "kl": 0.0, "learning_rate": 1.6666666666666667e-06, "loss": 0.31267982721328735, "num_tokens": 183080.0, "reward": 4.58039665222168, "reward_std": 1.3649483919143677, "rewards/coordinate_accuracy_reward_func/mean": 0.6928970813751221, "rewards/coordinate_accuracy_reward_func/std": 0.27080613374710083, "rewards/graph_topology_reward_func/mean": 3.825000047683716, "rewards/graph_topology_reward_func/std": 1.5216219425201416, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 9 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 728.375, "completions/clipped_ratio": 0.0, "completions/max_length": 1759.0, "completions/max_terminated_length": 1759.0, "completions/mean_length": 728.375, "completions/mean_terminated_length": 728.375, "completions/min_length": 403.0, "completions/min_terminated_length": 403.0, "epoch": 0.018796992481203006, "frac_reward_zero_std": 0.0, "grad_norm": 0.09685627371072769, "kl": 0.0, "learning_rate": 1.8750000000000003e-06, "loss": 0.0028924550861120224, "num_tokens": 202014.0, "reward": 5.31941032409668, "reward_std": 0.32870712876319885, "rewards/coordinate_accuracy_reward_func/mean": 0.7350351810455322, "rewards/coordinate_accuracy_reward_func/std": 0.08675692975521088, "rewards/graph_topology_reward_func/mean": 4.484375, "rewards/graph_topology_reward_func/std": 0.6549093723297119, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 10 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 979.0625, "completions/clipped_ratio": 0.3125, "completions/max_length": 2000.0, "completions/max_terminated_length": 763.0, "completions/mean_length": 979.0625, "completions/mean_terminated_length": 515.0, "completions/min_length": 422.0, "completions/min_terminated_length": 422.0, "epoch": 0.020676691729323307, "frac_reward_zero_std": 0.0, "grad_norm": 0.347843199968338, "kl": 0.0, "learning_rate": 2.0833333333333334e-06, "loss": 0.4994361996650696, "num_tokens": 225391.0, "reward": 4.135770797729492, "reward_std": 1.467184066772461, "rewards/coordinate_accuracy_reward_func/mean": 0.5691043734550476, "rewards/coordinate_accuracy_reward_func/std": 0.34202924370765686, "rewards/graph_topology_reward_func/mean": 3.5416665077209473, "rewards/graph_topology_reward_func/std": 2.1836938858032227, "rewards/xml_formatting_reward_func/mean": 0.02500000037252903, "rewards/xml_formatting_reward_func/std": 0.13416409492492676, "step": 11 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 471.5625, "completions/clipped_ratio": 0.0, "completions/max_length": 895.0, "completions/max_terminated_length": 895.0, "completions/mean_length": 471.5625, "completions/mean_terminated_length": 471.5625, "completions/min_length": 300.0, "completions/min_terminated_length": 300.0, "epoch": 0.022556390977443608, "frac_reward_zero_std": 0.5, "grad_norm": 0.02533583901822567, "kl": 0.0, "learning_rate": 2.2916666666666666e-06, "loss": -0.008149920962750912, "num_tokens": 239368.0, "reward": 5.7548604011535645, "reward_std": 0.3458578884601593, "rewards/coordinate_accuracy_reward_func/mean": 0.779860258102417, "rewards/coordinate_accuracy_reward_func/std": 0.03357023373246193, "rewards/graph_topology_reward_func/mean": 4.875, "rewards/graph_topology_reward_func/std": 0.5, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 12 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 760.6875, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1825.0, "completions/mean_length": 760.6875, "completions/mean_terminated_length": 678.0667114257812, "completions/min_length": 291.0, "completions/min_terminated_length": 291.0, "epoch": 0.02443609022556391, "frac_reward_zero_std": 0.5, "grad_norm": 0.42668941617012024, "kl": 0.0, "learning_rate": 2.5e-06, "loss": 0.1253407597541809, "num_tokens": 258691.0, "reward": 4.860828876495361, "reward_std": 1.2595736980438232, "rewards/coordinate_accuracy_reward_func/mean": 0.6108286380767822, "rewards/coordinate_accuracy_reward_func/std": 0.2667158246040344, "rewards/graph_topology_reward_func/mean": 4.1875, "rewards/graph_topology_reward_func/std": 1.6820127964019775, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 13 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 314.5625, "completions/clipped_ratio": 0.0, "completions/max_length": 426.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 314.5625, "completions/mean_terminated_length": 314.5625, "completions/min_length": 250.0, "completions/min_terminated_length": 250.0, "epoch": 0.02631578947368421, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "kl": 0.0, "learning_rate": 2.7083333333333334e-06, "loss": 0.0, "num_tokens": 269308.0, "reward": 5.900000095367432, "reward_std": 0.0, "rewards/coordinate_accuracy_reward_func/mean": 0.800000011920929, "rewards/coordinate_accuracy_reward_func/std": 0.0, "rewards/graph_topology_reward_func/mean": 5.0, "rewards/graph_topology_reward_func/std": 0.0, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 14 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 537.5625, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 728.0, "completions/mean_length": 537.5625, "completions/mean_terminated_length": 440.0666809082031, "completions/min_length": 281.0, "completions/min_terminated_length": 281.0, "epoch": 0.02819548872180451, "frac_reward_zero_std": 0.0, "grad_norm": 0.12479108572006226, "kl": 0.0, "learning_rate": 2.916666666666667e-06, "loss": 0.015275441110134125, "num_tokens": 283493.0, "reward": 4.17881965637207, "reward_std": 1.6195603609085083, "rewards/coordinate_accuracy_reward_func/mean": 0.5431054830551147, "rewards/coordinate_accuracy_reward_func/std": 0.37858161330223083, "rewards/graph_topology_reward_func/mean": 3.5357141494750977, "rewards/graph_topology_reward_func/std": 2.173745632171631, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 15 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1087.375, "completions/clipped_ratio": 0.0, "completions/max_length": 1560.0, "completions/max_terminated_length": 1560.0, "completions/mean_length": 1087.375, "completions/mean_terminated_length": 1087.375, "completions/min_length": 689.0, "completions/min_terminated_length": 689.0, "epoch": 0.03007518796992481, "frac_reward_zero_std": 0.0, "grad_norm": 0.029309239238500595, "kl": 0.0, "learning_rate": 3.125e-06, "loss": 0.01197317335754633, "num_tokens": 308747.0, "reward": 5.024444103240967, "reward_std": 0.3858163058757782, "rewards/coordinate_accuracy_reward_func/mean": 0.7664459943771362, "rewards/coordinate_accuracy_reward_func/std": 0.047260675579309464, "rewards/graph_topology_reward_func/mean": 4.157998085021973, "rewards/graph_topology_reward_func/std": 0.44025078415870667, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 16 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 916.25, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1258.0, "completions/mean_length": 916.25, "completions/mean_terminated_length": 844.0000610351562, "completions/min_length": 597.0, "completions/min_terminated_length": 597.0, "epoch": 0.03195488721804511, "frac_reward_zero_std": 0.0, "grad_norm": 0.08334092050790787, "kl": 0.0, "learning_rate": 3.3333333333333333e-06, "loss": 0.0710035115480423, "num_tokens": 330223.0, "reward": 5.054795742034912, "reward_std": 1.2215368747711182, "rewards/coordinate_accuracy_reward_func/mean": 0.7120075225830078, "rewards/coordinate_accuracy_reward_func/std": 0.2020058035850525, "rewards/graph_topology_reward_func/mean": 4.242788314819336, "rewards/graph_topology_reward_func/std": 1.2161568403244019, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 17 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 933.5625, "completions/clipped_ratio": 0.1875, "completions/max_length": 2000.0, "completions/max_terminated_length": 1149.0, "completions/mean_length": 933.5625, "completions/mean_terminated_length": 687.4615478515625, "completions/min_length": 454.0, "completions/min_terminated_length": 454.0, "epoch": 0.03383458646616541, "frac_reward_zero_std": 0.0, "grad_norm": 0.15413092076778412, "kl": 0.0, "learning_rate": 3.5416666666666673e-06, "loss": 0.22186844050884247, "num_tokens": 350920.0, "reward": 3.9765586853027344, "reward_std": 1.4142546653747559, "rewards/coordinate_accuracy_reward_func/mean": 0.5609335899353027, "rewards/coordinate_accuracy_reward_func/std": 0.341568261384964, "rewards/graph_topology_reward_func/mean": 3.390625, "rewards/graph_topology_reward_func/std": 2.091587781906128, "rewards/xml_formatting_reward_func/mean": 0.02500000037252903, "rewards/xml_formatting_reward_func/std": 0.13416409492492676, "step": 18 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1146.875, "completions/clipped_ratio": 0.25, "completions/max_length": 2000.0, "completions/max_terminated_length": 1954.0, "completions/mean_length": 1146.875, "completions/mean_terminated_length": 862.5, "completions/min_length": 504.0, "completions/min_terminated_length": 504.0, "epoch": 0.03571428571428571, "frac_reward_zero_std": 0.0, "grad_norm": 0.2560620605945587, "kl": 0.0, "learning_rate": 3.7500000000000005e-06, "loss": 0.573611855506897, "num_tokens": 377126.0, "reward": 3.601872444152832, "reward_std": 2.4138400554656982, "rewards/coordinate_accuracy_reward_func/mean": 0.4831221103668213, "rewards/coordinate_accuracy_reward_func/std": 0.34751877188682556, "rewards/graph_topology_reward_func/mean": 3.09375, "rewards/graph_topology_reward_func/std": 2.0348525047302246, "rewards/xml_formatting_reward_func/mean": 0.02500000037252903, "rewards/xml_formatting_reward_func/std": 0.13416409492492676, "step": 19 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 640.8125, "completions/clipped_ratio": 0.0, "completions/max_length": 1995.0, "completions/max_terminated_length": 1995.0, "completions/mean_length": 640.8125, "completions/mean_terminated_length": 640.8125, "completions/min_length": 285.0, "completions/min_terminated_length": 285.0, "epoch": 0.03759398496240601, "frac_reward_zero_std": 0.0, "grad_norm": 0.08432167768478394, "kl": 0.0, "learning_rate": 3.958333333333333e-06, "loss": 0.03268027305603027, "num_tokens": 393491.0, "reward": 4.919620037078857, "reward_std": 1.3752288818359375, "rewards/coordinate_accuracy_reward_func/mean": 0.7133700251579285, "rewards/coordinate_accuracy_reward_func/std": 0.19981735944747925, "rewards/graph_topology_reward_func/mean": 4.125, "rewards/graph_topology_reward_func/std": 1.3000000715255737, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 20 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 956.5625, "completions/clipped_ratio": 0.125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1764.0, "completions/mean_length": 956.5625, "completions/mean_terminated_length": 807.5000610351562, "completions/min_length": 266.0, "completions/min_terminated_length": 266.0, "epoch": 0.039473684210526314, "frac_reward_zero_std": 0.0, "grad_norm": 0.16231566667556763, "kl": 0.0, "learning_rate": 4.166666666666667e-06, "loss": 0.11739540100097656, "num_tokens": 415660.0, "reward": 4.395720481872559, "reward_std": 1.9851109981536865, "rewards/coordinate_accuracy_reward_func/mean": 0.6144704818725586, "rewards/coordinate_accuracy_reward_func/std": 0.310215562582016, "rewards/graph_topology_reward_func/mean": 3.71875, "rewards/graph_topology_reward_func/std": 1.6928157806396484, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 21 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 505.0, "completions/clipped_ratio": 0.0, "completions/max_length": 859.0, "completions/max_terminated_length": 859.0, "completions/mean_length": 505.0, "completions/mean_terminated_length": 505.0, "completions/min_length": 209.0, "completions/min_terminated_length": 209.0, "epoch": 0.041353383458646614, "frac_reward_zero_std": 0.0, "grad_norm": 0.09942260384559631, "kl": 0.0, "learning_rate": 4.3750000000000005e-06, "loss": -0.006339303217828274, "num_tokens": 430444.0, "reward": 5.276254177093506, "reward_std": 1.2513279914855957, "rewards/coordinate_accuracy_reward_func/mean": 0.7262544631958008, "rewards/coordinate_accuracy_reward_func/std": 0.2056841254234314, "rewards/graph_topology_reward_func/mean": 4.46875, "rewards/graph_topology_reward_func/std": 1.306633710861206, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 22 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 432.25, "completions/clipped_ratio": 0.0, "completions/max_length": 737.0, "completions/max_terminated_length": 737.0, "completions/mean_length": 432.25, "completions/mean_terminated_length": 432.25, "completions/min_length": 284.0, "completions/min_terminated_length": 284.0, "epoch": 0.043233082706766915, "frac_reward_zero_std": 0.5, "grad_norm": 0.0893731340765953, "kl": 0.0, "learning_rate": 4.583333333333333e-06, "loss": 0.0020187310874462128, "num_tokens": 442944.0, "reward": 5.143750190734863, "reward_std": 1.1481157541275024, "rewards/coordinate_accuracy_reward_func/mean": 0.75, "rewards/coordinate_accuracy_reward_func/std": 0.20000000298023224, "rewards/graph_topology_reward_func/mean": 4.3125, "rewards/graph_topology_reward_func/std": 1.5370426177978516, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 23 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1279.125, "completions/clipped_ratio": 0.375, "completions/max_length": 2000.0, "completions/max_terminated_length": 1599.0, "completions/mean_length": 1279.125, "completions/mean_terminated_length": 846.6000366210938, "completions/min_length": 565.0, "completions/min_terminated_length": 565.0, "epoch": 0.045112781954887216, "frac_reward_zero_std": 0.0, "grad_norm": 0.21305620670318604, "kl": 0.0, "learning_rate": 4.791666666666668e-06, "loss": 0.49527961015701294, "num_tokens": 469698.0, "reward": 2.82751727104187, "reward_std": 2.1899497509002686, "rewards/coordinate_accuracy_reward_func/mean": 0.5081116557121277, "rewards/coordinate_accuracy_reward_func/std": 0.3585614860057831, "rewards/graph_topology_reward_func/mean": 2.3131556510925293, "rewards/graph_topology_reward_func/std": 1.686226487159729, "rewards/xml_formatting_reward_func/mean": 0.0062500000931322575, "rewards/xml_formatting_reward_func/std": 0.14361406862735748, "step": 24 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 474.5, "completions/clipped_ratio": 0.0, "completions/max_length": 615.0, "completions/max_terminated_length": 615.0, "completions/mean_length": 474.5, "completions/mean_terminated_length": 474.5, "completions/min_length": 306.0, "completions/min_terminated_length": 306.0, "epoch": 0.046992481203007516, "frac_reward_zero_std": 0.0, "grad_norm": 0.11687883734703064, "kl": 0.0, "learning_rate": 5e-06, "loss": -0.020139899104833603, "num_tokens": 482874.0, "reward": 4.634520530700684, "reward_std": 1.7706841230392456, "rewards/coordinate_accuracy_reward_func/mean": 0.6657707691192627, "rewards/coordinate_accuracy_reward_func/std": 0.2784713804721832, "rewards/graph_topology_reward_func/mean": 3.90625, "rewards/graph_topology_reward_func/std": 1.652964472770691, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 25 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 516.1875, "completions/clipped_ratio": 0.0, "completions/max_length": 996.0, "completions/max_terminated_length": 996.0, "completions/mean_length": 516.1875, "completions/mean_terminated_length": 516.1875, "completions/min_length": 275.0, "completions/min_terminated_length": 275.0, "epoch": 0.04887218045112782, "frac_reward_zero_std": 0.0, "grad_norm": 0.02738766372203827, "kl": 0.0, "learning_rate": 4.999735579817769e-06, "loss": 0.018890798091888428, "num_tokens": 498957.0, "reward": 5.628383159637451, "reward_std": 0.31740158796310425, "rewards/coordinate_accuracy_reward_func/mean": 0.7158830165863037, "rewards/coordinate_accuracy_reward_func/std": 0.11106010526418686, "rewards/graph_topology_reward_func/mean": 4.8125, "rewards/graph_topology_reward_func/std": 0.4330127239227295, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 26 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 996.8125, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1721.0, "completions/mean_length": 996.8125, "completions/mean_terminated_length": 929.9334106445312, "completions/min_length": 595.0, "completions/min_terminated_length": 595.0, "epoch": 0.05075187969924812, "frac_reward_zero_std": 0.0, "grad_norm": 0.25558069348335266, "kl": 0.0, "learning_rate": 4.998942375205502e-06, "loss": 0.17509771883487701, "num_tokens": 521194.0, "reward": 4.330102920532227, "reward_std": 1.46968412399292, "rewards/coordinate_accuracy_reward_func/mean": 0.6217696666717529, "rewards/coordinate_accuracy_reward_func/std": 0.25099989771842957, "rewards/graph_topology_reward_func/mean": 3.6458334922790527, "rewards/graph_topology_reward_func/std": 1.5078498125076294, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 27 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 707.9375, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1173.0, "completions/mean_length": 707.9375, "completions/mean_terminated_length": 621.800048828125, "completions/min_length": 407.0, "completions/min_terminated_length": 407.0, "epoch": 0.05263157894736842, "frac_reward_zero_std": 0.0, "grad_norm": 0.1116558387875557, "kl": 0.0, "learning_rate": 4.997620553954645e-06, "loss": 0.18896029889583588, "num_tokens": 540089.0, "reward": 4.687322616577148, "reward_std": 1.2077451944351196, "rewards/coordinate_accuracy_reward_func/mean": 0.6248228549957275, "rewards/coordinate_accuracy_reward_func/std": 0.2564260959625244, "rewards/graph_topology_reward_func/mean": 4.0, "rewards/graph_topology_reward_func/std": 1.632993221282959, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 28 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 759.5, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1594.0, "completions/mean_length": 759.5, "completions/mean_terminated_length": 676.800048828125, "completions/min_length": 308.0, "completions/min_terminated_length": 308.0, "epoch": 0.05451127819548872, "frac_reward_zero_std": 0.0, "grad_norm": 0.04255201295018196, "kl": 0.0, "learning_rate": 4.995770395678171e-06, "loss": 0.1634421944618225, "num_tokens": 559249.0, "reward": 5.179400444030762, "reward_std": 0.9883494973182678, "rewards/coordinate_accuracy_reward_func/mean": 0.6981505155563354, "rewards/coordinate_accuracy_reward_func/std": 0.22082455456256866, "rewards/graph_topology_reward_func/mean": 4.400000095367432, "rewards/graph_topology_reward_func/std": 1.246862769126892, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 29 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 479.875, "completions/clipped_ratio": 0.0, "completions/max_length": 829.0, "completions/max_terminated_length": 829.0, "completions/mean_length": 479.875, "completions/mean_terminated_length": 479.875, "completions/min_length": 260.0, "completions/min_terminated_length": 260.0, "epoch": 0.05639097744360902, "frac_reward_zero_std": 0.0, "grad_norm": 0.06177043169736862, "kl": 0.0, "learning_rate": 4.993392291751431e-06, "loss": 0.00955201406031847, "num_tokens": 574223.0, "reward": 5.669327259063721, "reward_std": 0.5043275952339172, "rewards/coordinate_accuracy_reward_func/mean": 0.6943275332450867, "rewards/coordinate_accuracy_reward_func/std": 0.19961273670196533, "rewards/graph_topology_reward_func/mean": 4.875, "rewards/graph_topology_reward_func/std": 0.5, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 30 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 578.625, "completions/clipped_ratio": 0.0, "completions/max_length": 1154.0, "completions/max_terminated_length": 1154.0, "completions/mean_length": 578.625, "completions/mean_terminated_length": 578.625, "completions/min_length": 306.0, "completions/min_terminated_length": 306.0, "epoch": 0.05827067669172932, "frac_reward_zero_std": 0.5, "grad_norm": 0.11814803630113602, "kl": 0.0, "learning_rate": 4.990486745229364e-06, "loss": 0.07345657795667648, "num_tokens": 590441.0, "reward": 5.427145004272461, "reward_std": 1.0419297218322754, "rewards/coordinate_accuracy_reward_func/mean": 0.6583948731422424, "rewards/coordinate_accuracy_reward_func/std": 0.20998412370681763, "rewards/graph_topology_reward_func/mean": 4.6875, "rewards/graph_topology_reward_func/std": 1.25, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 31 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 843.25, "completions/clipped_ratio": 0.1875, "completions/max_length": 2000.0, "completions/max_terminated_length": 1154.0, "completions/mean_length": 843.25, "completions/mean_terminated_length": 576.3077392578125, "completions/min_length": 264.0, "completions/min_terminated_length": 264.0, "epoch": 0.06015037593984962, "frac_reward_zero_std": 0.0, "grad_norm": 0.19636015594005585, "kl": 0.0, "learning_rate": 4.9870543707400835e-06, "loss": 0.5660920739173889, "num_tokens": 610045.0, "reward": 4.40899658203125, "reward_std": 2.374026298522949, "rewards/coordinate_accuracy_reward_func/mean": 0.5839964747428894, "rewards/coordinate_accuracy_reward_func/std": 0.304964542388916, "rewards/graph_topology_reward_func/mean": 3.78125, "rewards/graph_topology_reward_func/std": 1.957624077796936, "rewards/xml_formatting_reward_func/mean": 0.04374999925494194, "rewards/xml_formatting_reward_func/std": 0.12093387544155121, "step": 32 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 868.6875, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1479.0, "completions/mean_length": 868.6875, "completions/mean_terminated_length": 793.2667236328125, "completions/min_length": 553.0, "completions/min_terminated_length": 553.0, "epoch": 0.06203007518796992, "frac_reward_zero_std": 0.0, "grad_norm": 0.11138980090618134, "kl": 0.0, "learning_rate": 4.983095894354858e-06, "loss": 0.1629563868045807, "num_tokens": 631016.0, "reward": 4.796245574951172, "reward_std": 1.2293848991394043, "rewards/coordinate_accuracy_reward_func/mean": 0.7036316990852356, "rewards/coordinate_accuracy_reward_func/std": 0.19788314402103424, "rewards/graph_topology_reward_func/mean": 4.011363506317139, "rewards/graph_topology_reward_func/std": 1.1640269756317139, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 33 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1026.6875, "completions/clipped_ratio": 0.1875, "completions/max_length": 2000.0, "completions/max_terminated_length": 1271.0, "completions/mean_length": 1026.6875, "completions/mean_terminated_length": 802.0769653320312, "completions/min_length": 424.0, "completions/min_terminated_length": 424.0, "epoch": 0.06390977443609022, "frac_reward_zero_std": 0.0, "grad_norm": 0.1571996659040451, "kl": 0.0, "learning_rate": 4.978612153434527e-06, "loss": 0.30387550592422485, "num_tokens": 655443.0, "reward": 3.698967933654785, "reward_std": 1.6794087886810303, "rewards/coordinate_accuracy_reward_func/mean": 0.6154453158378601, "rewards/coordinate_accuracy_reward_func/std": 0.3084288239479065, "rewards/graph_topology_reward_func/mean": 3.0210227966308594, "rewards/graph_topology_reward_func/std": 1.2906817197799683, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 34 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 554.3125, "completions/clipped_ratio": 0.0, "completions/max_length": 1059.0, "completions/max_terminated_length": 1059.0, "completions/mean_length": 554.3125, "completions/mean_terminated_length": 554.3125, "completions/min_length": 274.0, "completions/min_terminated_length": 274.0, "epoch": 0.06578947368421052, "frac_reward_zero_std": 0.0, "grad_norm": 0.11644691228866577, "kl": 0.0, "learning_rate": 4.973604096452361e-06, "loss": -0.0010448116809129715, "num_tokens": 670600.0, "reward": 4.505001068115234, "reward_std": 1.0484553575515747, "rewards/coordinate_accuracy_reward_func/mean": 0.7828419208526611, "rewards/coordinate_accuracy_reward_func/std": 0.027784282341599464, "rewards/graph_topology_reward_func/mean": 3.622159004211426, "rewards/graph_topology_reward_func/std": 1.2344591617584229, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 35 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 451.625, "completions/clipped_ratio": 0.0, "completions/max_length": 1122.0, "completions/max_terminated_length": 1122.0, "completions/mean_length": 451.625, "completions/mean_terminated_length": 451.625, "completions/min_length": 294.0, "completions/min_terminated_length": 294.0, "epoch": 0.06766917293233082, "frac_reward_zero_std": 0.5, "grad_norm": 0.018829086795449257, "kl": 0.0, "learning_rate": 4.968072782793436e-06, "loss": -0.004240840673446655, "num_tokens": 684082.0, "reward": 5.624947547912598, "reward_std": 0.26859766244888306, "rewards/coordinate_accuracy_reward_func/mean": 0.7749476432800293, "rewards/coordinate_accuracy_reward_func/std": 0.054724857211112976, "rewards/graph_topology_reward_func/mean": 4.75, "rewards/graph_topology_reward_func/std": 0.44721361994743347, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 36 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 593.0625, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 609.0, "completions/mean_length": 593.0625, "completions/mean_terminated_length": 499.2666931152344, "completions/min_length": 386.0, "completions/min_terminated_length": 386.0, "epoch": 0.06954887218045112, "frac_reward_zero_std": 0.0, "grad_norm": 0.12222662568092346, "kl": 0.0, "learning_rate": 4.962019382530521e-06, "loss": 0.240690216422081, "num_tokens": 700387.0, "reward": 5.314290523529053, "reward_std": 1.239711046218872, "rewards/coordinate_accuracy_reward_func/mean": 0.7330405116081238, "rewards/coordinate_accuracy_reward_func/std": 0.20051027834415436, "rewards/graph_topology_reward_func/mean": 4.5, "rewards/graph_topology_reward_func/std": 1.2747548818588257, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 37 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 774.4375, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1878.0, "completions/mean_length": 774.4375, "completions/mean_terminated_length": 692.7333984375, "completions/min_length": 355.0, "completions/min_terminated_length": 355.0, "epoch": 0.07142857142857142, "frac_reward_zero_std": 0.0, "grad_norm": 0.0919976755976677, "kl": 0.0, "learning_rate": 4.955445176176577e-06, "loss": 0.23787111043930054, "num_tokens": 720778.0, "reward": 4.445254325866699, "reward_std": 1.094590663909912, "rewards/coordinate_accuracy_reward_func/mean": 0.5870813727378845, "rewards/coordinate_accuracy_reward_func/std": 0.24998511373996735, "rewards/graph_topology_reward_func/mean": 3.795673131942749, "rewards/graph_topology_reward_func/std": 1.6730087995529175, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 38 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 731.625, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1213.0, "completions/mean_length": 731.625, "completions/mean_terminated_length": 647.0667114257812, "completions/min_length": 380.0, "completions/min_terminated_length": 380.0, "epoch": 0.07330827067669173, "frac_reward_zero_std": 0.0, "grad_norm": 0.14144465327262878, "kl": 0.0, "learning_rate": 4.948351554413879e-06, "loss": 0.2137073129415512, "num_tokens": 739732.0, "reward": 5.306427955627441, "reward_std": 1.2259447574615479, "rewards/coordinate_accuracy_reward_func/mean": 0.6939277648925781, "rewards/coordinate_accuracy_reward_func/std": 0.20565059781074524, "rewards/graph_topology_reward_func/mean": 4.53125, "rewards/graph_topology_reward_func/std": 1.2841178178787231, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 39 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 689.0625, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1022.0, "completions/mean_length": 689.0625, "completions/mean_terminated_length": 601.6666870117188, "completions/min_length": 348.0, "completions/min_terminated_length": 348.0, "epoch": 0.07518796992481203, "frac_reward_zero_std": 0.0, "grad_norm": 0.12783606350421906, "kl": 0.0, "learning_rate": 4.9407400177998335e-06, "loss": 0.21467936038970947, "num_tokens": 757525.0, "reward": 5.114828109741211, "reward_std": 1.168297529220581, "rewards/coordinate_accuracy_reward_func/mean": 0.6669114232063293, "rewards/coordinate_accuracy_reward_func/std": 0.20427648723125458, "rewards/graph_topology_reward_func/mean": 4.366666793823242, "rewards/graph_topology_reward_func/std": 1.272093653678894, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 40 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1092.875, "completions/clipped_ratio": 0.25, "completions/max_length": 2000.0, "completions/max_terminated_length": 1819.0, "completions/mean_length": 1092.875, "completions/mean_terminated_length": 790.5, "completions/min_length": 301.0, "completions/min_terminated_length": 301.0, "epoch": 0.07706766917293233, "frac_reward_zero_std": 0.0, "grad_norm": 0.22410424053668976, "kl": 0.0, "learning_rate": 4.93261217644956e-06, "loss": 0.5799803733825684, "num_tokens": 780771.0, "reward": 3.643676996231079, "reward_std": 2.4054880142211914, "rewards/coordinate_accuracy_reward_func/mean": 0.5698132514953613, "rewards/coordinate_accuracy_reward_func/std": 0.34493160247802734, "rewards/graph_topology_reward_func/mean": 3.048863649368286, "rewards/graph_topology_reward_func/std": 2.010528564453125, "rewards/xml_formatting_reward_func/mean": 0.02500000037252903, "rewards/xml_formatting_reward_func/std": 0.13416409492492676, "step": 41 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 719.1875, "completions/clipped_ratio": 0.125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1092.0, "completions/mean_length": 719.1875, "completions/mean_terminated_length": 536.2142944335938, "completions/min_length": 292.0, "completions/min_terminated_length": 292.0, "epoch": 0.07894736842105263, "frac_reward_zero_std": 0.0, "grad_norm": 0.2469155490398407, "kl": 0.0, "learning_rate": 4.9239697496952904e-06, "loss": 0.26751255989074707, "num_tokens": 799238.0, "reward": 4.185155868530273, "reward_std": 1.1655606031417847, "rewards/coordinate_accuracy_reward_func/mean": 0.6414055228233337, "rewards/coordinate_accuracy_reward_func/std": 0.32004937529563904, "rewards/graph_topology_reward_func/mean": 3.5, "rewards/graph_topology_reward_func/std": 1.9493589401245117, "rewards/xml_formatting_reward_func/mean": 0.04374999925494194, "rewards/xml_formatting_reward_func/std": 0.12093386799097061, "step": 42 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 410.9375, "completions/clipped_ratio": 0.0, "completions/max_length": 613.0, "completions/max_terminated_length": 613.0, "completions/mean_length": 410.9375, "completions/mean_terminated_length": 410.9375, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "epoch": 0.08082706766917293, "frac_reward_zero_std": 0.5, "grad_norm": 0.03335123509168625, "kl": 0.0, "learning_rate": 4.914814565722671e-06, "loss": -0.0009731263853609562, "num_tokens": 811573.0, "reward": 5.215703964233398, "reward_std": 0.4625944495201111, "rewards/coordinate_accuracy_reward_func/mean": 0.7823706865310669, "rewards/coordinate_accuracy_reward_func/std": 0.02345484122633934, "rewards/graph_topology_reward_func/mean": 4.333333492279053, "rewards/graph_topology_reward_func/std": 0.942808985710144, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 43 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1101.5, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1858.0, "completions/mean_length": 1101.5, "completions/mean_terminated_length": 1041.60009765625, "completions/min_length": 615.0, "completions/min_terminated_length": 615.0, "epoch": 0.08270676691729323, "frac_reward_zero_std": 0.0, "grad_norm": 0.13523919880390167, "kl": 0.0, "learning_rate": 4.905148561184033e-06, "loss": 0.0513731986284256, "num_tokens": 836189.0, "reward": 3.5792479515075684, "reward_std": 1.8848676681518555, "rewards/coordinate_accuracy_reward_func/mean": 0.6010575294494629, "rewards/coordinate_accuracy_reward_func/std": 0.303867369890213, "rewards/graph_topology_reward_func/mean": 2.9344406127929688, "rewards/graph_topology_reward_func/std": 1.7757271528244019, "rewards/xml_formatting_reward_func/mean": 0.04374999925494194, "rewards/xml_formatting_reward_func/std": 0.12093387544155121, "step": 44 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 785.375, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1257.0, "completions/mean_length": 785.375, "completions/mean_terminated_length": 704.4000244140625, "completions/min_length": 514.0, "completions/min_terminated_length": 514.0, "epoch": 0.08458646616541353, "frac_reward_zero_std": 0.5, "grad_norm": 0.06288287043571472, "kl": 0.0, "learning_rate": 4.894973780788722e-06, "loss": 0.16060885787010193, "num_tokens": 855747.0, "reward": 5.241495132446289, "reward_std": 0.9942088723182678, "rewards/coordinate_accuracy_reward_func/mean": 0.7383702993392944, "rewards/coordinate_accuracy_reward_func/std": 0.19829627871513367, "rewards/graph_topology_reward_func/mean": 4.421875, "rewards/graph_topology_reward_func/std": 1.260683536529541, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 45 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1327.8125, "completions/clipped_ratio": 0.375, "completions/max_length": 2000.0, "completions/max_terminated_length": 1670.0, "completions/mean_length": 1327.8125, "completions/mean_terminated_length": 924.5, "completions/min_length": 630.0, "completions/min_terminated_length": 630.0, "epoch": 0.08646616541353383, "frac_reward_zero_std": 0.0, "grad_norm": 0.29088926315307617, "kl": 0.0, "learning_rate": 4.884292376870567e-06, "loss": 0.6214227676391602, "num_tokens": 884848.0, "reward": 3.098896026611328, "reward_std": 2.6377406120300293, "rewards/coordinate_accuracy_reward_func/mean": 0.44979915022850037, "rewards/coordinate_accuracy_reward_func/std": 0.3657219409942627, "rewards/graph_topology_reward_func/mean": 2.6428465843200684, "rewards/graph_topology_reward_func/std": 2.076526165008545, "rewards/xml_formatting_reward_func/mean": 0.0062500000931322575, "rewards/xml_formatting_reward_func/std": 0.14361406862735748, "step": 46 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 977.9375, "completions/clipped_ratio": 0.1875, "completions/max_length": 2000.0, "completions/max_terminated_length": 1424.0, "completions/mean_length": 977.9375, "completions/mean_terminated_length": 742.0769653320312, "completions/min_length": 398.0, "completions/min_terminated_length": 398.0, "epoch": 0.08834586466165413, "frac_reward_zero_std": 0.0, "grad_norm": 0.3349037170410156, "kl": 0.0, "learning_rate": 4.873106608932585e-06, "loss": 0.2385222613811493, "num_tokens": 906783.0, "reward": 3.5078349113464355, "reward_std": 1.6186273097991943, "rewards/coordinate_accuracy_reward_func/mean": 0.6265847086906433, "rewards/coordinate_accuracy_reward_func/std": 0.3121328353881836, "rewards/graph_topology_reward_func/mean": 2.8375000953674316, "rewards/graph_topology_reward_func/std": 1.6053556203842163, "rewards/xml_formatting_reward_func/mean": 0.04374999925494194, "rewards/xml_formatting_reward_func/std": 0.12093386799097061, "step": 47 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 564.4375, "completions/clipped_ratio": 0.0, "completions/max_length": 967.0, "completions/max_terminated_length": 967.0, "completions/mean_length": 564.4375, "completions/mean_terminated_length": 564.4375, "completions/min_length": 387.0, "completions/min_terminated_length": 387.0, "epoch": 0.09022556390977443, "frac_reward_zero_std": 0.0, "grad_norm": 0.03590282425284386, "kl": 0.0, "learning_rate": 4.861418843169012e-06, "loss": -0.002396233845502138, "num_tokens": 922230.0, "reward": 5.4746198654174805, "reward_std": 0.2707287073135376, "rewards/coordinate_accuracy_reward_func/mean": 0.7496196031570435, "rewards/coordinate_accuracy_reward_func/std": 0.07695239782333374, "rewards/graph_topology_reward_func/mean": 4.625, "rewards/graph_topology_reward_func/std": 0.5, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 48 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 318.6875, "completions/clipped_ratio": 0.0, "completions/max_length": 564.0, "completions/max_terminated_length": 564.0, "completions/mean_length": 318.6875, "completions/mean_terminated_length": 318.6875, "completions/min_length": 201.0, "completions/min_terminated_length": 201.0, "epoch": 0.09210526315789473, "frac_reward_zero_std": 0.5, "grad_norm": 0.049704719334840775, "kl": 0.0, "learning_rate": 4.849231551964771e-06, "loss": 0.008765624836087227, "num_tokens": 932913.0, "reward": 4.962500095367432, "reward_std": 0.7763237953186035, "rewards/coordinate_accuracy_reward_func/mean": 0.800000011920929, "rewards/coordinate_accuracy_reward_func/std": 0.0, "rewards/graph_topology_reward_func/mean": 4.0625, "rewards/graph_topology_reward_func/std": 1.4361406564712524, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 49 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1132.625, "completions/clipped_ratio": 0.3125, "completions/max_length": 2000.0, "completions/max_terminated_length": 990.0, "completions/mean_length": 1132.625, "completions/mean_terminated_length": 738.3636474609375, "completions/min_length": 471.0, "completions/min_terminated_length": 471.0, "epoch": 0.09398496240601503, "frac_reward_zero_std": 0.0, "grad_norm": 0.23879152536392212, "kl": 0.0, "learning_rate": 4.836547313372472e-06, "loss": 0.43627607822418213, "num_tokens": 957899.0, "reward": 3.4175827503204346, "reward_std": 2.14681339263916, "rewards/coordinate_accuracy_reward_func/mean": 0.45508265495300293, "rewards/coordinate_accuracy_reward_func/std": 0.2982906401157379, "rewards/graph_topology_reward_func/mean": 2.9375, "rewards/graph_topology_reward_func/std": 1.7969882488250732, "rewards/xml_formatting_reward_func/mean": 0.02500000037252903, "rewards/xml_formatting_reward_func/std": 0.13416409492492676, "step": 50 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 561.8125, "completions/clipped_ratio": 0.0, "completions/max_length": 1035.0, "completions/max_terminated_length": 1035.0, "completions/mean_length": 561.8125, "completions/mean_terminated_length": 561.8125, "completions/min_length": 279.0, "completions/min_terminated_length": 279.0, "epoch": 0.09586466165413533, "frac_reward_zero_std": 0.0, "grad_norm": 0.08872710168361664, "kl": 0.0, "learning_rate": 4.823368810567056e-06, "loss": 0.02139318734407425, "num_tokens": 973176.0, "reward": 3.732870101928711, "reward_std": 1.4560611248016357, "rewards/coordinate_accuracy_reward_func/mean": 0.7141200304031372, "rewards/coordinate_accuracy_reward_func/std": 0.19907577335834503, "rewards/graph_topology_reward_func/mean": 2.9375, "rewards/graph_topology_reward_func/std": 1.3889445066452026, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 51 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 472.8125, "completions/clipped_ratio": 0.0, "completions/max_length": 691.0, "completions/max_terminated_length": 691.0, "completions/mean_length": 472.8125, "completions/mean_terminated_length": 472.8125, "completions/min_length": 294.0, "completions/min_terminated_length": 294.0, "epoch": 0.09774436090225563, "frac_reward_zero_std": 0.0, "grad_norm": 0.03164717182517052, "kl": 0.0, "learning_rate": 4.809698831278217e-06, "loss": -0.007672642823308706, "num_tokens": 986325.0, "reward": 5.431108474731445, "reward_std": 0.5168022513389587, "rewards/coordinate_accuracy_reward_func/mean": 0.7686082124710083, "rewards/coordinate_accuracy_reward_func/std": 0.05186959356069565, "rewards/graph_topology_reward_func/mean": 4.5625, "rewards/graph_topology_reward_func/std": 0.8139410614967346, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 52 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 525.5625, "completions/clipped_ratio": 0.0, "completions/max_length": 1059.0, "completions/max_terminated_length": 1059.0, "completions/mean_length": 525.5625, "completions/mean_terminated_length": 525.5625, "completions/min_length": 415.0, "completions/min_terminated_length": 415.0, "epoch": 0.09962406015037593, "frac_reward_zero_std": 0.0, "grad_norm": 0.013874426484107971, "kl": 0.0, "learning_rate": 4.7955402672006855e-06, "loss": 0.0058916304260492325, "num_tokens": 1000846.0, "reward": 5.055930137634277, "reward_std": 0.2548319399356842, "rewards/coordinate_accuracy_reward_func/mean": 0.7059301137924194, "rewards/coordinate_accuracy_reward_func/std": 0.16241170465946198, "rewards/graph_topology_reward_func/mean": 4.25, "rewards/graph_topology_reward_func/std": 0.8215838670730591, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 53 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 453.0, "completions/clipped_ratio": 0.0, "completions/max_length": 880.0, "completions/max_terminated_length": 880.0, "completions/mean_length": 453.0, "completions/mean_terminated_length": 453.0, "completions/min_length": 296.0, "completions/min_terminated_length": 296.0, "epoch": 0.10150375939849623, "frac_reward_zero_std": 0.0, "grad_norm": 0.021252281963825226, "kl": 0.0, "learning_rate": 4.780896113382536e-06, "loss": -0.009775628335773945, "num_tokens": 1013678.0, "reward": 5.638901233673096, "reward_std": 0.27519845962524414, "rewards/coordinate_accuracy_reward_func/mean": 0.7889013290405273, "rewards/coordinate_accuracy_reward_func/std": 0.016853881999850273, "rewards/graph_topology_reward_func/mean": 4.75, "rewards/graph_topology_reward_func/std": 0.44721361994743347, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 54 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 462.5, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 562.0, "completions/mean_length": 462.5, "completions/mean_terminated_length": 360.0000305175781, "completions/min_length": 296.0, "completions/min_terminated_length": 296.0, "epoch": 0.10338345864661654, "frac_reward_zero_std": 0.0, "grad_norm": 0.06714082509279251, "kl": 0.0, "learning_rate": 4.765769467591626e-06, "loss": 0.1220107153058052, "num_tokens": 1026662.0, "reward": 3.9009714126586914, "reward_std": 0.9124672412872314, "rewards/coordinate_accuracy_reward_func/mean": 0.6947215795516968, "rewards/coordinate_accuracy_reward_func/std": 0.27174893021583557, "rewards/graph_topology_reward_func/mean": 3.125, "rewards/graph_topology_reward_func/std": 1.8211718797683716, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 55 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 506.5625, "completions/clipped_ratio": 0.0, "completions/max_length": 996.0, "completions/max_terminated_length": 996.0, "completions/mean_length": 506.5625, "completions/mean_terminated_length": 506.5625, "completions/min_length": 254.0, "completions/min_terminated_length": 254.0, "epoch": 0.10526315789473684, "frac_reward_zero_std": 0.0, "grad_norm": 0.12259690463542938, "kl": 0.0, "learning_rate": 4.750163529660303e-06, "loss": 0.013522312045097351, "num_tokens": 1041055.0, "reward": 4.8603925704956055, "reward_std": 1.1629693508148193, "rewards/coordinate_accuracy_reward_func/mean": 0.7478925585746765, "rewards/coordinate_accuracy_reward_func/std": 0.19957560300827026, "rewards/graph_topology_reward_func/mean": 4.012499809265137, "rewards/graph_topology_reward_func/std": 1.2784757614135742, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 56 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 889.375, "completions/clipped_ratio": 0.1875, "completions/max_length": 2000.0, "completions/max_terminated_length": 873.0, "completions/mean_length": 889.375, "completions/mean_terminated_length": 633.0769653320312, "completions/min_length": 359.0, "completions/min_terminated_length": 359.0, "epoch": 0.10714285714285714, "frac_reward_zero_std": 0.0, "grad_norm": 0.17437641322612762, "kl": 0.0, "learning_rate": 4.734081600808531e-06, "loss": 0.4984560012817383, "num_tokens": 1061573.0, "reward": 4.086814880371094, "reward_std": 2.0909018516540527, "rewards/coordinate_accuracy_reward_func/mean": 0.6368147134780884, "rewards/coordinate_accuracy_reward_func/std": 0.31759390234947205, "rewards/graph_topology_reward_func/mean": 3.40625, "rewards/graph_topology_reward_func/std": 1.8904916048049927, "rewards/xml_formatting_reward_func/mean": 0.04374999925494194, "rewards/xml_formatting_reward_func/std": 0.12093387544155121, "step": 57 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 814.4375, "completions/clipped_ratio": 0.125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1143.0, "completions/mean_length": 814.4375, "completions/mean_terminated_length": 645.0714721679688, "completions/min_length": 305.0, "completions/min_terminated_length": 305.0, "epoch": 0.10902255639097744, "frac_reward_zero_std": 0.0, "grad_norm": 0.1329573541879654, "kl": 0.0, "learning_rate": 4.717527082945555e-06, "loss": 0.15290413796901703, "num_tokens": 1080716.0, "reward": 5.0059814453125, "reward_std": 0.8771085143089294, "rewards/coordinate_accuracy_reward_func/mean": 0.7059814929962158, "rewards/coordinate_accuracy_reward_func/std": 0.1980554461479187, "rewards/graph_topology_reward_func/mean": 4.21875, "rewards/graph_topology_reward_func/std": 1.2512494325637817, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 58 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 662.3125, "completions/clipped_ratio": 0.125, "completions/max_length": 2000.0, "completions/max_terminated_length": 809.0, "completions/mean_length": 662.3125, "completions/mean_terminated_length": 471.21429443359375, "completions/min_length": 324.0, "completions/min_terminated_length": 324.0, "epoch": 0.11090225563909774, "frac_reward_zero_std": 0.0, "grad_norm": 0.08007911592721939, "kl": 0.0, "learning_rate": 4.700503477950278e-06, "loss": 0.3740197420120239, "num_tokens": 1097249.0, "reward": 4.948113918304443, "reward_std": 1.3065218925476074, "rewards/coordinate_accuracy_reward_func/mean": 0.6981135606765747, "rewards/coordinate_accuracy_reward_func/std": 0.27256348729133606, "rewards/graph_topology_reward_func/mean": 4.1875, "rewards/graph_topology_reward_func/std": 1.6670832633972168, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 59 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 830.875, "completions/clipped_ratio": 0.0, "completions/max_length": 1250.0, "completions/max_terminated_length": 1250.0, "completions/mean_length": 830.875, "completions/mean_terminated_length": 830.875, "completions/min_length": 639.0, "completions/min_terminated_length": 639.0, "epoch": 0.11278195488721804, "frac_reward_zero_std": 0.0, "grad_norm": 0.026666993275284767, "kl": 0.0, "learning_rate": 4.6830143869304904e-06, "loss": 0.0031928857788443565, "num_tokens": 1118111.0, "reward": 5.2190656661987305, "reward_std": 0.42291462421417236, "rewards/coordinate_accuracy_reward_func/mean": 0.7775888442993164, "rewards/coordinate_accuracy_reward_func/std": 0.044992364943027496, "rewards/graph_topology_reward_func/mean": 4.341477394104004, "rewards/graph_topology_reward_func/std": 0.4250810444355011, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 60 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 522.75, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 730.0, "completions/mean_length": 522.75, "completions/mean_terminated_length": 424.2666931152344, "completions/min_length": 290.0, "completions/min_terminated_length": 290.0, "epoch": 0.11466165413533834, "frac_reward_zero_std": 0.0, "grad_norm": 0.08957364410161972, "kl": 0.0, "learning_rate": 4.665063509461098e-06, "loss": -0.01662450097501278, "num_tokens": 1132059.0, "reward": 5.1855010986328125, "reward_std": 1.1648216247558594, "rewards/coordinate_accuracy_reward_func/mean": 0.7448759078979492, "rewards/coordinate_accuracy_reward_func/std": 0.1996115893125534, "rewards/graph_topology_reward_func/mean": 4.359375, "rewards/graph_topology_reward_func/std": 1.3101168870925903, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 61 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 767.875, "completions/clipped_ratio": 0.125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1562.0, "completions/mean_length": 767.875, "completions/mean_terminated_length": 591.857177734375, "completions/min_length": 271.0, "completions/min_terminated_length": 271.0, "epoch": 0.11654135338345864, "frac_reward_zero_std": 0.5, "grad_norm": 0.11776944249868393, "kl": 0.0, "learning_rate": 4.646654642801533e-06, "loss": 0.33867067098617554, "num_tokens": 1149929.0, "reward": 4.986268043518066, "reward_std": 1.345542550086975, "rewards/coordinate_accuracy_reward_func/mean": 0.6737679243087769, "rewards/coordinate_accuracy_reward_func/std": 0.28157395124435425, "rewards/graph_topology_reward_func/mean": 4.25, "rewards/graph_topology_reward_func/std": 1.6931233406066895, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 62 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 585.25, "completions/clipped_ratio": 0.0, "completions/max_length": 1384.0, "completions/max_terminated_length": 1384.0, "completions/mean_length": 585.25, "completions/mean_terminated_length": 585.25, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "epoch": 0.11842105263157894, "frac_reward_zero_std": 0.0, "grad_norm": 0.07201402634382248, "kl": 0.0, "learning_rate": 4.627791681092499e-06, "loss": 0.012828437611460686, "num_tokens": 1164877.0, "reward": 5.319637298583984, "reward_std": 0.3318747282028198, "rewards/coordinate_accuracy_reward_func/mean": 0.700406551361084, "rewards/coordinate_accuracy_reward_func/std": 0.16134105622768402, "rewards/graph_topology_reward_func/mean": 4.519230842590332, "rewards/graph_topology_reward_func/std": 0.6208094358444214, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 63 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 580.4375, "completions/clipped_ratio": 0.0, "completions/max_length": 1082.0, "completions/max_terminated_length": 1082.0, "completions/mean_length": 580.4375, "completions/mean_terminated_length": 580.4375, "completions/min_length": 290.0, "completions/min_terminated_length": 290.0, "epoch": 0.12030075187969924, "frac_reward_zero_std": 0.0, "grad_norm": 0.013361654244363308, "kl": 0.0, "learning_rate": 4.608478614532215e-06, "loss": 0.011485453695058823, "num_tokens": 1181316.0, "reward": 5.718707084655762, "reward_std": 0.17716237902641296, "rewards/coordinate_accuracy_reward_func/mean": 0.6968316435813904, "rewards/coordinate_accuracy_reward_func/std": 0.1283807009458542, "rewards/graph_topology_reward_func/mean": 4.921875, "rewards/graph_topology_reward_func/std": 0.2183031290769577, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 64 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 723.5, "completions/clipped_ratio": 0.125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1177.0, "completions/mean_length": 723.5, "completions/mean_terminated_length": 541.1428833007812, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "epoch": 0.12218045112781954, "frac_reward_zero_std": 0.0, "grad_norm": 0.15604515373706818, "kl": 0.0, "learning_rate": 4.588719528532342e-06, "loss": 0.14646224677562714, "num_tokens": 1199180.0, "reward": 3.8986945152282715, "reward_std": 1.950113296508789, "rewards/coordinate_accuracy_reward_func/mean": 0.6895599961280823, "rewards/coordinate_accuracy_reward_func/std": 0.2706224024295807, "rewards/graph_topology_reward_func/mean": 3.146634578704834, "rewards/graph_topology_reward_func/std": 1.7150695323944092, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 65 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 844.125, "completions/clipped_ratio": 0.125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1400.0, "completions/mean_length": 844.125, "completions/mean_terminated_length": 679.0, "completions/min_length": 390.0, "completions/min_terminated_length": 390.0, "epoch": 0.12406015037593984, "frac_reward_zero_std": 0.0, "grad_norm": 0.15074588358402252, "kl": 0.0, "learning_rate": 4.568518602853776e-06, "loss": 0.18513138592243195, "num_tokens": 1218974.0, "reward": 5.08858585357666, "reward_std": 1.3887803554534912, "rewards/coordinate_accuracy_reward_func/mean": 0.7323356866836548, "rewards/coordinate_accuracy_reward_func/std": 0.2001328468322754, "rewards/graph_topology_reward_func/mean": 4.275000095367432, "rewards/graph_topology_reward_func/std": 1.2876335382461548, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 66 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1429.0625, "completions/clipped_ratio": 0.4375, "completions/max_length": 2000.0, "completions/max_terminated_length": 1867.0, "completions/mean_length": 1429.0625, "completions/mean_terminated_length": 985.0, "completions/min_length": 368.0, "completions/min_terminated_length": 368.0, "epoch": 0.12593984962406016, "frac_reward_zero_std": 0.0, "grad_norm": 0.24904194474220276, "kl": 0.0, "learning_rate": 4.54788011072248e-06, "loss": 0.4909200370311737, "num_tokens": 1248271.0, "reward": 2.8360483646392822, "reward_std": 2.6703362464904785, "rewards/coordinate_accuracy_reward_func/mean": 0.4735482931137085, "rewards/coordinate_accuracy_reward_func/std": 0.38267526030540466, "rewards/graph_topology_reward_func/mean": 2.375, "rewards/graph_topology_reward_func/std": 2.0936412811279297, "rewards/xml_formatting_reward_func/mean": -0.012500000186264515, "rewards/xml_formatting_reward_func/std": 0.15000000596046448, "step": 67 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 582.1875, "completions/clipped_ratio": 0.0, "completions/max_length": 1211.0, "completions/max_terminated_length": 1211.0, "completions/mean_length": 582.1875, "completions/mean_terminated_length": 582.1875, "completions/min_length": 369.0, "completions/min_terminated_length": 369.0, "epoch": 0.12781954887218044, "frac_reward_zero_std": 0.0, "grad_norm": 0.00783149991184473, "kl": 0.0, "learning_rate": 4.526808417925531e-06, "loss": 0.0033713490702211857, "num_tokens": 1265410.0, "reward": 5.53170108795166, "reward_std": 0.14254450798034668, "rewards/coordinate_accuracy_reward_func/mean": 0.6942012310028076, "rewards/coordinate_accuracy_reward_func/std": 0.09660802781581879, "rewards/graph_topology_reward_func/mean": 4.737500190734863, "rewards/graph_topology_reward_func/std": 0.30740848183631897, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 68 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 579.6875, "completions/clipped_ratio": 0.0, "completions/max_length": 1159.0, "completions/max_terminated_length": 1159.0, "completions/mean_length": 579.6875, "completions/mean_terminated_length": 579.6875, "completions/min_length": 381.0, "completions/min_terminated_length": 381.0, "epoch": 0.12969924812030076, "frac_reward_zero_std": 0.0, "grad_norm": 0.045327045023441315, "kl": 0.0, "learning_rate": 4.50530798188761e-06, "loss": 0.0046268668957054615, "num_tokens": 1281325.0, "reward": 4.991917610168457, "reward_std": 0.39091992378234863, "rewards/coordinate_accuracy_reward_func/mean": 0.7794175148010254, "rewards/coordinate_accuracy_reward_func/std": 0.034688711166381836, "rewards/graph_topology_reward_func/mean": 4.112500190734863, "rewards/graph_topology_reward_func/std": 0.8721939325332642, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 69 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 441.9375, "completions/clipped_ratio": 0.0, "completions/max_length": 895.0, "completions/max_terminated_length": 895.0, "completions/mean_length": 441.9375, "completions/mean_terminated_length": 441.9375, "completions/min_length": 290.0, "completions/min_terminated_length": 290.0, "epoch": 0.13157894736842105, "frac_reward_zero_std": 0.0, "grad_norm": 0.03369409963488579, "kl": 0.0, "learning_rate": 4.4833833507280884e-06, "loss": 0.012433069758117199, "num_tokens": 1294156.0, "reward": 5.659772872924805, "reward_std": 0.5469986200332642, "rewards/coordinate_accuracy_reward_func/mean": 0.7472731471061707, "rewards/coordinate_accuracy_reward_func/std": 0.05734746530652046, "rewards/graph_topology_reward_func/mean": 4.8125, "rewards/graph_topology_reward_func/std": 0.75, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 70 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 726.25, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1811.0, "completions/mean_length": 726.25, "completions/mean_terminated_length": 641.3333740234375, "completions/min_length": 262.0, "completions/min_terminated_length": 262.0, "epoch": 0.13345864661654136, "frac_reward_zero_std": 0.0, "grad_norm": 0.06025904417037964, "kl": 0.0, "learning_rate": 4.46103916229894e-06, "loss": 0.15248729288578033, "num_tokens": 1311360.0, "reward": 5.385895729064941, "reward_std": 1.0460965633392334, "rewards/coordinate_accuracy_reward_func/mean": 0.7108960151672363, "rewards/coordinate_accuracy_reward_func/std": 0.19815519452095032, "rewards/graph_topology_reward_func/mean": 4.59375, "rewards/graph_topology_reward_func/std": 1.2545750141143799, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 71 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 423.0, "completions/clipped_ratio": 0.0, "completions/max_length": 637.0, "completions/max_terminated_length": 637.0, "completions/mean_length": 423.0, "completions/mean_terminated_length": 423.0, "completions/min_length": 330.0, "completions/min_terminated_length": 330.0, "epoch": 0.13533834586466165, "frac_reward_zero_std": 0.0, "grad_norm": 0.03607521951198578, "kl": 0.0, "learning_rate": 4.438280143203665e-06, "loss": -0.0054114204831421375, "num_tokens": 1323712.0, "reward": 5.2677459716796875, "reward_std": 0.6807551383972168, "rewards/coordinate_accuracy_reward_func/mean": 0.7302459478378296, "rewards/coordinate_accuracy_reward_func/std": 0.07179661095142365, "rewards/graph_topology_reward_func/mean": 4.4375, "rewards/graph_topology_reward_func/std": 0.75, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 72 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 436.0625, "completions/clipped_ratio": 0.0, "completions/max_length": 835.0, "completions/max_terminated_length": 835.0, "completions/mean_length": 436.0625, "completions/mean_terminated_length": 436.0625, "completions/min_length": 230.0, "completions/min_terminated_length": 230.0, "epoch": 0.13721804511278196, "frac_reward_zero_std": 0.0, "grad_norm": 0.11193640530109406, "kl": 0.0, "learning_rate": 4.415111107797445e-06, "loss": 0.02248099073767662, "num_tokens": 1337409.0, "reward": 5.037131309509277, "reward_std": 1.4971914291381836, "rewards/coordinate_accuracy_reward_func/mean": 0.7371314167976379, "rewards/coordinate_accuracy_reward_func/std": 0.19967474043369293, "rewards/graph_topology_reward_func/mean": 4.21875, "rewards/graph_topology_reward_func/std": 1.356696367263794, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 73 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 444.125, "completions/clipped_ratio": 0.0, "completions/max_length": 770.0, "completions/max_terminated_length": 770.0, "completions/mean_length": 444.125, "completions/mean_terminated_length": 444.125, "completions/min_length": 252.0, "completions/min_terminated_length": 252.0, "epoch": 0.13909774436090225, "frac_reward_zero_std": 0.5, "grad_norm": 0.02847164124250412, "kl": 0.0, "learning_rate": 4.391536957168733e-06, "loss": 0.0011844169348478317, "num_tokens": 1350099.0, "reward": 5.633746147155762, "reward_std": 0.25180622935295105, "rewards/coordinate_accuracy_reward_func/mean": 0.7837458848953247, "rewards/coordinate_accuracy_reward_func/std": 0.03552357107400894, "rewards/graph_topology_reward_func/mean": 4.75, "rewards/graph_topology_reward_func/std": 0.44721361994743347, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 74 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 793.5625, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1952.0, "completions/mean_length": 793.5625, "completions/mean_terminated_length": 713.1333618164062, "completions/min_length": 275.0, "completions/min_terminated_length": 275.0, "epoch": 0.14097744360902256, "frac_reward_zero_std": 0.0, "grad_norm": 0.11117635667324066, "kl": 0.0, "learning_rate": 4.367562678102491e-06, "loss": 0.1720340996980667, "num_tokens": 1368908.0, "reward": 5.324819087982178, "reward_std": 1.0407084226608276, "rewards/coordinate_accuracy_reward_func/mean": 0.7435690760612488, "rewards/coordinate_accuracy_reward_func/std": 0.19857530295848846, "rewards/graph_topology_reward_func/mean": 4.5, "rewards/graph_topology_reward_func/std": 1.2712198495864868, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 75 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 347.125, "completions/clipped_ratio": 0.0, "completions/max_length": 600.0, "completions/max_terminated_length": 600.0, "completions/mean_length": 347.125, "completions/mean_terminated_length": 347.125, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "epoch": 0.14285714285714285, "frac_reward_zero_std": 0.5, "grad_norm": 0.034457214176654816, "kl": 0.0, "learning_rate": 4.34319334202531e-06, "loss": -0.001093149185180664, "num_tokens": 1380046.0, "reward": 5.708409309387207, "reward_std": 0.5288010835647583, "rewards/coordinate_accuracy_reward_func/mean": 0.7959089279174805, "rewards/coordinate_accuracy_reward_func/std": 0.016364559531211853, "rewards/graph_topology_reward_func/mean": 4.8125, "rewards/graph_topology_reward_func/std": 0.75, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 76 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 394.125, "completions/clipped_ratio": 0.0, "completions/max_length": 845.0, "completions/max_terminated_length": 845.0, "completions/mean_length": 394.125, "completions/mean_terminated_length": 394.125, "completions/min_length": 260.0, "completions/min_terminated_length": 260.0, "epoch": 0.14473684210526316, "frac_reward_zero_std": 0.0, "grad_norm": 0.09861151874065399, "kl": 0.0, "learning_rate": 4.318434103932622e-06, "loss": 0.0020296871662139893, "num_tokens": 1391936.0, "reward": 4.9148783683776855, "reward_std": 1.3600800037384033, "rewards/coordinate_accuracy_reward_func/mean": 0.7523783445358276, "rewards/coordinate_accuracy_reward_func/std": 0.05325964465737343, "rewards/graph_topology_reward_func/mean": 4.0625, "rewards/graph_topology_reward_func/std": 1.3275917768478394, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 77 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 478.5, "completions/clipped_ratio": 0.0, "completions/max_length": 1132.0, "completions/max_terminated_length": 1132.0, "completions/mean_length": 478.5, "completions/mean_terminated_length": 478.5, "completions/min_length": 281.0, "completions/min_terminated_length": 281.0, "epoch": 0.14661654135338345, "frac_reward_zero_std": 0.5, "grad_norm": 0.061929330229759216, "kl": 0.0, "learning_rate": 4.293290201298224e-06, "loss": 0.01751873269677162, "num_tokens": 1407416.0, "reward": 5.612481117248535, "reward_std": 0.3868803381919861, "rewards/coordinate_accuracy_reward_func/mean": 0.7937308549880981, "rewards/coordinate_accuracy_reward_func/std": 0.015431337058544159, "rewards/graph_topology_reward_func/mean": 4.71875, "rewards/graph_topology_reward_func/std": 0.6046693325042725, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 78 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 843.9375, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1678.0, "completions/mean_length": 843.9375, "completions/mean_terminated_length": 766.86669921875, "completions/min_length": 584.0, "completions/min_terminated_length": 584.0, "epoch": 0.14849624060150377, "frac_reward_zero_std": 0.0, "grad_norm": 0.23493611812591553, "kl": 0.0, "learning_rate": 4.267766952966369e-06, "loss": 0.16811293363571167, "num_tokens": 1427735.0, "reward": 4.790218353271484, "reward_std": 0.9868783950805664, "rewards/coordinate_accuracy_reward_func/mean": 0.7380332946777344, "rewards/coordinate_accuracy_reward_func/std": 0.19774849712848663, "rewards/graph_topology_reward_func/mean": 3.970935344696045, "rewards/graph_topology_reward_func/std": 1.1650488376617432, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 79 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 645.1875, "completions/clipped_ratio": 0.0, "completions/max_length": 942.0, "completions/max_terminated_length": 942.0, "completions/mean_length": 645.1875, "completions/mean_terminated_length": 645.1875, "completions/min_length": 494.0, "completions/min_terminated_length": 494.0, "epoch": 0.15037593984962405, "frac_reward_zero_std": 0.0, "grad_norm": 0.029840683564543724, "kl": 0.0, "learning_rate": 4.241869758026638e-06, "loss": -0.002414736896753311, "num_tokens": 1444490.0, "reward": 5.147690773010254, "reward_std": 0.3915814459323883, "rewards/coordinate_accuracy_reward_func/mean": 0.750815749168396, "rewards/coordinate_accuracy_reward_func/std": 0.06517978757619858, "rewards/graph_topology_reward_func/mean": 4.296875, "rewards/graph_topology_reward_func/std": 0.5789268016815186, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 80 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 869.5625, "completions/clipped_ratio": 0.125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1370.0, "completions/mean_length": 869.5625, "completions/mean_terminated_length": 708.0714721679688, "completions/min_length": 513.0, "completions/min_terminated_length": 513.0, "epoch": 0.15225563909774437, "frac_reward_zero_std": 0.0, "grad_norm": 0.17345793545246124, "kl": 0.0, "learning_rate": 4.215604094671835e-06, "loss": 0.23368486762046814, "num_tokens": 1466771.0, "reward": 4.584491729736328, "reward_std": 1.1114850044250488, "rewards/coordinate_accuracy_reward_func/mean": 0.6640374660491943, "rewards/coordinate_accuracy_reward_func/std": 0.2609242796897888, "rewards/graph_topology_reward_func/mean": 3.857954502105713, "rewards/graph_topology_reward_func/std": 1.6660503149032593, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 81 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 490.1875, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 788.0, "completions/mean_length": 490.1875, "completions/mean_terminated_length": 389.5333557128906, "completions/min_length": 243.0, "completions/min_terminated_length": 243.0, "epoch": 0.15413533834586465, "frac_reward_zero_std": 0.0, "grad_norm": 0.058151714503765106, "kl": 0.0, "learning_rate": 4.188975519039151e-06, "loss": 0.2854698896408081, "num_tokens": 1480198.0, "reward": 5.51440954208374, "reward_std": 1.0877041816711426, "rewards/coordinate_accuracy_reward_func/mean": 0.7456597685813904, "rewards/coordinate_accuracy_reward_func/std": 0.19930103421211243, "rewards/graph_topology_reward_func/mean": 4.6875, "rewards/graph_topology_reward_func/std": 1.25, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 82 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 760.375, "completions/clipped_ratio": 0.125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1147.0, "completions/mean_length": 760.375, "completions/mean_terminated_length": 583.2857666015625, "completions/min_length": 383.0, "completions/min_terminated_length": 383.0, "epoch": 0.15601503759398497, "frac_reward_zero_std": 0.0, "grad_norm": 0.3394562602043152, "kl": 0.0, "learning_rate": 4.161989664034844e-06, "loss": 0.27119284868240356, "num_tokens": 1498652.0, "reward": 3.38315749168396, "reward_std": 1.4021626710891724, "rewards/coordinate_accuracy_reward_func/mean": 0.6778002381324768, "rewards/coordinate_accuracy_reward_func/std": 0.26745790243148804, "rewards/graph_topology_reward_func/mean": 2.642857074737549, "rewards/graph_topology_reward_func/std": 1.26491117477417, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 83 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 425.25, "completions/clipped_ratio": 0.0, "completions/max_length": 561.0, "completions/max_terminated_length": 561.0, "completions/mean_length": 425.25, "completions/mean_terminated_length": 425.25, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "epoch": 0.15789473684210525, "frac_reward_zero_std": 0.0, "grad_norm": 0.02765234187245369, "kl": 0.0, "learning_rate": 4.134652238142674e-06, "loss": -0.00599747383967042, "num_tokens": 1512608.0, "reward": 3.693894624710083, "reward_std": 0.6546977758407593, "rewards/coordinate_accuracy_reward_func/mean": 0.6563946008682251, "rewards/coordinate_accuracy_reward_func/std": 0.21212811768054962, "rewards/graph_topology_reward_func/mean": 2.9375, "rewards/graph_topology_reward_func/std": 1.3307266235351562, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 84 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 733.1875, "completions/clipped_ratio": 0.0, "completions/max_length": 997.0, "completions/max_terminated_length": 997.0, "completions/mean_length": 733.1875, "completions/mean_terminated_length": 733.1875, "completions/min_length": 569.0, "completions/min_terminated_length": 569.0, "epoch": 0.15977443609022557, "frac_reward_zero_std": 0.0, "grad_norm": 0.0425846204161644, "kl": 0.0, "learning_rate": 4.106969024216348e-06, "loss": 0.014492754824459553, "num_tokens": 1532019.0, "reward": 5.234529495239258, "reward_std": 0.4689003825187683, "rewards/coordinate_accuracy_reward_func/mean": 0.650154173374176, "rewards/coordinate_accuracy_reward_func/std": 0.1325112283229828, "rewards/graph_topology_reward_func/mean": 4.484375, "rewards/graph_topology_reward_func/std": 0.6355361938476562, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 85 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 383.9375, "completions/clipped_ratio": 0.0, "completions/max_length": 478.0, "completions/max_terminated_length": 478.0, "completions/mean_length": 383.9375, "completions/mean_terminated_length": 383.9375, "completions/min_length": 286.0, "completions/min_terminated_length": 286.0, "epoch": 0.16165413533834586, "frac_reward_zero_std": 0.0, "grad_norm": 0.06096876785159111, "kl": 0.0, "learning_rate": 4.078945878256244e-06, "loss": 0.00455862283706665, "num_tokens": 1543746.0, "reward": 4.3796186447143555, "reward_std": 0.8436566591262817, "rewards/coordinate_accuracy_reward_func/mean": 0.7358686923980713, "rewards/coordinate_accuracy_reward_func/std": 0.20053471624851227, "rewards/graph_topology_reward_func/mean": 3.5625, "rewards/graph_topology_reward_func/std": 1.7500001192092896, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 86 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 634.75, "completions/clipped_ratio": 0.0, "completions/max_length": 1136.0, "completions/max_terminated_length": 1136.0, "completions/mean_length": 634.75, "completions/mean_terminated_length": 634.75, "completions/min_length": 359.0, "completions/min_terminated_length": 359.0, "epoch": 0.16353383458646617, "frac_reward_zero_std": 0.0, "grad_norm": 0.02554352954030037, "kl": 0.0, "learning_rate": 4.0505887281706505e-06, "loss": -0.008525269106030464, "num_tokens": 1560014.0, "reward": 5.268431663513184, "reward_std": 0.36519530415534973, "rewards/coordinate_accuracy_reward_func/mean": 0.6934319138526917, "rewards/coordinate_accuracy_reward_func/std": 0.15953516960144043, "rewards/graph_topology_reward_func/mean": 4.474999904632568, "rewards/graph_topology_reward_func/std": 0.6526867747306824, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 87 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 777.875, "completions/clipped_ratio": 0.1875, "completions/max_length": 2000.0, "completions/max_terminated_length": 1075.0, "completions/mean_length": 777.875, "completions/mean_terminated_length": 495.8461608886719, "completions/min_length": 288.0, "completions/min_terminated_length": 288.0, "epoch": 0.16541353383458646, "frac_reward_zero_std": 0.5, "grad_norm": 0.13790547847747803, "kl": 0.0, "learning_rate": 4.021903572521802e-06, "loss": 0.18941883742809296, "num_tokens": 1578748.0, "reward": 4.489029884338379, "reward_std": 1.0317115783691406, "rewards/coordinate_accuracy_reward_func/mean": 0.6854585409164429, "rewards/coordinate_accuracy_reward_func/std": 0.27036699652671814, "rewards/graph_topology_reward_func/mean": 3.7410714626312256, "rewards/graph_topology_reward_func/std": 1.6977126598358154, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 88 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 808.8125, "completions/clipped_ratio": 0.0, "completions/max_length": 1669.0, "completions/max_terminated_length": 1669.0, "completions/mean_length": 808.8125, "completions/mean_terminated_length": 808.8125, "completions/min_length": 601.0, "completions/min_terminated_length": 601.0, "epoch": 0.16729323308270677, "frac_reward_zero_std": 0.0, "grad_norm": 0.1298065036535263, "kl": 0.0, "learning_rate": 3.992896479256966e-06, "loss": 0.02378976345062256, "num_tokens": 1599193.0, "reward": 4.746280193328857, "reward_std": 1.2782974243164062, "rewards/coordinate_accuracy_reward_func/mean": 0.7082992196083069, "rewards/coordinate_accuracy_reward_func/std": 0.1931397169828415, "rewards/graph_topology_reward_func/mean": 3.956730842590332, "rewards/graph_topology_reward_func/std": 1.170889139175415, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 89 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 676.6875, "completions/clipped_ratio": 0.0, "completions/max_length": 1193.0, "completions/max_terminated_length": 1193.0, "completions/mean_length": 676.6875, "completions/mean_terminated_length": 676.6875, "completions/min_length": 557.0, "completions/min_terminated_length": 557.0, "epoch": 0.16917293233082706, "frac_reward_zero_std": 0.0, "grad_norm": 0.03102908469736576, "kl": 0.0, "learning_rate": 3.963573584424852e-06, "loss": -0.014245279133319855, "num_tokens": 1616836.0, "reward": 4.83237886428833, "reward_std": 0.6284863352775574, "rewards/coordinate_accuracy_reward_func/mean": 0.7551060914993286, "rewards/coordinate_accuracy_reward_func/std": 0.08354499191045761, "rewards/graph_topology_reward_func/mean": 3.9772725105285645, "rewards/graph_topology_reward_func/std": 0.7394883036613464, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 90 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 609.4375, "completions/clipped_ratio": 0.0, "completions/max_length": 1539.0, "completions/max_terminated_length": 1539.0, "completions/mean_length": 609.4375, "completions/mean_terminated_length": 609.4375, "completions/min_length": 315.0, "completions/min_terminated_length": 315.0, "epoch": 0.17105263157894737, "frac_reward_zero_std": 0.0, "grad_norm": 0.10814845561981201, "kl": 0.0, "learning_rate": 3.933941090877615e-06, "loss": 0.007238894701004028, "num_tokens": 1632875.0, "reward": 4.725651741027832, "reward_std": 0.917718768119812, "rewards/coordinate_accuracy_reward_func/mean": 0.77773517370224, "rewards/coordinate_accuracy_reward_func/std": 0.05891840532422066, "rewards/graph_topology_reward_func/mean": 3.847916603088379, "rewards/graph_topology_reward_func/std": 0.9604759216308594, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 91 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 494.9375, "completions/clipped_ratio": 0.0, "completions/max_length": 688.0, "completions/max_terminated_length": 688.0, "completions/mean_length": 494.9375, "completions/mean_terminated_length": 494.9375, "completions/min_length": 369.0, "completions/min_terminated_length": 369.0, "epoch": 0.17293233082706766, "frac_reward_zero_std": 0.0, "grad_norm": 0.030460018664598465, "kl": 0.0, "learning_rate": 3.9040052669587325e-06, "loss": -0.0013976418413221836, "num_tokens": 1647258.0, "reward": 5.704535484313965, "reward_std": 0.3163030445575714, "rewards/coordinate_accuracy_reward_func/mean": 0.7139102816581726, "rewards/coordinate_accuracy_reward_func/std": 0.13340230286121368, "rewards/graph_topology_reward_func/mean": 4.890625, "rewards/graph_topology_reward_func/std": 0.30233466625213623, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 92 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 693.6875, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1561.0, "completions/mean_length": 693.6875, "completions/mean_terminated_length": 606.6000366210938, "completions/min_length": 271.0, "completions/min_terminated_length": 271.0, "epoch": 0.17481203007518797, "frac_reward_zero_std": 0.0, "grad_norm": 0.17381823062896729, "kl": 0.0, "learning_rate": 3.8737724451770155e-06, "loss": 0.11337310075759888, "num_tokens": 1664117.0, "reward": 4.308534145355225, "reward_std": 1.8105342388153076, "rewards/coordinate_accuracy_reward_func/mean": 0.6975637078285217, "rewards/coordinate_accuracy_reward_func/std": 0.2723815143108368, "rewards/graph_topology_reward_func/mean": 3.5297203063964844, "rewards/graph_topology_reward_func/std": 1.70619535446167, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 93 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 517.75, "completions/clipped_ratio": 0.0, "completions/max_length": 824.0, "completions/max_terminated_length": 824.0, "completions/mean_length": 517.75, "completions/mean_terminated_length": 517.75, "completions/min_length": 445.0, "completions/min_terminated_length": 445.0, "epoch": 0.17669172932330826, "frac_reward_zero_std": 0.0, "grad_norm": 0.06615965813398361, "kl": 0.0, "learning_rate": 3.8432490208670605e-06, "loss": -0.0001715589314699173, "num_tokens": 1678161.0, "reward": 4.905134201049805, "reward_std": 0.43666183948516846, "rewards/coordinate_accuracy_reward_func/mean": 0.7895090579986572, "rewards/coordinate_accuracy_reward_func/std": 0.02889779955148697, "rewards/graph_topology_reward_func/mean": 4.015625, "rewards/graph_topology_reward_func/std": 0.45155981183052063, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 94 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 466.8125, "completions/clipped_ratio": 0.0, "completions/max_length": 625.0, "completions/max_terminated_length": 625.0, "completions/mean_length": 466.8125, "completions/mean_terminated_length": 466.8125, "completions/min_length": 355.0, "completions/min_terminated_length": 355.0, "epoch": 0.17857142857142858, "frac_reward_zero_std": 0.0, "grad_norm": 0.049933284521102905, "kl": 0.0, "learning_rate": 3.8124414508364005e-06, "loss": 0.010361225344240665, "num_tokens": 1692910.0, "reward": 5.224326133728027, "reward_std": 0.7026044130325317, "rewards/coordinate_accuracy_reward_func/mean": 0.749326229095459, "rewards/coordinate_accuracy_reward_func/std": 0.09528137743473053, "rewards/graph_topology_reward_func/mean": 4.375, "rewards/graph_topology_reward_func/std": 0.670820415019989, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 95 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 987.5625, "completions/clipped_ratio": 0.3125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1384.0, "completions/mean_length": 987.5625, "completions/mean_terminated_length": 527.3636474609375, "completions/min_length": 274.0, "completions/min_terminated_length": 274.0, "epoch": 0.18045112781954886, "frac_reward_zero_std": 0.5, "grad_norm": 0.13480573892593384, "kl": 0.0, "learning_rate": 3.7813562519996633e-06, "loss": 0.24449783563613892, "num_tokens": 1714591.0, "reward": 3.7636611461639404, "reward_std": 1.094565749168396, "rewards/coordinate_accuracy_reward_func/mean": 0.5170265436172485, "rewards/coordinate_accuracy_reward_func/std": 0.3714961111545563, "rewards/graph_topology_reward_func/mean": 3.22163462638855, "rewards/graph_topology_reward_func/std": 2.1820483207702637, "rewards/xml_formatting_reward_func/mean": 0.02500000037252903, "rewards/xml_formatting_reward_func/std": 0.13416409492492676, "step": 96 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 463.5625, "completions/clipped_ratio": 0.0, "completions/max_length": 833.0, "completions/max_terminated_length": 833.0, "completions/mean_length": 463.5625, "completions/mean_terminated_length": 463.5625, "completions/min_length": 261.0, "completions/min_terminated_length": 261.0, "epoch": 0.18233082706766918, "frac_reward_zero_std": 0.0, "grad_norm": 0.029042627662420273, "kl": 0.0, "learning_rate": 3.7500000000000005e-06, "loss": -0.0014519591350108385, "num_tokens": 1728120.0, "reward": 5.447628974914551, "reward_std": 0.22185976803302765, "rewards/coordinate_accuracy_reward_func/mean": 0.7851289510726929, "rewards/coordinate_accuracy_reward_func/std": 0.03899515047669411, "rewards/graph_topology_reward_func/mean": 4.5625, "rewards/graph_topology_reward_func/std": 0.5426785349845886, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 97 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1102.0625, "completions/clipped_ratio": 0.25, "completions/max_length": 2000.0, "completions/max_terminated_length": 1754.0, "completions/mean_length": 1102.0625, "completions/mean_terminated_length": 802.75, "completions/min_length": 456.0, "completions/min_terminated_length": 456.0, "epoch": 0.18421052631578946, "frac_reward_zero_std": 0.0, "grad_norm": 0.1899338811635971, "kl": 0.0, "learning_rate": 3.7183793278181063e-06, "loss": 0.2890927195549011, "num_tokens": 1752745.0, "reward": 3.571110725402832, "reward_std": 1.3117281198501587, "rewards/coordinate_accuracy_reward_func/mean": 0.5764137506484985, "rewards/coordinate_accuracy_reward_func/std": 0.3464476764202118, "rewards/graph_topology_reward_func/mean": 2.9696969985961914, "rewards/graph_topology_reward_func/std": 1.8334864377975464, "rewards/xml_formatting_reward_func/mean": 0.02500000037252903, "rewards/xml_formatting_reward_func/std": 0.13416409492492676, "step": 98 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 733.5, "completions/clipped_ratio": 0.0, "completions/max_length": 1412.0, "completions/max_terminated_length": 1412.0, "completions/mean_length": 733.5, "completions/mean_terminated_length": 733.5, "completions/min_length": 571.0, "completions/min_terminated_length": 571.0, "epoch": 0.18609022556390978, "frac_reward_zero_std": 0.0, "grad_norm": 0.03173709660768509, "kl": 0.0, "learning_rate": 3.6865009243691015e-06, "loss": -0.0038963891565799713, "num_tokens": 1771473.0, "reward": 5.194991111755371, "reward_std": 0.39840370416641235, "rewards/coordinate_accuracy_reward_func/mean": 0.7427181005477905, "rewards/coordinate_accuracy_reward_func/std": 0.10078544914722443, "rewards/graph_topology_reward_func/mean": 4.352272987365723, "rewards/graph_topology_reward_func/std": 0.420055091381073, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 99 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 477.3125, "completions/clipped_ratio": 0.0, "completions/max_length": 935.0, "completions/max_terminated_length": 935.0, "completions/mean_length": 477.3125, "completions/mean_terminated_length": 477.3125, "completions/min_length": 312.0, "completions/min_terminated_length": 312.0, "epoch": 0.18796992481203006, "frac_reward_zero_std": 0.0, "grad_norm": 0.03984377533197403, "kl": 0.0, "learning_rate": 3.654371533087586e-06, "loss": 0.006507607642561197, "num_tokens": 1785254.0, "reward": 5.138284206390381, "reward_std": 0.5264720916748047, "rewards/coordinate_accuracy_reward_func/mean": 0.7882839441299438, "rewards/coordinate_accuracy_reward_func/std": 0.025554845109581947, "rewards/graph_topology_reward_func/mean": 4.25, "rewards/graph_topology_reward_func/std": 0.7745966911315918, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 545.75, "completions/clipped_ratio": 0.0, "completions/max_length": 764.0, "completions/max_terminated_length": 764.0, "completions/mean_length": 545.75, "completions/mean_terminated_length": 545.75, "completions/min_length": 461.0, "completions/min_terminated_length": 461.0, "epoch": 0.18984962406015038, "frac_reward_zero_std": 0.0, "grad_norm": 0.03931056708097458, "kl": 0.0, "learning_rate": 3.621997950501156e-06, "loss": -0.0001787245273590088, "num_tokens": 1801138.0, "reward": 5.299640655517578, "reward_std": 0.6725379824638367, "rewards/coordinate_accuracy_reward_func/mean": 0.7152658700942993, "rewards/coordinate_accuracy_reward_func/std": 0.08963414281606674, "rewards/graph_topology_reward_func/mean": 4.484375, "rewards/graph_topology_reward_func/std": 0.7608588933944702, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 101 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 665.5625, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1269.0, "completions/mean_length": 665.5625, "completions/mean_terminated_length": 576.6000366210938, "completions/min_length": 241.0, "completions/min_terminated_length": 241.0, "epoch": 0.19172932330827067, "frac_reward_zero_std": 0.0, "grad_norm": 0.23963890969753265, "kl": 0.0, "learning_rate": 3.5893870247926986e-06, "loss": 0.15939615666866302, "num_tokens": 1818075.0, "reward": 4.841655254364014, "reward_std": 1.449117660522461, "rewards/coordinate_accuracy_reward_func/mean": 0.7177915573120117, "rewards/coordinate_accuracy_reward_func/std": 0.21610607206821442, "rewards/graph_topology_reward_func/mean": 4.042613506317139, "rewards/graph_topology_reward_func/std": 1.3845336437225342, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 102 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 418.625, "completions/clipped_ratio": 0.0, "completions/max_length": 704.0, "completions/max_terminated_length": 704.0, "completions/mean_length": 418.625, "completions/mean_terminated_length": 418.625, "completions/min_length": 266.0, "completions/min_terminated_length": 266.0, "epoch": 0.19360902255639098, "frac_reward_zero_std": 0.5, "grad_norm": 0.08563794195652008, "kl": 0.0, "learning_rate": 3.556545654351749e-06, "loss": -0.0065102651715278625, "num_tokens": 1831061.0, "reward": 5.50631046295166, "reward_std": 1.0734306573867798, "rewards/coordinate_accuracy_reward_func/mean": 0.7375602722167969, "rewards/coordinate_accuracy_reward_func/std": 0.19840113818645477, "rewards/graph_topology_reward_func/mean": 4.6875, "rewards/graph_topology_reward_func/std": 1.25, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 103 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 957.5, "completions/clipped_ratio": 0.25, "completions/max_length": 2000.0, "completions/max_terminated_length": 1115.0, "completions/mean_length": 957.5, "completions/mean_terminated_length": 610.0, "completions/min_length": 514.0, "completions/min_terminated_length": 514.0, "epoch": 0.19548872180451127, "frac_reward_zero_std": 0.0, "grad_norm": 0.2585947811603546, "kl": 0.0, "learning_rate": 3.5234807863152316e-06, "loss": 0.37977445125579834, "num_tokens": 1855101.0, "reward": 3.197767496109009, "reward_std": 1.3248279094696045, "rewards/coordinate_accuracy_reward_func/mean": 0.5425591468811035, "rewards/coordinate_accuracy_reward_func/std": 0.33239981532096863, "rewards/graph_topology_reward_func/mean": 2.6302082538604736, "rewards/graph_topology_reward_func/std": 1.5825930833816528, "rewards/xml_formatting_reward_func/mean": 0.02500000037252903, "rewards/xml_formatting_reward_func/std": 0.13416409492492676, "step": 104 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 516.625, "completions/clipped_ratio": 0.0, "completions/max_length": 1481.0, "completions/max_terminated_length": 1481.0, "completions/mean_length": 516.625, "completions/mean_terminated_length": 516.625, "completions/min_length": 281.0, "completions/min_terminated_length": 281.0, "epoch": 0.19736842105263158, "frac_reward_zero_std": 0.0, "grad_norm": 0.019215580075979233, "kl": 0.0, "learning_rate": 3.4901994150978926e-06, "loss": 0.0008287392556667328, "num_tokens": 1869623.0, "reward": 5.7782883644104, "reward_std": 0.2780331075191498, "rewards/coordinate_accuracy_reward_func/mean": 0.7782881855964661, "rewards/coordinate_accuracy_reward_func/std": 0.044969405978918076, "rewards/graph_topology_reward_func/mean": 4.900000095367432, "rewards/graph_topology_reward_func/std": 0.2828426957130432, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 543.25, "completions/clipped_ratio": 0.0, "completions/max_length": 900.0, "completions/max_terminated_length": 900.0, "completions/mean_length": 543.25, "completions/mean_terminated_length": 543.25, "completions/min_length": 324.0, "completions/min_terminated_length": 324.0, "epoch": 0.19924812030075187, "frac_reward_zero_std": 0.0, "grad_norm": 0.01700744219124317, "kl": 0.0, "learning_rate": 3.4567085809127247e-06, "loss": 0.0038414536975324154, "num_tokens": 1885307.0, "reward": 5.697380065917969, "reward_std": 0.17743521928787231, "rewards/coordinate_accuracy_reward_func/mean": 0.7848798632621765, "rewards/coordinate_accuracy_reward_func/std": 0.02732839621603489, "rewards/graph_topology_reward_func/mean": 4.8125, "rewards/graph_topology_reward_func/std": 0.2872281074523926, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 682.6875, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1195.0, "completions/mean_length": 682.6875, "completions/mean_terminated_length": 594.86669921875, "completions/min_length": 354.0, "completions/min_terminated_length": 354.0, "epoch": 0.20112781954887218, "frac_reward_zero_std": 0.0, "grad_norm": 0.1122838705778122, "kl": 0.0, "learning_rate": 3.4230153682817112e-06, "loss": 0.18668541312217712, "num_tokens": 1902166.0, "reward": 5.030940055847168, "reward_std": 1.367864727973938, "rewards/coordinate_accuracy_reward_func/mean": 0.6871896386146545, "rewards/coordinate_accuracy_reward_func/std": 0.26855412125587463, "rewards/graph_topology_reward_func/mean": 4.28125, "rewards/graph_topology_reward_func/std": 1.6928157806396484, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 776.8125, "completions/clipped_ratio": 0.0, "completions/max_length": 1429.0, "completions/max_terminated_length": 1429.0, "completions/mean_length": 776.8125, "completions/mean_terminated_length": 776.8125, "completions/min_length": 490.0, "completions/min_terminated_length": 490.0, "epoch": 0.20300751879699247, "frac_reward_zero_std": 0.0, "grad_norm": 0.15962724387645721, "kl": 0.0, "learning_rate": 3.389126904537192e-06, "loss": 0.08957692980766296, "num_tokens": 1922131.0, "reward": 4.641499996185303, "reward_std": 1.2996433973312378, "rewards/coordinate_accuracy_reward_func/mean": 0.6258751153945923, "rewards/coordinate_accuracy_reward_func/std": 0.2598617374897003, "rewards/graph_topology_reward_func/mean": 3.953125, "rewards/graph_topology_reward_func/std": 1.6335512399673462, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 108 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 597.125, "completions/clipped_ratio": 0.0, "completions/max_length": 1030.0, "completions/max_terminated_length": 1030.0, "completions/mean_length": 597.125, "completions/mean_terminated_length": 597.125, "completions/min_length": 428.0, "completions/min_terminated_length": 428.0, "epoch": 0.20488721804511278, "frac_reward_zero_std": 0.0, "grad_norm": 0.04359513893723488, "kl": 0.0, "learning_rate": 3.3550503583141726e-06, "loss": 0.008886045776307583, "num_tokens": 1938677.0, "reward": 5.513735771179199, "reward_std": 0.31733420491218567, "rewards/coordinate_accuracy_reward_func/mean": 0.7762353420257568, "rewards/coordinate_accuracy_reward_func/std": 0.05075686052441597, "rewards/graph_topology_reward_func/mean": 4.637499809265137, "rewards/graph_topology_reward_func/std": 0.5572252869606018, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 109 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 704.375, "completions/clipped_ratio": 0.0, "completions/max_length": 1020.0, "completions/max_terminated_length": 1020.0, "completions/mean_length": 704.375, "completions/mean_terminated_length": 704.375, "completions/min_length": 516.0, "completions/min_terminated_length": 516.0, "epoch": 0.20676691729323307, "frac_reward_zero_std": 0.0, "grad_norm": 0.08052542060613632, "kl": 0.0, "learning_rate": 3.3207929380339034e-06, "loss": 0.004702674224972725, "num_tokens": 1956587.0, "reward": 4.780728340148926, "reward_std": 0.6317276954650879, "rewards/coordinate_accuracy_reward_func/mean": 0.7597491145133972, "rewards/coordinate_accuracy_reward_func/std": 0.07506606727838516, "rewards/graph_topology_reward_func/mean": 3.9209790229797363, "rewards/graph_topology_reward_func/std": 0.6787975430488586, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 110 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1148.875, "completions/clipped_ratio": 0.25, "completions/max_length": 2000.0, "completions/max_terminated_length": 1394.0, "completions/mean_length": 1148.875, "completions/mean_terminated_length": 865.1666870117188, "completions/min_length": 601.0, "completions/min_terminated_length": 601.0, "epoch": 0.20864661654135339, "frac_reward_zero_std": 0.0, "grad_norm": 0.17186102271080017, "kl": 0.0, "learning_rate": 3.2863618903790346e-06, "loss": 0.3771874010562897, "num_tokens": 1982825.0, "reward": 3.5402417182922363, "reward_std": 1.425986886024475, "rewards/coordinate_accuracy_reward_func/mean": 0.543607234954834, "rewards/coordinate_accuracy_reward_func/std": 0.34286805987358093, "rewards/graph_topology_reward_func/mean": 2.9716343879699707, "rewards/graph_topology_reward_func/std": 1.7980194091796875, "rewards/xml_formatting_reward_func/mean": 0.02500000037252903, "rewards/xml_formatting_reward_func/std": 0.13416409492492676, "step": 111 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 413.4375, "completions/clipped_ratio": 0.0, "completions/max_length": 603.0, "completions/max_terminated_length": 603.0, "completions/mean_length": 413.4375, "completions/mean_terminated_length": 413.4375, "completions/min_length": 313.0, "completions/min_terminated_length": 313.0, "epoch": 0.21052631578947367, "frac_reward_zero_std": 0.0, "grad_norm": 0.018049633130431175, "kl": 0.0, "learning_rate": 3.2517644987606827e-06, "loss": 0.0019588605500757694, "num_tokens": 1996016.0, "reward": 5.211120128631592, "reward_std": 0.2952108681201935, "rewards/coordinate_accuracy_reward_func/mean": 0.7673702239990234, "rewards/coordinate_accuracy_reward_func/std": 0.04208545386791229, "rewards/graph_topology_reward_func/mean": 4.34375, "rewards/graph_topology_reward_func/std": 0.7685213088989258, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 112 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 652.4375, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1233.0, "completions/mean_length": 652.4375, "completions/mean_terminated_length": 562.6000366210938, "completions/min_length": 424.0, "completions/min_terminated_length": 424.0, "epoch": 0.212406015037594, "frac_reward_zero_std": 0.0, "grad_norm": 0.09911539405584335, "kl": 0.0, "learning_rate": 3.217008081777726e-06, "loss": 0.20954665541648865, "num_tokens": 2013175.0, "reward": 4.788642883300781, "reward_std": 1.3177140951156616, "rewards/coordinate_accuracy_reward_func/mean": 0.7198928594589233, "rewards/coordinate_accuracy_reward_func/std": 0.19539377093315125, "rewards/graph_topology_reward_func/mean": 3.987499952316284, "rewards/graph_topology_reward_func/std": 1.2098898887634277, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 113 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 494.8125, "completions/clipped_ratio": 0.0, "completions/max_length": 709.0, "completions/max_terminated_length": 709.0, "completions/mean_length": 494.8125, "completions/mean_terminated_length": 494.8125, "completions/min_length": 314.0, "completions/min_terminated_length": 314.0, "epoch": 0.21428571428571427, "frac_reward_zero_std": 0.0, "grad_norm": 0.018193848431110382, "kl": 0.0, "learning_rate": 3.182099991668653e-06, "loss": -0.0034578945487737656, "num_tokens": 2027956.0, "reward": 5.349656581878662, "reward_std": 0.2180614471435547, "rewards/coordinate_accuracy_reward_func/mean": 0.7638610601425171, "rewards/coordinate_accuracy_reward_func/std": 0.05042881891131401, "rewards/graph_topology_reward_func/mean": 4.485795497894287, "rewards/graph_topology_reward_func/std": 0.5909600853919983, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 114 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 524.8125, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 545.0, "completions/mean_length": 524.8125, "completions/mean_terminated_length": 426.4666748046875, "completions/min_length": 256.0, "completions/min_terminated_length": 256.0, "epoch": 0.2161654135338346, "frac_reward_zero_std": 0.0, "grad_norm": 0.09459604322910309, "kl": 0.0, "learning_rate": 3.147047612756302e-06, "loss": 0.16467620432376862, "num_tokens": 2042641.0, "reward": 4.142458915710449, "reward_std": 1.5458855628967285, "rewards/coordinate_accuracy_reward_func/mean": 0.7435007095336914, "rewards/coordinate_accuracy_reward_func/std": 0.19925178587436676, "rewards/graph_topology_reward_func/mean": 3.3177082538604736, "rewards/graph_topology_reward_func/std": 1.4476094245910645, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 625.625, "completions/clipped_ratio": 0.0, "completions/max_length": 1318.0, "completions/max_terminated_length": 1318.0, "completions/mean_length": 625.625, "completions/mean_terminated_length": 625.625, "completions/min_length": 393.0, "completions/min_terminated_length": 393.0, "epoch": 0.21804511278195488, "frac_reward_zero_std": 0.0, "grad_norm": 0.018846703693270683, "kl": 0.0, "learning_rate": 3.1118583598858097e-06, "loss": 0.007049616426229477, "num_tokens": 2058939.0, "reward": 5.1254706382751465, "reward_std": 0.3533204197883606, "rewards/coordinate_accuracy_reward_func/mean": 0.650470495223999, "rewards/coordinate_accuracy_reward_func/std": 0.11946040391921997, "rewards/graph_topology_reward_func/mean": 4.375, "rewards/graph_topology_reward_func/std": 0.4281744360923767, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 116 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 643.8125, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1192.0, "completions/mean_length": 643.8125, "completions/mean_terminated_length": 553.4000244140625, "completions/min_length": 313.0, "completions/min_terminated_length": 313.0, "epoch": 0.2199248120300752, "frac_reward_zero_std": 0.0, "grad_norm": 0.11112863570451736, "kl": 0.0, "learning_rate": 3.0765396768561005e-06, "loss": 0.27851736545562744, "num_tokens": 2075704.0, "reward": 5.286303997039795, "reward_std": 1.2659155130386353, "rewards/coordinate_accuracy_reward_func/mean": 0.7050539255142212, "rewards/coordinate_accuracy_reward_func/std": 0.20165081322193146, "rewards/graph_topology_reward_func/mean": 4.5, "rewards/graph_topology_reward_func/std": 1.2449899911880493, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 117 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 665.9375, "completions/clipped_ratio": 0.0, "completions/max_length": 1521.0, "completions/max_terminated_length": 1521.0, "completions/mean_length": 665.9375, "completions/mean_terminated_length": 665.9375, "completions/min_length": 349.0, "completions/min_terminated_length": 349.0, "epoch": 0.22180451127819548, "frac_reward_zero_std": 0.0, "grad_norm": 0.03289644420146942, "kl": 0.0, "learning_rate": 3.0410990348452572e-06, "loss": 0.0077062007039785385, "num_tokens": 2092791.0, "reward": 5.291324615478516, "reward_std": 0.2650687098503113, "rewards/coordinate_accuracy_reward_func/mean": 0.7634400725364685, "rewards/coordinate_accuracy_reward_func/std": 0.060397855937480927, "rewards/graph_topology_reward_func/mean": 4.427884578704834, "rewards/graph_topology_reward_func/std": 0.6630387306213379, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 118 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 460.6875, "completions/clipped_ratio": 0.0, "completions/max_length": 956.0, "completions/max_terminated_length": 956.0, "completions/mean_length": 460.6875, "completions/mean_terminated_length": 460.6875, "completions/min_length": 240.0, "completions/min_terminated_length": 240.0, "epoch": 0.2236842105263158, "frac_reward_zero_std": 0.5, "grad_norm": 0.004578461404889822, "kl": 0.0, "learning_rate": 3.0055439308300954e-06, "loss": -0.0007549161091446877, "num_tokens": 2106882.0, "reward": 5.865910530090332, "reward_std": 0.03279118612408638, "rewards/coordinate_accuracy_reward_func/mean": 0.7659105062484741, "rewards/coordinate_accuracy_reward_func/std": 0.05698000267148018, "rewards/graph_topology_reward_func/mean": 5.0, "rewards/graph_topology_reward_func/std": 0.0, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 119 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 579.3125, "completions/clipped_ratio": 0.0, "completions/max_length": 1918.0, "completions/max_terminated_length": 1918.0, "completions/mean_length": 579.3125, "completions/mean_terminated_length": 579.3125, "completions/min_length": 302.0, "completions/min_terminated_length": 302.0, "epoch": 0.22556390977443608, "frac_reward_zero_std": 0.0, "grad_norm": 0.0727643296122551, "kl": 0.0, "learning_rate": 2.96988188600028e-06, "loss": -0.019806761294603348, "num_tokens": 2122407.0, "reward": 5.494058132171631, "reward_std": 0.6452251076698303, "rewards/coordinate_accuracy_reward_func/mean": 0.7690579891204834, "rewards/coordinate_accuracy_reward_func/std": 0.07416322082281113, "rewards/graph_topology_reward_func/mean": 4.625, "rewards/graph_topology_reward_func/std": 0.670820415019989, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 120 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 472.9375, "completions/clipped_ratio": 0.0, "completions/max_length": 718.0, "completions/max_terminated_length": 718.0, "completions/mean_length": 472.9375, "completions/mean_terminated_length": 472.9375, "completions/min_length": 282.0, "completions/min_terminated_length": 282.0, "epoch": 0.2274436090225564, "frac_reward_zero_std": 0.5, "grad_norm": 0.03178076446056366, "kl": 0.0, "learning_rate": 2.9341204441673267e-06, "loss": -0.0018915310502052307, "num_tokens": 2135734.0, "reward": 4.140468597412109, "reward_std": 0.27307915687561035, "rewards/coordinate_accuracy_reward_func/mean": 0.7904680967330933, "rewards/coordinate_accuracy_reward_func/std": 0.01598774455487728, "rewards/graph_topology_reward_func/mean": 3.25, "rewards/graph_topology_reward_func/std": 1.341640830039978, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 121 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 426.4375, "completions/clipped_ratio": 0.0, "completions/max_length": 762.0, "completions/max_terminated_length": 762.0, "completions/mean_length": 426.4375, "completions/mean_terminated_length": 426.4375, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "epoch": 0.22932330827067668, "frac_reward_zero_std": 0.5, "grad_norm": 0.01970127411186695, "kl": 0.0, "learning_rate": 2.898267170168807e-06, "loss": 0.006058077793568373, "num_tokens": 2148845.0, "reward": 5.76971435546875, "reward_std": 0.23257319629192352, "rewards/coordinate_accuracy_reward_func/mean": 0.7947139739990234, "rewards/coordinate_accuracy_reward_func/std": 0.013148589059710503, "rewards/graph_topology_reward_func/mean": 4.875, "rewards/graph_topology_reward_func/std": 0.3415650427341461, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 122 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 592.4375, "completions/clipped_ratio": 0.0, "completions/max_length": 1838.0, "completions/max_terminated_length": 1838.0, "completions/mean_length": 592.4375, "completions/mean_terminated_length": 592.4375, "completions/min_length": 322.0, "completions/min_terminated_length": 322.0, "epoch": 0.231203007518797, "frac_reward_zero_std": 0.5, "grad_norm": 0.016404060646891594, "kl": 0.0, "learning_rate": 2.862329648268117e-06, "loss": 0.01149224303662777, "num_tokens": 2165284.0, "reward": 5.444645881652832, "reward_std": 0.16667158901691437, "rewards/coordinate_accuracy_reward_func/mean": 0.7946456670761108, "rewards/coordinate_accuracy_reward_func/std": 0.021417245268821716, "rewards/graph_topology_reward_func/mean": 4.550000190734863, "rewards/graph_topology_reward_func/std": 0.513809323310852, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 123 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 568.125, "completions/clipped_ratio": 0.0, "completions/max_length": 937.0, "completions/max_terminated_length": 937.0, "completions/mean_length": 568.125, "completions/mean_terminated_length": 568.125, "completions/min_length": 400.0, "completions/min_terminated_length": 400.0, "epoch": 0.23308270676691728, "frac_reward_zero_std": 0.0, "grad_norm": 0.037805359810590744, "kl": 0.0, "learning_rate": 2.82631548055013e-06, "loss": 0.0003531182010192424, "num_tokens": 2180486.0, "reward": 5.399947643280029, "reward_std": 0.49450063705444336, "rewards/coordinate_accuracy_reward_func/mean": 0.7686973810195923, "rewards/coordinate_accuracy_reward_func/std": 0.03766566514968872, "rewards/graph_topology_reward_func/mean": 4.53125, "rewards/graph_topology_reward_func/std": 0.4989572763442993, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 124 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 570.25, "completions/clipped_ratio": 0.0, "completions/max_length": 878.0, "completions/max_terminated_length": 878.0, "completions/mean_length": 570.25, "completions/mean_terminated_length": 570.25, "completions/min_length": 385.0, "completions/min_terminated_length": 385.0, "epoch": 0.2349624060150376, "frac_reward_zero_std": 0.0, "grad_norm": 0.0476958230137825, "kl": 0.0, "learning_rate": 2.7902322853130758e-06, "loss": 0.002704191952943802, "num_tokens": 2196618.0, "reward": 5.1059675216674805, "reward_std": 0.665777325630188, "rewards/coordinate_accuracy_reward_func/mean": 0.7934672832489014, "rewards/coordinate_accuracy_reward_func/std": 0.012001155875623226, "rewards/graph_topology_reward_func/mean": 4.212499618530273, "rewards/graph_topology_reward_func/std": 0.6790925860404968, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 738.4375, "completions/clipped_ratio": 0.0, "completions/max_length": 1608.0, "completions/max_terminated_length": 1608.0, "completions/mean_length": 738.4375, "completions/mean_terminated_length": 738.4375, "completions/min_length": 449.0, "completions/min_terminated_length": 449.0, "epoch": 0.23684210526315788, "frac_reward_zero_std": 0.0, "grad_norm": 0.048013124614953995, "kl": 0.0, "learning_rate": 2.754087695457005e-06, "loss": 0.005289211869239807, "num_tokens": 2214721.0, "reward": 5.055622100830078, "reward_std": 0.5367113351821899, "rewards/coordinate_accuracy_reward_func/mean": 0.777431070804596, "rewards/coordinate_accuracy_reward_func/std": 0.04514532536268234, "rewards/graph_topology_reward_func/mean": 4.1781907081604, "rewards/graph_topology_reward_func/std": 0.5270512104034424, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 126 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 572.5625, "completions/clipped_ratio": 0.0, "completions/max_length": 977.0, "completions/max_terminated_length": 977.0, "completions/mean_length": 572.5625, "completions/mean_terminated_length": 572.5625, "completions/min_length": 469.0, "completions/min_terminated_length": 469.0, "epoch": 0.2387218045112782, "frac_reward_zero_std": 0.0, "grad_norm": 0.05460721254348755, "kl": 0.0, "learning_rate": 2.717889356869146e-06, "loss": -0.002888401970267296, "num_tokens": 2229642.0, "reward": 4.757171630859375, "reward_std": 1.1693437099456787, "rewards/coordinate_accuracy_reward_func/mean": 0.7259218692779541, "rewards/coordinate_accuracy_reward_func/std": 0.19826240837574005, "rewards/graph_topology_reward_func/mean": 3.950000047683716, "rewards/graph_topology_reward_func/std": 1.1706123352050781, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 127 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 612.375, "completions/clipped_ratio": 0.0, "completions/max_length": 760.0, "completions/max_terminated_length": 760.0, "completions/mean_length": 612.375, "completions/mean_terminated_length": 612.375, "completions/min_length": 561.0, "completions/min_terminated_length": 561.0, "epoch": 0.24060150375939848, "frac_reward_zero_std": 0.0, "grad_norm": 0.025732504203915596, "kl": 0.0, "learning_rate": 2.681644926806527e-06, "loss": -0.001954960636794567, "num_tokens": 2246944.0, "reward": 4.97055196762085, "reward_std": 0.3940858244895935, "rewards/coordinate_accuracy_reward_func/mean": 0.7268022298812866, "rewards/coordinate_accuracy_reward_func/std": 0.10905970633029938, "rewards/graph_topology_reward_func/mean": 4.143750190734863, "rewards/graph_topology_reward_func/std": 0.4194738566875458, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 128 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 463.8125, "completions/clipped_ratio": 0.0, "completions/max_length": 591.0, "completions/max_terminated_length": 591.0, "completions/mean_length": 463.8125, "completions/mean_terminated_length": 463.8125, "completions/min_length": 331.0, "completions/min_terminated_length": 331.0, "epoch": 0.2424812030075188, "frac_reward_zero_std": 0.0, "grad_norm": 0.02850574254989624, "kl": 0.0, "learning_rate": 2.6453620722761897e-06, "loss": -0.00020107068121433258, "num_tokens": 2261181.0, "reward": 4.851731777191162, "reward_std": 0.47068262100219727, "rewards/coordinate_accuracy_reward_func/mean": 0.7829817533493042, "rewards/coordinate_accuracy_reward_func/std": 0.04943789169192314, "rewards/graph_topology_reward_func/mean": 3.96875, "rewards/graph_topology_reward_func/std": 0.7685213088989258, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 129 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 590.1875, "completions/clipped_ratio": 0.0, "completions/max_length": 1175.0, "completions/max_terminated_length": 1175.0, "completions/mean_length": 590.1875, "completions/mean_terminated_length": 590.1875, "completions/min_length": 278.0, "completions/min_terminated_length": 278.0, "epoch": 0.24436090225563908, "frac_reward_zero_std": 0.0, "grad_norm": 0.048575449734926224, "kl": 0.0, "learning_rate": 2.6090484684133406e-06, "loss": -0.005354724824428558, "num_tokens": 2277920.0, "reward": 5.6975932121276855, "reward_std": 0.2721836268901825, "rewards/coordinate_accuracy_reward_func/mean": 0.7453204393386841, "rewards/coordinate_accuracy_reward_func/std": 0.058557894080877304, "rewards/graph_topology_reward_func/mean": 4.852272987365723, "rewards/graph_topology_reward_func/std": 0.36609017848968506, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 130 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 503.125, "completions/clipped_ratio": 0.0, "completions/max_length": 1133.0, "completions/max_terminated_length": 1133.0, "completions/mean_length": 503.125, "completions/mean_terminated_length": 503.125, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "epoch": 0.2462406015037594, "frac_reward_zero_std": 0.5, "grad_norm": 0.03102295473217964, "kl": 0.0, "learning_rate": 2.572711796857779e-06, "loss": -0.009234796278178692, "num_tokens": 2291730.0, "reward": 5.379921913146973, "reward_std": 0.25161734223365784, "rewards/coordinate_accuracy_reward_func/mean": 0.7965884208679199, "rewards/coordinate_accuracy_reward_func/std": 0.01112093310803175, "rewards/graph_topology_reward_func/mean": 4.483333587646484, "rewards/graph_topology_reward_func/std": 0.6359595060348511, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 131 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 718.4375, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1676.0, "completions/mean_length": 718.4375, "completions/mean_terminated_length": 633.0000610351562, "completions/min_length": 302.0, "completions/min_terminated_length": 302.0, "epoch": 0.24812030075187969, "frac_reward_zero_std": 0.0, "grad_norm": 0.25009000301361084, "kl": 0.0, "learning_rate": 2.5363597441289574e-06, "loss": 0.16868962347507477, "num_tokens": 2309801.0, "reward": 3.8706369400024414, "reward_std": 1.3697330951690674, "rewards/coordinate_accuracy_reward_func/mean": 0.7143868207931519, "rewards/coordinate_accuracy_reward_func/std": 0.19591256976127625, "rewards/graph_topology_reward_func/mean": 3.075000047683716, "rewards/graph_topology_reward_func/std": 1.1670476198196411, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 132 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1159.75, "completions/clipped_ratio": 0.3125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1998.0, "completions/mean_length": 1159.75, "completions/mean_terminated_length": 777.8181762695312, "completions/min_length": 455.0, "completions/min_terminated_length": 455.0, "epoch": 0.25, "frac_reward_zero_std": 0.0, "grad_norm": 0.39297711849212646, "kl": 0.0, "learning_rate": 2.5e-06, "loss": 0.14496468007564545, "num_tokens": 2334821.0, "reward": 3.353425979614258, "reward_std": 1.4770066738128662, "rewards/coordinate_accuracy_reward_func/mean": 0.5855279564857483, "rewards/coordinate_accuracy_reward_func/std": 0.3498726785182953, "rewards/graph_topology_reward_func/mean": 2.7428977489471436, "rewards/graph_topology_reward_func/std": 1.691347360610962, "rewards/xml_formatting_reward_func/mean": 0.02500000037252903, "rewards/xml_formatting_reward_func/std": 0.13416409492492676, "step": 133 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 795.0625, "completions/clipped_ratio": 0.1875, "completions/max_length": 2000.0, "completions/max_terminated_length": 898.0, "completions/mean_length": 795.0625, "completions/mean_terminated_length": 517.0, "completions/min_length": 260.0, "completions/min_terminated_length": 260.0, "epoch": 0.2518796992481203, "frac_reward_zero_std": 0.0, "grad_norm": 0.16456486284732819, "kl": 0.0, "learning_rate": 2.4636402558710434e-06, "loss": 0.3731682002544403, "num_tokens": 2354694.0, "reward": 4.173561096191406, "reward_std": 1.4876573085784912, "rewards/coordinate_accuracy_reward_func/mean": 0.6778881549835205, "rewards/coordinate_accuracy_reward_func/std": 0.2735919654369354, "rewards/graph_topology_reward_func/mean": 3.433173179626465, "rewards/graph_topology_reward_func/std": 1.7353311777114868, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 134 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 917.5, "completions/clipped_ratio": 0.1875, "completions/max_length": 2000.0, "completions/max_terminated_length": 1901.0, "completions/mean_length": 917.5, "completions/mean_terminated_length": 667.6923217773438, "completions/min_length": 225.0, "completions/min_terminated_length": 225.0, "epoch": 0.25375939849624063, "frac_reward_zero_std": 0.0, "grad_norm": 0.17802436649799347, "kl": 0.0, "learning_rate": 2.4272882031422216e-06, "loss": 0.15840819478034973, "num_tokens": 2376382.0, "reward": 3.472425937652588, "reward_std": 1.9411141872406006, "rewards/coordinate_accuracy_reward_func/mean": 0.6233872771263123, "rewards/coordinate_accuracy_reward_func/std": 0.3209688365459442, "rewards/graph_topology_reward_func/mean": 2.805288314819336, "rewards/graph_topology_reward_func/std": 1.7756000757217407, "rewards/xml_formatting_reward_func/mean": 0.04374999925494194, "rewards/xml_formatting_reward_func/std": 0.12093386799097061, "step": 135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 592.3125, "completions/clipped_ratio": 0.0, "completions/max_length": 1532.0, "completions/max_terminated_length": 1532.0, "completions/mean_length": 592.3125, "completions/mean_terminated_length": 592.3125, "completions/min_length": 278.0, "completions/min_terminated_length": 278.0, "epoch": 0.2556390977443609, "frac_reward_zero_std": 0.5, "grad_norm": 0.034459155052900314, "kl": 0.0, "learning_rate": 2.3909515315866606e-06, "loss": -0.00017467141151428223, "num_tokens": 2392147.0, "reward": 5.615148544311523, "reward_std": 0.21713413298130035, "rewards/coordinate_accuracy_reward_func/mean": 0.78678297996521, "rewards/coordinate_accuracy_reward_func/std": 0.029292764142155647, "rewards/graph_topology_reward_func/mean": 4.728365421295166, "rewards/graph_topology_reward_func/std": 0.3983369767665863, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 136 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 498.5, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 747.0, "completions/mean_length": 498.5, "completions/mean_terminated_length": 398.4000244140625, "completions/min_length": 281.0, "completions/min_terminated_length": 281.0, "epoch": 0.2575187969924812, "frac_reward_zero_std": 0.0, "grad_norm": 0.18134041130542755, "kl": 0.0, "learning_rate": 2.3546379277238107e-06, "loss": 0.2235431969165802, "num_tokens": 2406699.0, "reward": 5.218072414398193, "reward_std": 1.0251038074493408, "rewards/coordinate_accuracy_reward_func/mean": 0.7305721044540405, "rewards/coordinate_accuracy_reward_func/std": 0.19820043444633484, "rewards/graph_topology_reward_func/mean": 4.40625, "rewards/graph_topology_reward_func/std": 1.3193275928497314, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 137 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 693.9375, "completions/clipped_ratio": 0.0, "completions/max_length": 1074.0, "completions/max_terminated_length": 1074.0, "completions/mean_length": 693.9375, "completions/mean_terminated_length": 693.9375, "completions/min_length": 489.0, "completions/min_terminated_length": 489.0, "epoch": 0.2593984962406015, "frac_reward_zero_std": 0.0, "grad_norm": 0.020451784133911133, "kl": 0.0, "learning_rate": 2.318355073193474e-06, "loss": -0.008083818480372429, "num_tokens": 2423738.0, "reward": 4.938825607299805, "reward_std": 0.30009815096855164, "rewards/coordinate_accuracy_reward_func/mean": 0.7763258814811707, "rewards/coordinate_accuracy_reward_func/std": 0.03791728615760803, "rewards/graph_topology_reward_func/mean": 4.0625, "rewards/graph_topology_reward_func/std": 0.4425306022167206, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 138 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 725.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1713.0, "completions/mean_length": 725.0, "completions/mean_terminated_length": 640.0000610351562, "completions/min_length": 364.0, "completions/min_terminated_length": 364.0, "epoch": 0.26127819548872183, "frac_reward_zero_std": 0.0, "grad_norm": 0.12592196464538574, "kl": 0.0, "learning_rate": 2.2821106431308546e-06, "loss": 0.13406142592430115, "num_tokens": 2441770.0, "reward": 4.872073173522949, "reward_std": 0.8568293452262878, "rewards/coordinate_accuracy_reward_func/mean": 0.7066879868507385, "rewards/coordinate_accuracy_reward_func/std": 0.19240723550319672, "rewards/graph_topology_reward_func/mean": 4.084134578704834, "rewards/graph_topology_reward_func/std": 1.3079349994659424, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 139 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 372.75, "completions/clipped_ratio": 0.0, "completions/max_length": 513.0, "completions/max_terminated_length": 513.0, "completions/mean_length": 372.75, "completions/mean_terminated_length": 372.75, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "epoch": 0.2631578947368421, "frac_reward_zero_std": 0.0, "grad_norm": 0.05156349018216133, "kl": 0.0, "learning_rate": 2.2459123045429953e-06, "loss": -0.0017967447638511658, "num_tokens": 2453318.0, "reward": 4.461216926574707, "reward_std": 0.8703586459159851, "rewards/coordinate_accuracy_reward_func/mean": 0.7987164258956909, "rewards/coordinate_accuracy_reward_func/std": 0.005134448409080505, "rewards/graph_topology_reward_func/mean": 3.5625, "rewards/graph_topology_reward_func/std": 1.209338665008545, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 140 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 770.5, "completions/clipped_ratio": 0.1875, "completions/max_length": 2000.0, "completions/max_terminated_length": 1310.0, "completions/mean_length": 770.5, "completions/mean_terminated_length": 486.7692565917969, "completions/min_length": 294.0, "completions/min_terminated_length": 294.0, "epoch": 0.2650375939849624, "frac_reward_zero_std": 0.5, "grad_norm": 0.30383050441741943, "kl": 0.0, "learning_rate": 2.2097677146869242e-06, "loss": 0.22337713837623596, "num_tokens": 2472350.0, "reward": 4.861433029174805, "reward_std": 1.285253643989563, "rewards/coordinate_accuracy_reward_func/mean": 0.6895580887794495, "rewards/coordinate_accuracy_reward_func/std": 0.27000245451927185, "rewards/graph_topology_reward_func/mean": 4.109375, "rewards/graph_topology_reward_func/std": 1.7004135847091675, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 141 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 342.25, "completions/clipped_ratio": 0.0, "completions/max_length": 441.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 342.25, "completions/mean_terminated_length": 342.25, "completions/min_length": 263.0, "completions/min_terminated_length": 263.0, "epoch": 0.2669172932330827, "frac_reward_zero_std": 0.5, "grad_norm": 0.05587203428149223, "kl": 0.0, "learning_rate": 2.173684519449872e-06, "loss": 0.0018749982118606567, "num_tokens": 2483410.0, "reward": 5.150000095367432, "reward_std": 0.6943650841712952, "rewards/coordinate_accuracy_reward_func/mean": 0.800000011920929, "rewards/coordinate_accuracy_reward_func/std": 0.0, "rewards/graph_topology_reward_func/mean": 4.25, "rewards/graph_topology_reward_func/std": 1.2247449159622192, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 142 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 938.25, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1650.0, "completions/mean_length": 938.25, "completions/mean_terminated_length": 867.4667358398438, "completions/min_length": 604.0, "completions/min_terminated_length": 604.0, "epoch": 0.26879699248120303, "frac_reward_zero_std": 0.0, "grad_norm": 0.18139538168907166, "kl": 0.0, "learning_rate": 2.1376703517318835e-06, "loss": 0.19561165571212769, "num_tokens": 2504886.0, "reward": 5.233395576477051, "reward_std": 1.1588438749313354, "rewards/coordinate_accuracy_reward_func/mean": 0.7124605774879456, "rewards/coordinate_accuracy_reward_func/std": 0.20150022208690643, "rewards/graph_topology_reward_func/mean": 4.439685344696045, "rewards/graph_topology_reward_func/std": 1.2255727052688599, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 143 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 337.5625, "completions/clipped_ratio": 0.0, "completions/max_length": 499.0, "completions/max_terminated_length": 499.0, "completions/mean_length": 337.5625, "completions/mean_terminated_length": 337.5625, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "epoch": 0.2706766917293233, "frac_reward_zero_std": 0.5, "grad_norm": 0.03617791458964348, "kl": 0.0, "learning_rate": 2.101732829831194e-06, "loss": 0.0011982591822743416, "num_tokens": 2515871.0, "reward": 5.393293380737305, "reward_std": 0.5274629592895508, "rewards/coordinate_accuracy_reward_func/mean": 0.7932931184768677, "rewards/coordinate_accuracy_reward_func/std": 0.019011234864592552, "rewards/graph_topology_reward_func/mean": 4.5, "rewards/graph_topology_reward_func/std": 0.8944272398948669, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 576.375, "completions/clipped_ratio": 0.0, "completions/max_length": 1196.0, "completions/max_terminated_length": 1196.0, "completions/mean_length": 576.375, "completions/mean_terminated_length": 576.375, "completions/min_length": 380.0, "completions/min_terminated_length": 380.0, "epoch": 0.2725563909774436, "frac_reward_zero_std": 0.0, "grad_norm": 0.11900322884321213, "kl": 0.0, "learning_rate": 2.0658795558326745e-06, "loss": -0.022828510031104088, "num_tokens": 2531381.0, "reward": 4.640824317932129, "reward_std": 1.6513330936431885, "rewards/coordinate_accuracy_reward_func/mean": 0.7220743298530579, "rewards/coordinate_accuracy_reward_func/std": 0.2011108547449112, "rewards/graph_topology_reward_func/mean": 3.8375000953674316, "rewards/graph_topology_reward_func/std": 1.5217862129211426, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 559.1875, "completions/clipped_ratio": 0.0, "completions/max_length": 898.0, "completions/max_terminated_length": 898.0, "completions/mean_length": 559.1875, "completions/mean_terminated_length": 559.1875, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "epoch": 0.2744360902255639, "frac_reward_zero_std": 0.0, "grad_norm": 0.04900432005524635, "kl": 0.0, "learning_rate": 2.0301181139997206e-06, "loss": 0.004935018718242645, "num_tokens": 2547336.0, "reward": 4.00078821182251, "reward_std": 0.6956571340560913, "rewards/coordinate_accuracy_reward_func/mean": 0.7396768927574158, "rewards/coordinate_accuracy_reward_func/std": 0.20122812688350677, "rewards/graph_topology_reward_func/mean": 3.179861068725586, "rewards/graph_topology_reward_func/std": 1.5674372911453247, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 146 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 724.3125, "completions/clipped_ratio": 0.0, "completions/max_length": 1442.0, "completions/max_terminated_length": 1442.0, "completions/mean_length": 724.3125, "completions/mean_terminated_length": 724.3125, "completions/min_length": 552.0, "completions/min_terminated_length": 552.0, "epoch": 0.27631578947368424, "frac_reward_zero_std": 0.0, "grad_norm": 0.04552697017788887, "kl": 0.0, "learning_rate": 1.994456069169906e-06, "loss": -0.01499214582145214, "num_tokens": 2566781.0, "reward": 5.46744966506958, "reward_std": 0.2689211964607239, "rewards/coordinate_accuracy_reward_func/mean": 0.7174495458602905, "rewards/coordinate_accuracy_reward_func/std": 0.08860702067613602, "rewards/graph_topology_reward_func/mean": 4.650000095367432, "rewards/graph_topology_reward_func/std": 0.4000000059604645, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 147 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 403.6875, "completions/clipped_ratio": 0.0, "completions/max_length": 643.0, "completions/max_terminated_length": 643.0, "completions/mean_length": 403.6875, "completions/mean_terminated_length": 403.6875, "completions/min_length": 247.0, "completions/min_terminated_length": 247.0, "epoch": 0.2781954887218045, "frac_reward_zero_std": 0.0, "grad_norm": 0.1353675127029419, "kl": 0.0, "learning_rate": 1.958900965154743e-06, "loss": -0.010168902575969696, "num_tokens": 2578824.0, "reward": 5.47345495223999, "reward_std": 1.2021632194519043, "rewards/coordinate_accuracy_reward_func/mean": 0.748454749584198, "rewards/coordinate_accuracy_reward_func/std": 0.1996263712644577, "rewards/graph_topology_reward_func/mean": 4.625, "rewards/graph_topology_reward_func/std": 1.2583057880401611, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 148 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 713.25, "completions/clipped_ratio": 0.0, "completions/max_length": 1039.0, "completions/max_terminated_length": 1039.0, "completions/mean_length": 713.25, "completions/mean_terminated_length": 713.25, "completions/min_length": 629.0, "completions/min_terminated_length": 629.0, "epoch": 0.2800751879699248, "frac_reward_zero_std": 0.0, "grad_norm": 0.05676470324397087, "kl": 0.0, "learning_rate": 1.9234603231439e-06, "loss": 0.00953211635351181, "num_tokens": 2598124.0, "reward": 4.7954206466674805, "reward_std": 0.5885616540908813, "rewards/coordinate_accuracy_reward_func/mean": 0.7579202651977539, "rewards/coordinate_accuracy_reward_func/std": 0.044387850910425186, "rewards/graph_topology_reward_func/mean": 3.9375, "rewards/graph_topology_reward_func/std": 0.7274384498596191, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 735.625, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1017.0, "completions/mean_length": 735.625, "completions/mean_terminated_length": 651.3333740234375, "completions/min_length": 546.0, "completions/min_terminated_length": 546.0, "epoch": 0.2819548872180451, "frac_reward_zero_std": 0.0, "grad_norm": 0.16689381003379822, "kl": 0.0, "learning_rate": 1.8881416401141905e-06, "loss": 0.20352552831172943, "num_tokens": 2616358.0, "reward": 5.142414093017578, "reward_std": 1.3101907968521118, "rewards/coordinate_accuracy_reward_func/mean": 0.6980957984924316, "rewards/coordinate_accuracy_reward_func/std": 0.19453704357147217, "rewards/graph_topology_reward_func/mean": 4.363068103790283, "rewards/graph_topology_reward_func/std": 1.2410449981689453, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 150 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 594.6875, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1326.0, "completions/mean_length": 594.6875, "completions/mean_terminated_length": 501.0000305175781, "completions/min_length": 249.0, "completions/min_terminated_length": 249.0, "epoch": 0.28383458646616544, "frac_reward_zero_std": 0.5, "grad_norm": 0.0931449756026268, "kl": 0.0, "learning_rate": 1.852952387243698e-06, "loss": 0.17352606356143951, "num_tokens": 2631457.0, "reward": 5.106843948364258, "reward_std": 0.9219955801963806, "rewards/coordinate_accuracy_reward_func/mean": 0.713093638420105, "rewards/coordinate_accuracy_reward_func/std": 0.2010815590620041, "rewards/graph_topology_reward_func/mean": 4.3125, "rewards/graph_topology_reward_func/std": 1.2435835599899292, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 151 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 304.3125, "completions/clipped_ratio": 0.0, "completions/max_length": 635.0, "completions/max_terminated_length": 635.0, "completions/mean_length": 304.3125, "completions/mean_terminated_length": 304.3125, "completions/min_length": 215.0, "completions/min_terminated_length": 215.0, "epoch": 0.2857142857142857, "frac_reward_zero_std": 0.0, "grad_norm": 0.08837702870368958, "kl": 0.0, "learning_rate": 1.8179000083313483e-06, "loss": 0.0, "num_tokens": 2641910.0, "reward": 4.775000095367432, "reward_std": 1.4961488246917725, "rewards/coordinate_accuracy_reward_func/mean": 0.800000011920929, "rewards/coordinate_accuracy_reward_func/std": 0.0, "rewards/graph_topology_reward_func/mean": 3.875, "rewards/graph_topology_reward_func/std": 1.5, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 152 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 408.6875, "completions/clipped_ratio": 0.0, "completions/max_length": 537.0, "completions/max_terminated_length": 537.0, "completions/mean_length": 408.6875, "completions/mean_terminated_length": 408.6875, "completions/min_length": 288.0, "completions/min_terminated_length": 288.0, "epoch": 0.287593984962406, "frac_reward_zero_std": 0.5, "grad_norm": 0.07994116842746735, "kl": 0.0, "learning_rate": 1.7829919182222752e-06, "loss": -0.003338746726512909, "num_tokens": 2654033.0, "reward": 5.096208572387695, "reward_std": 0.9097648859024048, "rewards/coordinate_accuracy_reward_func/mean": 0.7441254258155823, "rewards/coordinate_accuracy_reward_func/std": 0.19950078427791595, "rewards/graph_topology_reward_func/mean": 4.270833492279053, "rewards/graph_topology_reward_func/std": 1.236594796180725, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 153 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 611.875, "completions/clipped_ratio": 0.0, "completions/max_length": 911.0, "completions/max_terminated_length": 911.0, "completions/mean_length": 611.875, "completions/mean_terminated_length": 611.875, "completions/min_length": 482.0, "completions/min_terminated_length": 482.0, "epoch": 0.2894736842105263, "frac_reward_zero_std": 0.0, "grad_norm": 0.017979193478822708, "kl": 0.0, "learning_rate": 1.7482355012393177e-06, "loss": 0.001868298277258873, "num_tokens": 2672543.0, "reward": 5.179852485656738, "reward_std": 0.2804794907569885, "rewards/coordinate_accuracy_reward_func/mean": 0.6798523664474487, "rewards/coordinate_accuracy_reward_func/std": 0.12331018596887589, "rewards/graph_topology_reward_func/mean": 4.400000095367432, "rewards/graph_topology_reward_func/std": 0.6928203105926514, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 574.25, "completions/clipped_ratio": 0.0, "completions/max_length": 737.0, "completions/max_terminated_length": 737.0, "completions/mean_length": 574.25, "completions/mean_terminated_length": 574.25, "completions/min_length": 497.0, "completions/min_terminated_length": 497.0, "epoch": 0.29135338345864664, "frac_reward_zero_std": 0.0, "grad_norm": 0.030934562906622887, "kl": 0.0, "learning_rate": 1.7136381096209665e-06, "loss": -0.003694351762533188, "num_tokens": 2688195.0, "reward": 5.021483898162842, "reward_std": 0.3518449664115906, "rewards/coordinate_accuracy_reward_func/mean": 0.7881506681442261, "rewards/coordinate_accuracy_reward_func/std": 0.03399653360247612, "rewards/graph_topology_reward_func/mean": 4.133333206176758, "rewards/graph_topology_reward_func/std": 0.679433286190033, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 308.3125, "completions/clipped_ratio": 0.0, "completions/max_length": 522.0, "completions/max_terminated_length": 522.0, "completions/mean_length": 308.3125, "completions/mean_terminated_length": 308.3125, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "epoch": 0.2932330827067669, "frac_reward_zero_std": 0.5, "grad_norm": 0.020805474370718002, "kl": 0.0, "learning_rate": 1.6792070619660977e-06, "loss": -0.001031249761581421, "num_tokens": 2698712.0, "reward": 4.212500095367432, "reward_std": 0.5303300619125366, "rewards/coordinate_accuracy_reward_func/mean": 0.800000011920929, "rewards/coordinate_accuracy_reward_func/std": 0.0, "rewards/graph_topology_reward_func/mean": 3.3125, "rewards/graph_topology_reward_func/std": 1.5370426177978516, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 156 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 345.9375, "completions/clipped_ratio": 0.0, "completions/max_length": 399.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 345.9375, "completions/mean_terminated_length": 345.9375, "completions/min_length": 292.0, "completions/min_terminated_length": 292.0, "epoch": 0.2951127819548872, "frac_reward_zero_std": 0.5, "grad_norm": 0.0004282262525521219, "kl": 0.0, "learning_rate": 1.6449496416858285e-06, "loss": -4.612804332282394e-06, "num_tokens": 2709831.0, "reward": 5.89697790145874, "reward_std": 0.005613335873931646, "rewards/coordinate_accuracy_reward_func/mean": 0.7969778776168823, "rewards/coordinate_accuracy_reward_func/std": 0.008280018344521523, "rewards/graph_topology_reward_func/mean": 5.0, "rewards/graph_topology_reward_func/std": 0.0, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 700.625, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1376.0, "completions/mean_length": 700.625, "completions/mean_terminated_length": 614.0000610351562, "completions/min_length": 507.0, "completions/min_terminated_length": 507.0, "epoch": 0.29699248120300753, "frac_reward_zero_std": 0.0, "grad_norm": 0.1973080188035965, "kl": 0.0, "learning_rate": 1.6108730954628093e-06, "loss": 0.22786815464496613, "num_tokens": 2728897.0, "reward": 4.722937107086182, "reward_std": 1.4986801147460938, "rewards/coordinate_accuracy_reward_func/mean": 0.6260621547698975, "rewards/coordinate_accuracy_reward_func/std": 0.1985456943511963, "rewards/graph_topology_reward_func/mean": 4.015625, "rewards/graph_topology_reward_func/std": 1.3153858184814453, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 158 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 513.75, "completions/clipped_ratio": 0.0, "completions/max_length": 1209.0, "completions/max_terminated_length": 1209.0, "completions/mean_length": 513.75, "completions/mean_terminated_length": 513.75, "completions/min_length": 211.0, "completions/min_terminated_length": 211.0, "epoch": 0.29887218045112784, "frac_reward_zero_std": 0.5, "grad_norm": 0.035573236644268036, "kl": 0.0, "learning_rate": 1.5769846317182894e-06, "loss": -0.004293807782232761, "num_tokens": 2743405.0, "reward": 5.74672794342041, "reward_std": 0.15857812762260437, "rewards/coordinate_accuracy_reward_func/mean": 0.7404782772064209, "rewards/coordinate_accuracy_reward_func/std": 0.09929351508617401, "rewards/graph_topology_reward_func/mean": 4.90625, "rewards/graph_topology_reward_func/std": 0.20155644416809082, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 159 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 639.625, "completions/clipped_ratio": 0.0, "completions/max_length": 1169.0, "completions/max_terminated_length": 1169.0, "completions/mean_length": 639.625, "completions/mean_terminated_length": 639.625, "completions/min_length": 331.0, "completions/min_terminated_length": 331.0, "epoch": 0.3007518796992481, "frac_reward_zero_std": 0.0, "grad_norm": 0.03734319657087326, "kl": 0.0, "learning_rate": 1.5432914190872757e-06, "loss": 0.002792090643197298, "num_tokens": 2759927.0, "reward": 4.936060905456543, "reward_std": 0.28994032740592957, "rewards/coordinate_accuracy_reward_func/mean": 0.7769259810447693, "rewards/coordinate_accuracy_reward_func/std": 0.035359788686037064, "rewards/graph_topology_reward_func/mean": 4.059134483337402, "rewards/graph_topology_reward_func/std": 0.6831688284873962, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 160 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 476.0625, "completions/clipped_ratio": 0.0, "completions/max_length": 792.0, "completions/max_terminated_length": 792.0, "completions/mean_length": 476.0625, "completions/mean_terminated_length": 476.0625, "completions/min_length": 256.0, "completions/min_terminated_length": 256.0, "epoch": 0.3026315789473684, "frac_reward_zero_std": 0.5, "grad_norm": 0.007356899790465832, "kl": 0.0, "learning_rate": 1.509800584902108e-06, "loss": 9.041372686624527e-05, "num_tokens": 2773656.0, "reward": 5.852182388305664, "reward_std": 0.10402965545654297, "rewards/coordinate_accuracy_reward_func/mean": 0.7896824479103088, "rewards/coordinate_accuracy_reward_func/std": 0.031076546758413315, "rewards/graph_topology_reward_func/mean": 4.962500095367432, "rewards/graph_topology_reward_func/std": 0.1499999761581421, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 161 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 515.5625, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 556.0, "completions/mean_length": 515.5625, "completions/mean_terminated_length": 416.60003662109375, "completions/min_length": 263.0, "completions/min_terminated_length": 263.0, "epoch": 0.30451127819548873, "frac_reward_zero_std": 0.0, "grad_norm": 0.1177961602807045, "kl": 0.0, "learning_rate": 1.4765192136847686e-06, "loss": 0.20104043185710907, "num_tokens": 2788193.0, "reward": 4.65786075592041, "reward_std": 1.4174928665161133, "rewards/coordinate_accuracy_reward_func/mean": 0.7328604459762573, "rewards/coordinate_accuracy_reward_func/std": 0.20594097673892975, "rewards/graph_topology_reward_func/mean": 3.84375, "rewards/graph_topology_reward_func/std": 1.4545189142227173, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 162 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 589.625, "completions/clipped_ratio": 0.0, "completions/max_length": 944.0, "completions/max_terminated_length": 944.0, "completions/mean_length": 589.625, "completions/mean_terminated_length": 589.625, "completions/min_length": 362.0, "completions/min_terminated_length": 362.0, "epoch": 0.30639097744360905, "frac_reward_zero_std": 0.0, "grad_norm": 0.02669627033174038, "kl": 0.0, "learning_rate": 1.443454345648252e-06, "loss": 0.001730161253362894, "num_tokens": 2803739.0, "reward": 5.011600494384766, "reward_std": 0.47284775972366333, "rewards/coordinate_accuracy_reward_func/mean": 0.7697731852531433, "rewards/coordinate_accuracy_reward_func/std": 0.039067476987838745, "rewards/graph_topology_reward_func/mean": 4.141826629638672, "rewards/graph_topology_reward_func/std": 0.659286379814148, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 163 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 575.375, "completions/clipped_ratio": 0.0, "completions/max_length": 1041.0, "completions/max_terminated_length": 1041.0, "completions/mean_length": 575.375, "completions/mean_terminated_length": 575.375, "completions/min_length": 373.0, "completions/min_terminated_length": 373.0, "epoch": 0.3082706766917293, "frac_reward_zero_std": 0.0, "grad_norm": 0.06398235261440277, "kl": 0.0, "learning_rate": 1.4106129752073023e-06, "loss": 0.009084941819310188, "num_tokens": 2820657.0, "reward": 4.633889198303223, "reward_std": 0.530742883682251, "rewards/coordinate_accuracy_reward_func/mean": 0.7630555629730225, "rewards/coordinate_accuracy_reward_func/std": 0.05858412757515907, "rewards/graph_topology_reward_func/mean": 3.7708332538604736, "rewards/graph_topology_reward_func/std": 0.5639641880989075, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 1011.6875, "completions/clipped_ratio": 0.1875, "completions/max_length": 2000.0, "completions/max_terminated_length": 1391.0, "completions/mean_length": 1011.6875, "completions/mean_terminated_length": 783.6154174804688, "completions/min_length": 523.0, "completions/min_terminated_length": 523.0, "epoch": 0.3101503759398496, "frac_reward_zero_std": 0.0, "grad_norm": 0.12504038214683533, "kl": 0.0, "learning_rate": 1.3780020494988447e-06, "loss": 0.3048320412635803, "num_tokens": 2844124.0, "reward": 4.46811056137085, "reward_std": 1.9337897300720215, "rewards/coordinate_accuracy_reward_func/mean": 0.627485454082489, "rewards/coordinate_accuracy_reward_func/std": 0.2557498812675476, "rewards/graph_topology_reward_func/mean": 3.778125047683716, "rewards/graph_topology_reward_func/std": 1.5359002351760864, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 636.1875, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 856.0, "completions/mean_length": 636.1875, "completions/mean_terminated_length": 545.2667236328125, "completions/min_length": 496.0, "completions/min_terminated_length": 496.0, "epoch": 0.31203007518796994, "frac_reward_zero_std": 0.0, "grad_norm": 0.035180557519197464, "kl": 0.0, "learning_rate": 1.3456284669124159e-06, "loss": -0.03274478763341904, "num_tokens": 2862591.0, "reward": 4.764005661010742, "reward_std": 0.6239866018295288, "rewards/coordinate_accuracy_reward_func/mean": 0.7629634737968445, "rewards/coordinate_accuracy_reward_func/std": 0.040863048285245895, "rewards/graph_topology_reward_func/mean": 3.9010415077209473, "rewards/graph_topology_reward_func/std": 0.6377178430557251, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 166 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 455.8125, "completions/clipped_ratio": 0.0, "completions/max_length": 779.0, "completions/max_terminated_length": 779.0, "completions/mean_length": 455.8125, "completions/mean_terminated_length": 455.8125, "completions/min_length": 317.0, "completions/min_terminated_length": 317.0, "epoch": 0.31390977443609025, "frac_reward_zero_std": 0.0, "grad_norm": 0.1166972890496254, "kl": 0.0, "learning_rate": 1.313499075630899e-06, "loss": -0.0006230361759662628, "num_tokens": 2877036.0, "reward": 5.15005350112915, "reward_std": 0.7186349630355835, "rewards/coordinate_accuracy_reward_func/mean": 0.7635949850082397, "rewards/coordinate_accuracy_reward_func/std": 0.04973550885915756, "rewards/graph_topology_reward_func/mean": 4.2864580154418945, "rewards/graph_topology_reward_func/std": 0.7103525400161743, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 167 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 720.9375, "completions/clipped_ratio": 0.125, "completions/max_length": 2000.0, "completions/max_terminated_length": 1149.0, "completions/mean_length": 720.9375, "completions/mean_terminated_length": 538.2142944335938, "completions/min_length": 429.0, "completions/min_terminated_length": 429.0, "epoch": 0.3157894736842105, "frac_reward_zero_std": 0.0, "grad_norm": 0.06108391657471657, "kl": 0.0, "learning_rate": 1.2816206721818944e-06, "loss": 0.18485940992832184, "num_tokens": 2895723.0, "reward": 5.2382378578186035, "reward_std": 1.1937769651412964, "rewards/coordinate_accuracy_reward_func/mean": 0.6257379651069641, "rewards/coordinate_accuracy_reward_func/std": 0.19281578063964844, "rewards/graph_topology_reward_func/mean": 4.53125, "rewards/graph_topology_reward_func/std": 1.2545750141143799, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 168 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 353.6875, "completions/clipped_ratio": 0.0, "completions/max_length": 420.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 353.6875, "completions/mean_terminated_length": 353.6875, "completions/min_length": 294.0, "completions/min_terminated_length": 294.0, "epoch": 0.3176691729323308, "frac_reward_zero_std": 0.0, "grad_norm": 0.043140385299921036, "kl": 0.0, "learning_rate": 1.2500000000000007e-06, "loss": -0.0014187810011208057, "num_tokens": 2908486.0, "reward": 4.935418605804443, "reward_std": 0.31448137760162354, "rewards/coordinate_accuracy_reward_func/mean": 0.716668963432312, "rewards/coordinate_accuracy_reward_func/std": 0.10445355623960495, "rewards/graph_topology_reward_func/mean": 4.118750095367432, "rewards/graph_topology_reward_func/std": 0.7386643290519714, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 169 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 446.5625, "completions/clipped_ratio": 0.0, "completions/max_length": 806.0, "completions/max_terminated_length": 806.0, "completions/mean_length": 446.5625, "completions/mean_terminated_length": 446.5625, "completions/min_length": 216.0, "completions/min_terminated_length": 216.0, "epoch": 0.31954887218045114, "frac_reward_zero_std": 0.5, "grad_norm": 0.013107211329042912, "kl": 0.0, "learning_rate": 1.218643748000337e-06, "loss": -0.0032877016346901655, "num_tokens": 2922783.0, "reward": 5.269986629486084, "reward_std": 0.17157143354415894, "rewards/coordinate_accuracy_reward_func/mean": 0.7949864864349365, "rewards/coordinate_accuracy_reward_func/std": 0.014189116656780243, "rewards/graph_topology_reward_func/mean": 4.375, "rewards/graph_topology_reward_func/std": 0.6892024278640747, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 170 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 579.8125, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 782.0, "completions/mean_length": 579.8125, "completions/mean_terminated_length": 485.13336181640625, "completions/min_length": 283.0, "completions/min_terminated_length": 283.0, "epoch": 0.32142857142857145, "frac_reward_zero_std": 0.0, "grad_norm": 0.013990349136292934, "kl": 0.0, "learning_rate": 1.1875585491636e-06, "loss": 0.006113918498158455, "num_tokens": 2938348.0, "reward": 5.206358432769775, "reward_std": 0.2964138090610504, "rewards/coordinate_accuracy_reward_func/mean": 0.7910176515579224, "rewards/coordinate_accuracy_reward_func/std": 0.020533369854092598, "rewards/graph_topology_reward_func/mean": 4.315340995788574, "rewards/graph_topology_reward_func/std": 0.8072448968887329, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 171 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 733.1875, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 1630.0, "completions/mean_length": 733.1875, "completions/mean_terminated_length": 648.7333374023438, "completions/min_length": 372.0, "completions/min_terminated_length": 372.0, "epoch": 0.3233082706766917, "frac_reward_zero_std": 0.0, "grad_norm": 0.18502187728881836, "kl": 0.0, "learning_rate": 1.1567509791329402e-06, "loss": 0.14066901803016663, "num_tokens": 2956655.0, "reward": 4.273937225341797, "reward_std": 1.2004225254058838, "rewards/coordinate_accuracy_reward_func/mean": 0.6569726467132568, "rewards/coordinate_accuracy_reward_func/std": 0.21254704892635345, "rewards/graph_topology_reward_func/mean": 3.5357141494750977, "rewards/graph_topology_reward_func/std": 1.1547741889953613, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 172 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 368.875, "completions/clipped_ratio": 0.0, "completions/max_length": 630.0, "completions/max_terminated_length": 630.0, "completions/mean_length": 368.875, "completions/mean_terminated_length": 368.875, "completions/min_length": 251.0, "completions/min_terminated_length": 251.0, "epoch": 0.325187969924812, "frac_reward_zero_std": 0.0, "grad_norm": 0.05937810614705086, "kl": 0.0, "learning_rate": 1.1262275548229852e-06, "loss": -0.00021551363170146942, "num_tokens": 2969133.0, "reward": 5.303394317626953, "reward_std": 0.806463360786438, "rewards/coordinate_accuracy_reward_func/mean": 0.7658941745758057, "rewards/coordinate_accuracy_reward_func/std": 0.04670072719454765, "rewards/graph_topology_reward_func/mean": 4.4375, "rewards/graph_topology_reward_func/std": 1.209338665008545, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 173 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 484.8125, "completions/clipped_ratio": 0.0, "completions/max_length": 670.0, "completions/max_terminated_length": 670.0, "completions/mean_length": 484.8125, "completions/mean_terminated_length": 484.8125, "completions/min_length": 353.0, "completions/min_terminated_length": 353.0, "epoch": 0.32706766917293234, "frac_reward_zero_std": 0.0, "grad_norm": 0.05529944226145744, "kl": 0.0, "learning_rate": 1.0959947330412681e-06, "loss": 0.0003865892067551613, "num_tokens": 2983586.0, "reward": 5.045216083526611, "reward_std": 0.7420950531959534, "rewards/coordinate_accuracy_reward_func/mean": 0.7889660596847534, "rewards/coordinate_accuracy_reward_func/std": 0.021292582154273987, "rewards/graph_topology_reward_func/mean": 4.15625, "rewards/graph_topology_reward_func/std": 1.0562314987182617, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 174 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 601.4375, "completions/clipped_ratio": 0.0, "completions/max_length": 1112.0, "completions/max_terminated_length": 1112.0, "completions/mean_length": 601.4375, "completions/mean_terminated_length": 601.4375, "completions/min_length": 411.0, "completions/min_terminated_length": 411.0, "epoch": 0.32894736842105265, "frac_reward_zero_std": 0.0, "grad_norm": 0.0171047393232584, "kl": 0.0, "learning_rate": 1.0660589091223854e-06, "loss": 0.0018669969867914915, "num_tokens": 2999497.0, "reward": 5.113887310028076, "reward_std": 0.16257959604263306, "rewards/coordinate_accuracy_reward_func/mean": 0.7881444096565247, "rewards/coordinate_accuracy_reward_func/std": 0.013344109058380127, "rewards/graph_topology_reward_func/mean": 4.225743293762207, "rewards/graph_topology_reward_func/std": 0.8275272250175476, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 591.75, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 751.0, "completions/mean_length": 591.75, "completions/mean_terminated_length": 497.86669921875, "completions/min_length": 329.0, "completions/min_terminated_length": 329.0, "epoch": 0.3308270676691729, "frac_reward_zero_std": 0.0, "grad_norm": 0.04783705994486809, "kl": 0.0, "learning_rate": 1.0364264155751489e-06, "loss": 0.20845438539981842, "num_tokens": 3015925.0, "reward": 5.2569990158081055, "reward_std": 1.0345845222473145, "rewards/coordinate_accuracy_reward_func/mean": 0.7132493257522583, "rewards/coordinate_accuracy_reward_func/std": 0.19983521103858948, "rewards/graph_topology_reward_func/mean": 4.462500095367432, "rewards/graph_topology_reward_func/std": 1.2643179893493652, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 548.4375, "completions/clipped_ratio": 0.0, "completions/max_length": 842.0, "completions/max_terminated_length": 842.0, "completions/mean_length": 548.4375, "completions/mean_terminated_length": 548.4375, "completions/min_length": 360.0, "completions/min_terminated_length": 360.0, "epoch": 0.33270676691729323, "frac_reward_zero_std": 0.0, "grad_norm": 0.028469592332839966, "kl": 0.0, "learning_rate": 1.0071035207430352e-06, "loss": -0.001276304479688406, "num_tokens": 3032700.0, "reward": 5.705867767333984, "reward_std": 0.24063822627067566, "rewards/coordinate_accuracy_reward_func/mean": 0.7791632413864136, "rewards/coordinate_accuracy_reward_func/std": 0.034453220665454865, "rewards/graph_topology_reward_func/mean": 4.826704502105713, "rewards/graph_topology_reward_func/std": 0.3506070077419281, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 177 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 387.9375, "completions/clipped_ratio": 0.0, "completions/max_length": 720.0, "completions/max_terminated_length": 720.0, "completions/mean_length": 387.9375, "completions/mean_terminated_length": 387.9375, "completions/min_length": 247.0, "completions/min_terminated_length": 247.0, "epoch": 0.33458646616541354, "frac_reward_zero_std": 0.0, "grad_norm": 0.08210897445678711, "kl": 0.0, "learning_rate": 9.780964274781984e-07, "loss": -0.02495812252163887, "num_tokens": 3044491.0, "reward": 4.684155464172363, "reward_std": 1.458109974861145, "rewards/coordinate_accuracy_reward_func/mean": 0.7279053926467896, "rewards/coordinate_accuracy_reward_func/std": 0.20056889951229095, "rewards/graph_topology_reward_func/mean": 3.875, "rewards/graph_topology_reward_func/std": 1.408308744430542, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 178 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 671.1875, "completions/clipped_ratio": 0.0, "completions/max_length": 1049.0, "completions/max_terminated_length": 1049.0, "completions/mean_length": 671.1875, "completions/mean_terminated_length": 671.1875, "completions/min_length": 530.0, "completions/min_terminated_length": 530.0, "epoch": 0.33646616541353386, "frac_reward_zero_std": 0.0, "grad_norm": 0.13130497932434082, "kl": 0.0, "learning_rate": 9.494112718293503e-07, "loss": 0.05091579258441925, "num_tokens": 3061518.0, "reward": 4.737955093383789, "reward_std": 1.1741849184036255, "rewards/coordinate_accuracy_reward_func/mean": 0.7078415751457214, "rewards/coordinate_accuracy_reward_func/std": 0.19945164024829865, "rewards/graph_topology_reward_func/mean": 3.9488635063171387, "rewards/graph_topology_reward_func/std": 1.1326454877853394, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 179 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 579.0, "completions/clipped_ratio": 0.125, "completions/max_length": 2000.0, "completions/max_terminated_length": 580.0, "completions/mean_length": 579.0, "completions/mean_terminated_length": 376.0000305175781, "completions/min_length": 273.0, "completions/min_terminated_length": 273.0, "epoch": 0.3383458646616541, "frac_reward_zero_std": 0.0, "grad_norm": 0.22466161847114563, "kl": 0.0, "learning_rate": 9.210541217437566e-07, "loss": 0.3866468667984009, "num_tokens": 3076366.0, "reward": 4.722480773925781, "reward_std": 1.4534472227096558, "rewards/coordinate_accuracy_reward_func/mean": 0.6974806189537048, "rewards/coordinate_accuracy_reward_func/std": 0.2723557949066162, "rewards/graph_topology_reward_func/mean": 3.9625000953674316, "rewards/graph_topology_reward_func/std": 1.8463928699493408, "rewards/xml_formatting_reward_func/mean": 0.0625, "rewards/xml_formatting_reward_func/std": 0.10246950387954712, "step": 180 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 432.5625, "completions/clipped_ratio": 0.0, "completions/max_length": 543.0, "completions/max_terminated_length": 543.0, "completions/mean_length": 432.5625, "completions/mean_terminated_length": 432.5625, "completions/min_length": 355.0, "completions/min_terminated_length": 355.0, "epoch": 0.34022556390977443, "frac_reward_zero_std": 0.0, "grad_norm": 0.03955727815628052, "kl": 0.0, "learning_rate": 8.930309757836517e-07, "loss": -4.878919571638107e-05, "num_tokens": 3090007.0, "reward": 5.128998756408691, "reward_std": 0.5744249224662781, "rewards/coordinate_accuracy_reward_func/mean": 0.7477483749389648, "rewards/coordinate_accuracy_reward_func/std": 0.08042123913764954, "rewards/graph_topology_reward_func/mean": 4.28125, "rewards/graph_topology_reward_func/std": 0.6046693325042725, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 181 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 288.25, "completions/clipped_ratio": 0.0, "completions/max_length": 434.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 288.25, "completions/mean_terminated_length": 288.25, "completions/min_length": 204.0, "completions/min_terminated_length": 204.0, "epoch": 0.34210526315789475, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "kl": 0.0, "learning_rate": 8.653477618573261e-07, "loss": 0.0, "num_tokens": 3100203.0, "reward": 5.900000095367432, "reward_std": 0.0, "rewards/coordinate_accuracy_reward_func/mean": 0.800000011920929, "rewards/coordinate_accuracy_reward_func/std": 0.0, "rewards/graph_topology_reward_func/mean": 5.0, "rewards/graph_topology_reward_func/std": 0.0, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 182 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 598.25, "completions/clipped_ratio": 0.0, "completions/max_length": 1102.0, "completions/max_terminated_length": 1102.0, "completions/mean_length": 598.25, "completions/mean_terminated_length": 598.25, "completions/min_length": 344.0, "completions/min_terminated_length": 344.0, "epoch": 0.34398496240601506, "frac_reward_zero_std": 0.0, "grad_norm": 0.03291003778576851, "kl": 0.0, "learning_rate": 8.380103359651554e-07, "loss": -0.006858679465949535, "num_tokens": 3116207.0, "reward": 5.749535083770752, "reward_std": 0.28908368945121765, "rewards/coordinate_accuracy_reward_func/mean": 0.7745348215103149, "rewards/coordinate_accuracy_reward_func/std": 0.03943701460957527, "rewards/graph_topology_reward_func/mean": 4.875, "rewards/graph_topology_reward_func/std": 0.37148353457450867, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 183 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 380.625, "completions/clipped_ratio": 0.0, "completions/max_length": 574.0, "completions/max_terminated_length": 574.0, "completions/mean_length": 380.625, "completions/mean_terminated_length": 380.625, "completions/min_length": 331.0, "completions/min_terminated_length": 331.0, "epoch": 0.3458646616541353, "frac_reward_zero_std": 0.0, "grad_norm": 0.048465847969055176, "kl": 0.0, "learning_rate": 8.110244809608494e-07, "loss": -0.0010321722365915775, "num_tokens": 3128585.0, "reward": 5.5084309577941895, "reward_std": 0.39158231019973755, "rewards/coordinate_accuracy_reward_func/mean": 0.6896809339523315, "rewards/coordinate_accuracy_reward_func/std": 0.0917106568813324, "rewards/graph_topology_reward_func/mean": 4.71875, "rewards/graph_topology_reward_func/std": 0.6046693325042725, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 184 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 278.1875, "completions/clipped_ratio": 0.0, "completions/max_length": 450.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 278.1875, "completions/mean_terminated_length": 278.1875, "completions/min_length": 202.0, "completions/min_terminated_length": 202.0, "epoch": 0.34774436090225563, "frac_reward_zero_std": 0.0, "grad_norm": 0.0719396322965622, "kl": 0.0, "learning_rate": 7.843959053281663e-07, "loss": 0.002976951189339161, "num_tokens": 3138620.0, "reward": 3.4562501907348633, "reward_std": 1.3497915267944336, "rewards/coordinate_accuracy_reward_func/mean": 0.75, "rewards/coordinate_accuracy_reward_func/std": 0.20000000298023224, "rewards/graph_topology_reward_func/mean": 2.625, "rewards/graph_topology_reward_func/std": 1.5, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 319.0625, "completions/clipped_ratio": 0.0, "completions/max_length": 460.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 319.0625, "completions/mean_terminated_length": 319.0625, "completions/min_length": 252.0, "completions/min_terminated_length": 252.0, "epoch": 0.34962406015037595, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "kl": 0.0, "learning_rate": 7.581302419733633e-07, "loss": 0.0, "num_tokens": 3149309.0, "reward": 5.900000095367432, "reward_std": 0.0, "rewards/coordinate_accuracy_reward_func/mean": 0.800000011920929, "rewards/coordinate_accuracy_reward_func/std": 0.0, "rewards/graph_topology_reward_func/mean": 5.0, "rewards/graph_topology_reward_func/std": 0.0, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 186 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 766.25, "completions/clipped_ratio": 0.0, "completions/max_length": 1128.0, "completions/max_terminated_length": 1128.0, "completions/mean_length": 766.25, "completions/mean_terminated_length": 766.25, "completions/min_length": 597.0, "completions/min_terminated_length": 597.0, "epoch": 0.35150375939849626, "frac_reward_zero_std": 0.0, "grad_norm": 0.026454389095306396, "kl": 0.0, "learning_rate": 7.322330470336314e-07, "loss": 0.009922749362885952, "num_tokens": 3170289.0, "reward": 4.969075679779053, "reward_std": 0.40688395500183105, "rewards/coordinate_accuracy_reward_func/mean": 0.7753257751464844, "rewards/coordinate_accuracy_reward_func/std": 0.023236757144331932, "rewards/graph_topology_reward_func/mean": 4.09375, "rewards/graph_topology_reward_func/std": 0.4552929699420929, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 187 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 572.5, "completions/clipped_ratio": 0.0, "completions/max_length": 1177.0, "completions/max_terminated_length": 1177.0, "completions/mean_length": 572.5, "completions/mean_terminated_length": 572.5, "completions/min_length": 370.0, "completions/min_terminated_length": 370.0, "epoch": 0.3533834586466165, "frac_reward_zero_std": 0.0, "grad_norm": 0.054417192935943604, "kl": 0.0, "learning_rate": 7.067097987017762e-07, "loss": 0.012284297496080399, "num_tokens": 3186601.0, "reward": 4.651228904724121, "reward_std": 0.5644460916519165, "rewards/coordinate_accuracy_reward_func/mean": 0.7137286067008972, "rewards/coordinate_accuracy_reward_func/std": 0.12591156363487244, "rewards/graph_topology_reward_func/mean": 3.8375000953674316, "rewards/graph_topology_reward_func/std": 0.9222255945205688, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 188 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 381.8125, "completions/clipped_ratio": 0.0, "completions/max_length": 549.0, "completions/max_terminated_length": 549.0, "completions/mean_length": 381.8125, "completions/mean_terminated_length": 381.8125, "completions/min_length": 290.0, "completions/min_terminated_length": 290.0, "epoch": 0.35526315789473684, "frac_reward_zero_std": 0.0, "grad_norm": 0.02148415334522724, "kl": 0.0, "learning_rate": 6.815658960673782e-07, "loss": -0.0002451036125421524, "num_tokens": 3198294.0, "reward": 5.8556036949157715, "reward_std": 0.11381106823682785, "rewards/coordinate_accuracy_reward_func/mean": 0.7931035757064819, "rewards/coordinate_accuracy_reward_func/std": 0.011101078242063522, "rewards/graph_topology_reward_func/mean": 4.962500095367432, "rewards/graph_topology_reward_func/std": 0.1499999761581421, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 189 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 899.125, "completions/clipped_ratio": 0.0, "completions/max_length": 1364.0, "completions/max_terminated_length": 1364.0, "completions/mean_length": 899.125, "completions/mean_terminated_length": 899.125, "completions/min_length": 558.0, "completions/min_terminated_length": 558.0, "epoch": 0.35714285714285715, "frac_reward_zero_std": 0.0, "grad_norm": 0.03022579476237297, "kl": 0.0, "learning_rate": 6.568066579746901e-07, "loss": -0.025115644559264183, "num_tokens": 3218968.0, "reward": 4.99937629699707, "reward_std": 0.4107770323753357, "rewards/coordinate_accuracy_reward_func/mean": 0.7721033096313477, "rewards/coordinate_accuracy_reward_func/std": 0.04534292593598366, "rewards/graph_topology_reward_func/mean": 4.127272605895996, "rewards/graph_topology_reward_func/std": 0.5472999215126038, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 190 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 472.9375, "completions/clipped_ratio": 0.0625, "completions/max_length": 2000.0, "completions/max_terminated_length": 680.0, "completions/mean_length": 472.9375, "completions/mean_terminated_length": 371.13336181640625, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "epoch": 0.35902255639097747, "frac_reward_zero_std": 0.0, "grad_norm": 0.04399430751800537, "kl": 0.0, "learning_rate": 6.324373218975105e-07, "loss": -0.03081013448536396, "num_tokens": 3233687.0, "reward": 4.101182460784912, "reward_std": 0.8298507928848267, "rewards/coordinate_accuracy_reward_func/mean": 0.7386823892593384, "rewards/coordinate_accuracy_reward_func/std": 0.1994287520647049, "rewards/graph_topology_reward_func/mean": 3.28125, "rewards/graph_topology_reward_func/std": 1.6928157806396484, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 191 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 384.0, "completions/clipped_ratio": 0.0, "completions/max_length": 741.0, "completions/max_terminated_length": 741.0, "completions/mean_length": 384.0, "completions/mean_terminated_length": 384.0, "completions/min_length": 220.0, "completions/min_terminated_length": 220.0, "epoch": 0.3609022556390977, "frac_reward_zero_std": 0.5, "grad_norm": 0.09448885917663574, "kl": 0.0, "learning_rate": 6.084630428312679e-07, "loss": -0.029936250299215317, "num_tokens": 3245767.0, "reward": 5.339325904846191, "reward_std": 0.9784765839576721, "rewards/coordinate_accuracy_reward_func/mean": 0.7393261194229126, "rewards/coordinate_accuracy_reward_func/std": 0.19929765164852142, "rewards/graph_topology_reward_func/mean": 4.5, "rewards/graph_topology_reward_func/std": 1.26491117477417, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 192 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 476.25, "completions/clipped_ratio": 0.0, "completions/max_length": 636.0, "completions/max_terminated_length": 636.0, "completions/mean_length": 476.25, "completions/mean_terminated_length": 476.25, "completions/min_length": 385.0, "completions/min_terminated_length": 385.0, "epoch": 0.36278195488721804, "frac_reward_zero_std": 0.0, "grad_norm": 0.0175445768982172, "kl": 0.0, "learning_rate": 5.848888922025553e-07, "loss": 0.0013170763850212097, "num_tokens": 3259675.0, "reward": 5.333982944488525, "reward_std": 0.20313739776611328, "rewards/coordinate_accuracy_reward_func/mean": 0.7964825630187988, "rewards/coordinate_accuracy_reward_func/std": 0.006885330658406019, "rewards/graph_topology_reward_func/mean": 4.4375, "rewards/graph_topology_reward_func/std": 0.6422616243362427, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 193 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 670.5625, "completions/clipped_ratio": 0.0, "completions/max_length": 1924.0, "completions/max_terminated_length": 1924.0, "completions/mean_length": 670.5625, "completions/mean_terminated_length": 670.5625, "completions/min_length": 284.0, "completions/min_terminated_length": 284.0, "epoch": 0.36466165413533835, "frac_reward_zero_std": 0.5, "grad_norm": 0.20753991603851318, "kl": 0.0, "learning_rate": 5.617198567963353e-07, "loss": 0.007608555257320404, "num_tokens": 3276516.0, "reward": 4.894202709197998, "reward_std": 0.8837202787399292, "rewards/coordinate_accuracy_reward_func/mean": 0.7192027568817139, "rewards/coordinate_accuracy_reward_func/std": 0.20769073069095612, "rewards/graph_topology_reward_func/mean": 4.09375, "rewards/graph_topology_reward_func/std": 1.3443554639816284, "rewards/xml_formatting_reward_func/mean": 0.08124999701976776, "rewards/xml_formatting_reward_func/std": 0.07500000298023224, "step": 194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 604.875, "completions/clipped_ratio": 0.0, "completions/max_length": 1213.0, "completions/max_terminated_length": 1213.0, "completions/mean_length": 604.875, "completions/mean_terminated_length": 604.875, "completions/min_length": 347.0, "completions/min_terminated_length": 347.0, "epoch": 0.36654135338345867, "frac_reward_zero_std": 0.0, "grad_norm": 0.028032662346959114, "kl": 0.0, "learning_rate": 5.389608377010608e-07, "loss": -0.004686424508690834, "num_tokens": 3293906.0, "reward": 5.749141693115234, "reward_std": 0.30656805634498596, "rewards/coordinate_accuracy_reward_func/mean": 0.7599366903305054, "rewards/coordinate_accuracy_reward_func/std": 0.05416753143072128, "rewards/graph_topology_reward_func/mean": 4.889204502105713, "rewards/graph_topology_reward_func/std": 0.37664929032325745, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 479.1875, "completions/clipped_ratio": 0.0, "completions/max_length": 586.0, "completions/max_terminated_length": 586.0, "completions/mean_length": 479.1875, "completions/mean_terminated_length": 479.1875, "completions/min_length": 350.0, "completions/min_terminated_length": 350.0, "epoch": 0.3684210526315789, "frac_reward_zero_std": 0.0, "grad_norm": 0.0023689414374530315, "kl": 0.0, "learning_rate": 5.166166492719124e-07, "loss": -0.00012168975081294775, "num_tokens": 3308725.0, "reward": 5.854145050048828, "reward_std": 0.036410748958587646, "rewards/coordinate_accuracy_reward_func/mean": 0.754144549369812, "rewards/coordinate_accuracy_reward_func/std": 0.06727547943592072, "rewards/graph_topology_reward_func/mean": 5.0, "rewards/graph_topology_reward_func/std": 0.0, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 196 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 296.0, "completions/clipped_ratio": 0.0, "completions/max_length": 488.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 296.0, "completions/mean_terminated_length": 296.0, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "epoch": 0.37030075187969924, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "kl": 0.0, "learning_rate": 4.946920181123904e-07, "loss": 0.0, "num_tokens": 3319045.0, "reward": 5.900000095367432, "reward_std": 0.0, "rewards/coordinate_accuracy_reward_func/mean": 0.800000011920929, "rewards/coordinate_accuracy_reward_func/std": 0.0, "rewards/graph_topology_reward_func/mean": 5.0, "rewards/graph_topology_reward_func/std": 0.0, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 197 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 589.1875, "completions/clipped_ratio": 0.0, "completions/max_length": 683.0, "completions/max_terminated_length": 683.0, "completions/mean_length": 589.1875, "completions/mean_terminated_length": 589.1875, "completions/min_length": 502.0, "completions/min_terminated_length": 502.0, "epoch": 0.37218045112781956, "frac_reward_zero_std": 0.0, "grad_norm": 0.014526073820888996, "kl": 0.0, "learning_rate": 4.7319158207446953e-07, "loss": 0.0015484420582652092, "num_tokens": 3335752.0, "reward": 5.257194519042969, "reward_std": 0.3002506494522095, "rewards/coordinate_accuracy_reward_func/mean": 0.7821943759918213, "rewards/coordinate_accuracy_reward_func/std": 0.025324687361717224, "rewards/graph_topology_reward_func/mean": 4.375, "rewards/graph_topology_reward_func/std": 0.30276504158973694, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 198 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 579.0625, "completions/clipped_ratio": 0.0, "completions/max_length": 1096.0, "completions/max_terminated_length": 1096.0, "completions/mean_length": 579.0625, "completions/mean_terminated_length": 579.0625, "completions/min_length": 415.0, "completions/min_terminated_length": 415.0, "epoch": 0.37406015037593987, "frac_reward_zero_std": 0.0, "grad_norm": 0.037435997277498245, "kl": 0.0, "learning_rate": 4.5211988927752026e-07, "loss": -0.0044632647186517715, "num_tokens": 3352345.0, "reward": 4.87026309967041, "reward_std": 0.5742481350898743, "rewards/coordinate_accuracy_reward_func/mean": 0.7131201028823853, "rewards/coordinate_accuracy_reward_func/std": 0.09645906835794449, "rewards/graph_topology_reward_func/mean": 4.057143211364746, "rewards/graph_topology_reward_func/std": 0.5720948576927185, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 199 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completion_length": 540.6875, "completions/clipped_ratio": 0.0, "completions/max_length": 1110.0, "completions/max_terminated_length": 1110.0, "completions/mean_length": 540.6875, "completions/mean_terminated_length": 540.6875, "completions/min_length": 340.0, "completions/min_terminated_length": 340.0, "epoch": 0.37593984962406013, "frac_reward_zero_std": 0.0, "grad_norm": 0.04394283518195152, "kl": 0.0, "learning_rate": 4.3148139714622365e-07, "loss": 0.00016671977937221527, "num_tokens": 3367956.0, "reward": 5.367794036865234, "reward_std": 0.6558822989463806, "rewards/coordinate_accuracy_reward_func/mean": 0.7677936553955078, "rewards/coordinate_accuracy_reward_func/std": 0.048047129064798355, "rewards/graph_topology_reward_func/mean": 4.5, "rewards/graph_topology_reward_func/std": 0.632455587387085, "rewards/xml_formatting_reward_func/mean": 0.10000000149011612, "rewards/xml_formatting_reward_func/std": 0.0, "step": 200 } ], "logging_steps": 1, "max_steps": 240, "num_input_tokens_seen": 3367956, "num_train_epochs": 1, "save_steps": 100, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 8, "trial_name": null, "trial_params": null }