{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.5, "eval_steps": 500, "global_step": 10, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4.0, "completions/max_terminated_length": 4.0, "completions/mean_length": 4.0, "completions/mean_terminated_length": 4.0, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "epoch": 0.1, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "kl": 0.0, "learning_rate": 5e-07, "loss": 0.0, "num_tokens": 1056.0, "reward": 2.501505732536316, "reward_std": 0.0, "rewards/concensus_correctness_reward_func/mean": 0.968999981880188, "rewards/concensus_correctness_reward_func/std": 1.1189048290252686, "rewards/consensus_reward_func/mean": 1.5, "rewards/consensus_reward_func/std": 0.5773502588272095, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.03250573016703129, "rewards/question_recreation_reward_func/std": 0.0031549197155982256, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 2 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 18.0, "completions/max_terminated_length": 18.0, "completions/mean_length": 7.5, "completions/mean_terminated_length": 7.5, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "epoch": 0.2, "frac_reward_zero_std": 0.75, "grad_norm": 64.29496765136719, "kl": 0.0, "learning_rate": 4.415111107797445e-07, "loss": 0.1375, "num_tokens": 2140.0, "reward": 2.162332773208618, "reward_std": 0.691501259803772, "rewards/concensus_correctness_reward_func/mean": 0.9030000418424606, "rewards/concensus_correctness_reward_func/std": 0.6905781626701355, "rewards/consensus_reward_func/mean": 1.25, "rewards/consensus_reward_func/std": 1.0773502588272095, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.009332760702818632, "rewards/question_recreation_reward_func/std": 0.00458485318813473, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 4 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 11.5, "completions/max_terminated_length": 11.5, "completions/mean_length": 7.75, "completions/mean_terminated_length": 7.75, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "epoch": 0.3, "frac_reward_zero_std": 1.0, "grad_norm": 0.005112084560096264, "kl": 0.005986911244690418, "learning_rate": 2.934120444167326e-07, "loss": 0.0, "num_tokens": 3226.0, "reward": 2.0115684112533927, "reward_std": 0.0, "rewards/concensus_correctness_reward_func/mean": 1.0, "rewards/concensus_correctness_reward_func/std": 0.0, "rewards/consensus_reward_func/mean": 1.0, "rewards/consensus_reward_func/std": 0.0, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.01156841404736042, "rewards/question_recreation_reward_func/std": 0.006134120747447014, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 6 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 4.0, "completions/max_terminated_length": 4.0, "completions/mean_length": 4.0, "completions/mean_terminated_length": 4.0, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "epoch": 0.4, "frac_reward_zero_std": 1.0, "grad_norm": 0.00172651547472924, "kl": 7.607042789459229e-05, "learning_rate": 1.2500000000000005e-07, "loss": 0.0, "num_tokens": 4282.0, "reward": 3.807758629322052, "reward_std": 0.0, "rewards/concensus_correctness_reward_func/mean": 2.254500061273575, "rewards/concensus_correctness_reward_func/std": 0.3516063094139099, "rewards/consensus_reward_func/mean": 1.5, "rewards/consensus_reward_func/std": 0.5773502588272095, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.05325863230973482, "rewards/question_recreation_reward_func/std": 0.0009418438421562314, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 8 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 11.5, "completions/max_terminated_length": 11.5, "completions/mean_length": 7.75, "completions/mean_terminated_length": 7.75, "completions/min_length": 4.0, "completions/min_terminated_length": 4.0, "epoch": 0.5, "frac_reward_zero_std": 0.75, "grad_norm": 0.00020738673629239202, "kl": 0.005593676585704088, "learning_rate": 1.507684480352292e-08, "loss": 0.0, "num_tokens": 5368.0, "reward": 3.2286359071731567, "reward_std": 2.38063839788083e-05, "rewards/concensus_correctness_reward_func/mean": 2.180999994277954, "rewards/concensus_correctness_reward_func/std": 2.0057148337364197, "rewards/consensus_reward_func/mean": 1.0, "rewards/consensus_reward_func/std": 1.154700517654419, "rewards/cumulative_reward_2/mean": 0.0, "rewards/cumulative_reward_2/std": 0.0, "rewards/final_correctness_reward_func/mean": 0.0, "rewards/final_correctness_reward_func/std": 0.0, "rewards/question_recreation_reward_func/mean": 0.04763588309288025, "rewards/question_recreation_reward_func/std": 0.010593654587864876, "rewards/soft_format_reward_func/mean": 0.0, "rewards/soft_format_reward_func/std": 0.0, "rewards/strict_format_reward_func/mean": 0.0, "rewards/strict_format_reward_func/std": 0.0, "rewards/xmlcount_reward_func/mean": 0.0, "rewards/xmlcount_reward_func/std": 0.0, "step": 10 }, { "epoch": 0.5, "step": 10, "total_flos": 0.0, "train_loss": 0.027499935874280367, "train_runtime": 464.5869, "train_samples_per_second": 0.086, "train_steps_per_second": 0.022 } ], "logging_steps": 2, "max_steps": 10, "num_input_tokens_seen": 5368, "num_train_epochs": 1, "save_steps": 10, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 2, "trial_name": null, "trial_params": null }