{"loss": 0.39252201, "grad_norm": 0.13297546, "learning_rate": 5.725e-05, "memory(GiB)": 149.46, "train_speed(iter/s)": 0.014869, "completion_length": 8022.76757812, "response_clip_ratio": 0.21875, "rewards/CosineReward": 0.1098307, "rewards/RepetitionPenalty": -7.1e-07, "reward": 0.10982999, "reward_std": 0.22277765, "kl": 3.62109375, "clip_ratio": 2.524e-05, "epoch": 5.08333333, "global_step/max_steps": "61/120", "percentage": "50.83%", "elapsed_time": "1h 8m 14s", "remaining_time": "1h 6m 0s"} {"loss": 0.39131755, "grad_norm": 0.0502551, "learning_rate": 5.58e-05, "memory(GiB)": 149.46, "train_speed(iter/s)": 0.0146, "kl": 3.34375, "clip_ratio": 0.00028464, "epoch": 5.16666667, "global_step/max_steps": "62/120", "percentage": "51.67%", "elapsed_time": "1h 10m 38s", "remaining_time": "1h 6m 5s"} {"loss": 0.38100323, "grad_norm": 0.24896182, "learning_rate": 5.436e-05, "memory(GiB)": 149.57, "train_speed(iter/s)": 0.007539, "completion_length": 7439.90039062, "response_clip_ratio": 0.17773438, "rewards/CosineReward": 0.21994188, "rewards/RepetitionPenalty": -2.85e-06, "reward": 0.21993902, "reward_std": 0.23494867, "kl": 2.83984375, "clip_ratio": 2.52e-05, "epoch": 5.25, "global_step/max_steps": "63/120", "percentage": "52.50%", "elapsed_time": "2h 19m 9s", "remaining_time": "2h 5m 53s"} {"loss": 0.37775791, "grad_norm": 0.08133992, "learning_rate": 5.291e-05, "memory(GiB)": 149.57, "train_speed(iter/s)": 0.007528, "kl": 3.1015625, "clip_ratio": 2.427e-05, "epoch": 5.33333333, "global_step/max_steps": "64/120", "percentage": "53.33%", "elapsed_time": "2h 21m 34s", "remaining_time": "2h 3m 52s"} {"loss": 0.37870449, "grad_norm": 0.10720343, "learning_rate": 5.145e-05, "memory(GiB)": 149.57, "train_speed(iter/s)": 0.00516, "completion_length": 7485.83203125, "response_clip_ratio": 0.17382812, "rewards/CosineReward": 0.1747364, "rewards/RepetitionPenalty": -1.32e-06, "reward": 0.17473509, "reward_std": 0.2002282, "kl": 3.4609375, "clip_ratio": 2.008e-05, "epoch": 5.41666667, "global_step/max_steps": "65/120", "percentage": "54.17%", "elapsed_time": "3h 29m 50s", "remaining_time": "2h 57m 33s"} {"loss": 0.37883586, "grad_norm": 0.22383407, "learning_rate": 5e-05, "memory(GiB)": 149.57, "train_speed(iter/s)": 0.005179, "kl": 3.66796875, "clip_ratio": 2.14e-05, "epoch": 5.5, "global_step/max_steps": "66/120", "percentage": "55.00%", "elapsed_time": "3h 32m 17s", "remaining_time": "2h 53m 41s"} {"loss": 0.42400631, "grad_norm": 0.04868019, "learning_rate": 4.855e-05, "memory(GiB)": 149.57, "train_speed(iter/s)": 0.003967, "completion_length": 7483.6328125, "response_clip_ratio": 0.15625, "rewards/CosineReward": 0.18850263, "rewards/RepetitionPenalty": -4.2e-07, "reward": 0.18850221, "reward_std": 0.22211298, "kl": 3.5859375, "clip_ratio": 2.306e-05, "epoch": 5.58333333, "global_step/max_steps": "67/120", "percentage": "55.83%", "elapsed_time": "4h 41m 22s", "remaining_time": "3h 42m 34s"} {"loss": 0.42287919, "grad_norm": 0.0949711, "learning_rate": 4.709e-05, "memory(GiB)": 149.57, "train_speed(iter/s)": 0.003991, "kl": 3.421875, "clip_ratio": 6.947e-05, "epoch": 5.66666667, "global_step/max_steps": "68/120", "percentage": "56.67%", "elapsed_time": "4h 43m 49s", "remaining_time": "3h 37m 2s"} {"loss": 0.41051257, "grad_norm": 0.06014465, "learning_rate": 4.564e-05, "memory(GiB)": 149.57, "train_speed(iter/s)": 0.003261, "completion_length": 7528.89257812, "response_clip_ratio": 0.18554688, "rewards/CosineReward": 0.1917028, "rewards/RepetitionPenalty": -6.3e-07, "reward": 0.19170217, "reward_std": 0.23879585, "kl": 3.3515625, "clip_ratio": 1.951e-05, "epoch": 5.75, "global_step/max_steps": "69/120", "percentage": "57.50%", "elapsed_time": "5h 52m 29s", "remaining_time": "4h 20m 32s"} {"loss": 0.40983129, "grad_norm": 0.04519659, "learning_rate": 4.42e-05, "memory(GiB)": 149.57, "train_speed(iter/s)": 0.003285, "kl": 3.4453125, "clip_ratio": 2.672e-05, "epoch": 5.83333333, "global_step/max_steps": "70/120", "percentage": "58.33%", "elapsed_time": "5h 55m 0s", "remaining_time": "4h 13m 34s"} {"loss": 0.40380001, "grad_norm": 0.05076796, "learning_rate": 4.275e-05, "memory(GiB)": 149.57, "train_speed(iter/s)": 0.00279, "completion_length": 7564.25, "response_clip_ratio": 0.19140625, "rewards/CosineReward": 0.18386444, "rewards/RepetitionPenalty": -2e-08, "reward": 0.18386443, "reward_std": 0.20012712, "kl": 3.40625, "clip_ratio": 2.048e-05, "epoch": 6.08333333, "global_step/max_steps": "71/120", "percentage": "59.17%", "elapsed_time": "7h 3m 57s", "remaining_time": "4h 52m 35s"} {"loss": 0.40342596, "grad_norm": 0.04416068, "learning_rate": 4.132e-05, "memory(GiB)": 149.57, "train_speed(iter/s)": 0.002813, "epoch": 6.16666667, "global_step/max_steps": "72/120", "percentage": "60.00%", "elapsed_time": "7h 6m 24s", "remaining_time": "4h 44m 16s"} {"eval_loss": 0.31386203, "eval_completion_length": 9389.85742188, "eval_response_clip_ratio": 0.42857146, "eval_rewards/CosineReward": 0.14122973, "eval_rewards/RepetitionPenalty": 0.0, "eval_reward": 0.14122972, "eval_reward_std": 0.16340539, "eval_kl": 4.875, "eval_clip_ratio": 4.801e-05, "eval_runtime": 1007.6533, "eval_samples_per_second": 0.007, "eval_steps_per_second": 0.001, "epoch": 6.16666667, "global_step/max_steps": "72/120", "percentage": "60.00%", "elapsed_time": "7h 23m 12s", "remaining_time": "4h 55m 28s"}