brikdavies commited on
Commit
f67fae8
·
verified ·
1 Parent(s): 3005e9a

Upload checkpoint-50/trainer_state.json with huggingface_hub

Browse files
Files changed (1) hide show
  1. checkpoint-50/trainer_state.json +1734 -0
checkpoint-50/trainer_state.json ADDED
@@ -0,0 +1,1734 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.021367521367521368,
6
+ "eval_steps": 500,
7
+ "global_step": 50,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "clip_ratio/high_max": 0.0,
14
+ "clip_ratio/high_mean": 0.0,
15
+ "clip_ratio/low_mean": 0.0,
16
+ "clip_ratio/low_min": 0.0,
17
+ "clip_ratio/region_mean": 0.0,
18
+ "completions/clipped_ratio": 0.5555555820465088,
19
+ "completions/max_length": 3000.0,
20
+ "completions/max_terminated_length": 2837.0,
21
+ "completions/mean_length": 2626.97216796875,
22
+ "completions/mean_terminated_length": 2160.6875,
23
+ "completions/min_length": 1285.0,
24
+ "completions/min_terminated_length": 1285.0,
25
+ "entropy": 0.17718915144602457,
26
+ "epoch": 0.00042735042735042735,
27
+ "frac_reward_zero_std": 0.0,
28
+ "grad_norm": 0.07994551956653595,
29
+ "learning_rate": 0.0,
30
+ "loss": 0.0468,
31
+ "num_tokens": 102485.0,
32
+ "reward": -1.153435230255127,
33
+ "reward_std": 1.0195810794830322,
34
+ "rewards/hint_following/mean": -0.2777777910232544,
35
+ "rewards/hint_following/std": 0.8819171190261841,
36
+ "rewards/length_penalty/mean": -0.8756573796272278,
37
+ "rewards/length_penalty/std": 0.17644810676574707,
38
+ "sampling/importance_sampling_ratio/max": 3.0,
39
+ "sampling/importance_sampling_ratio/mean": 0.9937517642974854,
40
+ "sampling/importance_sampling_ratio/min": 0.16920268535614014,
41
+ "sampling/sampling_logp_difference/max": 1.7766579389572144,
42
+ "sampling/sampling_logp_difference/mean": 0.02078823931515217,
43
+ "step": 1,
44
+ "step_time": 67.38753714214545
45
+ },
46
+ {
47
+ "clip_ratio/high_max": 0.0,
48
+ "clip_ratio/high_mean": 0.0,
49
+ "clip_ratio/low_mean": 0.0,
50
+ "clip_ratio/low_min": 0.0,
51
+ "clip_ratio/region_mean": 0.0,
52
+ "completions/clipped_ratio": 0.25,
53
+ "completions/max_length": 3000.0,
54
+ "completions/max_terminated_length": 2961.0,
55
+ "completions/mean_length": 1809.4444580078125,
56
+ "completions/mean_terminated_length": 1412.5926513671875,
57
+ "completions/min_length": 858.0,
58
+ "completions/min_terminated_length": 858.0,
59
+ "entropy": 0.1950028215845426,
60
+ "epoch": 0.0008547008547008547,
61
+ "frac_reward_zero_std": 0.0,
62
+ "grad_norm": 0.07918090373277664,
63
+ "learning_rate": 3.846153846153847e-06,
64
+ "loss": 0.022,
65
+ "num_tokens": 177279.0,
66
+ "reward": -0.4642592668533325,
67
+ "reward_std": 1.043834924697876,
68
+ "rewards/hint_following/mean": 0.1388888955116272,
69
+ "rewards/hint_following/std": 0.798311710357666,
70
+ "rewards/length_penalty/mean": -0.6031482219696045,
71
+ "rewards/length_penalty/std": 0.28368300199508667,
72
+ "sampling/importance_sampling_ratio/max": 2.0890045166015625,
73
+ "sampling/importance_sampling_ratio/mean": 0.9933527708053589,
74
+ "sampling/importance_sampling_ratio/min": 0.29880645871162415,
75
+ "sampling/sampling_logp_difference/max": 1.2079591751098633,
76
+ "sampling/sampling_logp_difference/mean": 0.021868562325835228,
77
+ "step": 2,
78
+ "step_time": 59.25448451703414
79
+ },
80
+ {
81
+ "clip_ratio/high_max": 0.007054112308348219,
82
+ "clip_ratio/high_mean": 0.007054112308348219,
83
+ "clip_ratio/low_mean": 0.0024281112127937376,
84
+ "clip_ratio/low_min": 0.0024281112127937376,
85
+ "clip_ratio/region_mean": 0.009482223695764938,
86
+ "completions/clipped_ratio": 0.1666666716337204,
87
+ "completions/max_length": 3000.0,
88
+ "completions/max_terminated_length": 2411.0,
89
+ "completions/mean_length": 1750.8055419921875,
90
+ "completions/mean_terminated_length": 1500.966796875,
91
+ "completions/min_length": 939.0,
92
+ "completions/min_terminated_length": 939.0,
93
+ "entropy": 0.1983068659901619,
94
+ "epoch": 0.001282051282051282,
95
+ "frac_reward_zero_std": 0.0,
96
+ "grad_norm": 0.06695634126663208,
97
+ "learning_rate": 7.692307692307694e-06,
98
+ "loss": 0.0035,
99
+ "num_tokens": 245288.0,
100
+ "reward": -0.25026851892471313,
101
+ "reward_std": 0.9555131196975708,
102
+ "rewards/hint_following/mean": 0.3333333432674408,
103
+ "rewards/hint_following/std": 0.7559289932250977,
104
+ "rewards/length_penalty/mean": -0.5836018323898315,
105
+ "rewards/length_penalty/std": 0.22866536676883698,
106
+ "sampling/importance_sampling_ratio/max": 2.208913803100586,
107
+ "sampling/importance_sampling_ratio/mean": 0.9931869506835938,
108
+ "sampling/importance_sampling_ratio/min": 0.30834895372390747,
109
+ "sampling/sampling_logp_difference/max": 1.176523208618164,
110
+ "sampling/sampling_logp_difference/mean": 0.022443262860178947,
111
+ "step": 3,
112
+ "step_time": 55.47524960897863
113
+ },
114
+ {
115
+ "clip_ratio/high_max": 0.0025329499427850046,
116
+ "clip_ratio/high_mean": 0.0025329499427850046,
117
+ "clip_ratio/low_mean": 0.0070530143566429615,
118
+ "clip_ratio/low_min": 0.0070530143566429615,
119
+ "clip_ratio/region_mean": 0.009585964183012644,
120
+ "completions/clipped_ratio": 0.472222238779068,
121
+ "completions/max_length": 3000.0,
122
+ "completions/max_terminated_length": 2857.0,
123
+ "completions/mean_length": 2469.27783203125,
124
+ "completions/mean_terminated_length": 1994.4210205078125,
125
+ "completions/min_length": 627.0,
126
+ "completions/min_terminated_length": 627.0,
127
+ "entropy": 0.1949674834807714,
128
+ "epoch": 0.0017094017094017094,
129
+ "frac_reward_zero_std": 0.0,
130
+ "grad_norm": 0.08172055333852768,
131
+ "learning_rate": 1.153846153846154e-05,
132
+ "loss": 0.0355,
133
+ "num_tokens": 339768.0,
134
+ "reward": -1.0453147888183594,
135
+ "reward_std": 0.8787956237792969,
136
+ "rewards/hint_following/mean": -0.2222222238779068,
137
+ "rewards/hint_following/std": 0.7601169347763062,
138
+ "rewards/length_penalty/mean": -0.8230926394462585,
139
+ "rewards/length_penalty/std": 0.2654260993003845,
140
+ "sampling/importance_sampling_ratio/max": 3.0,
141
+ "sampling/importance_sampling_ratio/mean": 0.993114173412323,
142
+ "sampling/importance_sampling_ratio/min": 0.19437530636787415,
143
+ "sampling/sampling_logp_difference/max": 1.6379644870758057,
144
+ "sampling/sampling_logp_difference/mean": 0.022510718554258347,
145
+ "step": 4,
146
+ "step_time": 62.77226335310843
147
+ },
148
+ {
149
+ "clip_ratio/high_max": 0.001851275311006854,
150
+ "clip_ratio/high_mean": 0.001851275311006854,
151
+ "clip_ratio/low_mean": 0.00705820966201524,
152
+ "clip_ratio/low_min": 0.00705820966201524,
153
+ "clip_ratio/region_mean": 0.008909484914814433,
154
+ "completions/clipped_ratio": 0.4166666567325592,
155
+ "completions/max_length": 3000.0,
156
+ "completions/max_terminated_length": 2915.0,
157
+ "completions/mean_length": 2362.8056640625,
158
+ "completions/mean_terminated_length": 1907.666748046875,
159
+ "completions/min_length": 1109.0,
160
+ "completions/min_terminated_length": 1109.0,
161
+ "entropy": 0.18256587783495584,
162
+ "epoch": 0.002136752136752137,
163
+ "frac_reward_zero_std": 0.0,
164
+ "grad_norm": 0.072978176176548,
165
+ "learning_rate": 1.5384615384615387e-05,
166
+ "loss": 0.1055,
167
+ "num_tokens": 431687.0,
168
+ "reward": -0.9264907836914062,
169
+ "reward_std": 1.0278923511505127,
170
+ "rewards/hint_following/mean": -0.1388888955116272,
171
+ "rewards/hint_following/std": 0.8333333730697632,
172
+ "rewards/length_penalty/mean": -0.787601888179779,
173
+ "rewards/length_penalty/std": 0.2262839674949646,
174
+ "sampling/importance_sampling_ratio/max": 2.5575737953186035,
175
+ "sampling/importance_sampling_ratio/mean": 0.9932507872581482,
176
+ "sampling/importance_sampling_ratio/min": 0.2952674627304077,
177
+ "sampling/sampling_logp_difference/max": 1.2198736667633057,
178
+ "sampling/sampling_logp_difference/mean": 0.020803121849894524,
179
+ "step": 5,
180
+ "step_time": 61.06088067602832
181
+ },
182
+ {
183
+ "clip_ratio/high_max": 0.0030960943549871445,
184
+ "clip_ratio/high_mean": 0.0030960943549871445,
185
+ "clip_ratio/low_mean": 0.006137795591106017,
186
+ "clip_ratio/low_min": 0.006137795591106017,
187
+ "clip_ratio/region_mean": 0.009233890101313591,
188
+ "completions/clipped_ratio": 0.4166666567325592,
189
+ "completions/max_length": 3000.0,
190
+ "completions/max_terminated_length": 2929.0,
191
+ "completions/mean_length": 2417.444580078125,
192
+ "completions/mean_terminated_length": 2001.3333740234375,
193
+ "completions/min_length": 1434.0,
194
+ "completions/min_terminated_length": 1434.0,
195
+ "entropy": 0.1919809157649676,
196
+ "epoch": 0.002564102564102564,
197
+ "frac_reward_zero_std": 0.0,
198
+ "grad_norm": 0.07703683525323868,
199
+ "learning_rate": 1.923076923076923e-05,
200
+ "loss": 0.1019,
201
+ "num_tokens": 525555.0,
202
+ "reward": -0.7224815487861633,
203
+ "reward_std": 1.1388682126998901,
204
+ "rewards/hint_following/mean": 0.0833333358168602,
205
+ "rewards/hint_following/std": 0.9673232436180115,
206
+ "rewards/length_penalty/mean": -0.8058148622512817,
207
+ "rewards/length_penalty/std": 0.18899519741535187,
208
+ "sampling/importance_sampling_ratio/max": 2.9181816577911377,
209
+ "sampling/importance_sampling_ratio/mean": 0.9935172200202942,
210
+ "sampling/importance_sampling_ratio/min": 0.14165793359279633,
211
+ "sampling/sampling_logp_difference/max": 1.9543399810791016,
212
+ "sampling/sampling_logp_difference/mean": 0.021658122539520264,
213
+ "step": 6,
214
+ "step_time": 64.93633897381369
215
+ },
216
+ {
217
+ "clip_ratio/high_max": 0.004218692930104832,
218
+ "clip_ratio/high_mean": 0.004218692930104832,
219
+ "clip_ratio/low_mean": 0.0048820926652600365,
220
+ "clip_ratio/low_min": 0.0048820926652600365,
221
+ "clip_ratio/region_mean": 0.009100785556559762,
222
+ "completions/clipped_ratio": 0.2222222238779068,
223
+ "completions/max_length": 3000.0,
224
+ "completions/max_terminated_length": 2933.0,
225
+ "completions/mean_length": 2029.0833740234375,
226
+ "completions/mean_terminated_length": 1751.6785888671875,
227
+ "completions/min_length": 1009.0,
228
+ "completions/min_terminated_length": 1009.0,
229
+ "entropy": 0.20896865924199423,
230
+ "epoch": 0.0029914529914529917,
231
+ "frac_reward_zero_std": 0.0,
232
+ "grad_norm": 0.07187572866678238,
233
+ "learning_rate": 2.307692307692308e-05,
234
+ "loss": 0.0798,
235
+ "num_tokens": 606186.0,
236
+ "reward": -0.4819166362285614,
237
+ "reward_std": 0.9970327019691467,
238
+ "rewards/hint_following/mean": 0.1944444477558136,
239
+ "rewards/hint_following/std": 0.7862913012504578,
240
+ "rewards/length_penalty/mean": -0.676361083984375,
241
+ "rewards/length_penalty/std": 0.2366827428340912,
242
+ "sampling/importance_sampling_ratio/max": 2.6964023113250732,
243
+ "sampling/importance_sampling_ratio/mean": 0.9927431344985962,
244
+ "sampling/importance_sampling_ratio/min": 0.21346372365951538,
245
+ "sampling/sampling_logp_difference/max": 1.5442883968353271,
246
+ "sampling/sampling_logp_difference/mean": 0.022996528074145317,
247
+ "step": 7,
248
+ "step_time": 59.43198294797912
249
+ },
250
+ {
251
+ "clip_ratio/high_max": 0.007654782539854447,
252
+ "clip_ratio/high_mean": 0.007654782539854447,
253
+ "clip_ratio/low_mean": 0.002519401100774606,
254
+ "clip_ratio/low_min": 0.002519401100774606,
255
+ "clip_ratio/region_mean": 0.010174183640629053,
256
+ "completions/clipped_ratio": 0.1666666716337204,
257
+ "completions/max_length": 3000.0,
258
+ "completions/max_terminated_length": 2957.0,
259
+ "completions/mean_length": 2227.97216796875,
260
+ "completions/mean_terminated_length": 2073.56689453125,
261
+ "completions/min_length": 1197.0,
262
+ "completions/min_terminated_length": 1197.0,
263
+ "entropy": 0.19751630226771036,
264
+ "epoch": 0.003418803418803419,
265
+ "frac_reward_zero_std": 0.0,
266
+ "grad_norm": 0.08499564230442047,
267
+ "learning_rate": 2.6923076923076923e-05,
268
+ "loss": 0.0572,
269
+ "num_tokens": 692279.0,
270
+ "reward": -0.4093240797519684,
271
+ "reward_std": 0.8091796040534973,
272
+ "rewards/hint_following/mean": 0.3333333432674408,
273
+ "rewards/hint_following/std": 0.7171371579170227,
274
+ "rewards/length_penalty/mean": -0.7426574230194092,
275
+ "rewards/length_penalty/std": 0.19112928211688995,
276
+ "sampling/importance_sampling_ratio/max": 2.491806745529175,
277
+ "sampling/importance_sampling_ratio/mean": 0.9929631352424622,
278
+ "sampling/importance_sampling_ratio/min": 0.3287244141101837,
279
+ "sampling/sampling_logp_difference/max": 1.1125354766845703,
280
+ "sampling/sampling_logp_difference/mean": 0.021952621638774872,
281
+ "step": 8,
282
+ "step_time": 60.087568536167964
283
+ },
284
+ {
285
+ "clip_ratio/high_max": 0.007754642438764374,
286
+ "clip_ratio/high_mean": 0.007754642438764374,
287
+ "clip_ratio/low_mean": 0.0032427439582534134,
288
+ "clip_ratio/low_min": 0.0032427439582534134,
289
+ "clip_ratio/region_mean": 0.010997386649250984,
290
+ "completions/clipped_ratio": 0.1111111119389534,
291
+ "completions/max_length": 3000.0,
292
+ "completions/max_terminated_length": 2948.0,
293
+ "completions/mean_length": 2034.2222900390625,
294
+ "completions/mean_terminated_length": 1913.5,
295
+ "completions/min_length": 778.0,
296
+ "completions/min_terminated_length": 778.0,
297
+ "entropy": 0.21248381833235422,
298
+ "epoch": 0.0038461538461538464,
299
+ "frac_reward_zero_std": 0.0,
300
+ "grad_norm": 0.07896988093852997,
301
+ "learning_rate": 3.0769230769230774e-05,
302
+ "loss": 0.0952,
303
+ "num_tokens": 771241.0,
304
+ "reward": -0.28918519616127014,
305
+ "reward_std": 0.7873722314834595,
306
+ "rewards/hint_following/mean": 0.3888888955116272,
307
+ "rewards/hint_following/std": 0.6877615451812744,
308
+ "rewards/length_penalty/mean": -0.6780741214752197,
309
+ "rewards/length_penalty/std": 0.22194787859916687,
310
+ "sampling/importance_sampling_ratio/max": 2.757654905319214,
311
+ "sampling/importance_sampling_ratio/mean": 0.9926807880401611,
312
+ "sampling/importance_sampling_ratio/min": 0.19047939777374268,
313
+ "sampling/sampling_logp_difference/max": 1.6582112312316895,
314
+ "sampling/sampling_logp_difference/mean": 0.02367428131401539,
315
+ "step": 9,
316
+ "step_time": 58.2687593010487
317
+ },
318
+ {
319
+ "clip_ratio/high_max": 0.0022279491822700948,
320
+ "clip_ratio/high_mean": 0.0022279491822700948,
321
+ "clip_ratio/low_mean": 0.0019324645400047302,
322
+ "clip_ratio/low_min": 0.0019324645400047302,
323
+ "clip_ratio/region_mean": 0.004160413751378655,
324
+ "completions/clipped_ratio": 0.3611111044883728,
325
+ "completions/max_length": 3000.0,
326
+ "completions/max_terminated_length": 2960.0,
327
+ "completions/mean_length": 2266.111083984375,
328
+ "completions/mean_terminated_length": 1851.3043212890625,
329
+ "completions/min_length": 784.0,
330
+ "completions/min_terminated_length": 784.0,
331
+ "entropy": 0.20578551292419434,
332
+ "epoch": 0.004273504273504274,
333
+ "frac_reward_zero_std": 0.0,
334
+ "grad_norm": 0.06615690886974335,
335
+ "learning_rate": 3.461538461538462e-05,
336
+ "loss": 0.0543,
337
+ "num_tokens": 858521.0,
338
+ "reward": -0.5331481099128723,
339
+ "reward_std": 1.1243869066238403,
340
+ "rewards/hint_following/mean": 0.2222222238779068,
341
+ "rewards/hint_following/std": 0.929242730140686,
342
+ "rewards/length_penalty/mean": -0.7553703784942627,
343
+ "rewards/length_penalty/std": 0.23448802530765533,
344
+ "sampling/importance_sampling_ratio/max": 2.0015952587127686,
345
+ "sampling/importance_sampling_ratio/mean": 0.9925298094749451,
346
+ "sampling/importance_sampling_ratio/min": 0.30108708143234253,
347
+ "sampling/sampling_logp_difference/max": 1.2003557682037354,
348
+ "sampling/sampling_logp_difference/mean": 0.021938424557447433,
349
+ "step": 10,
350
+ "step_time": 60.113012868212536
351
+ },
352
+ {
353
+ "clip_ratio/high_max": 0.002628828883947184,
354
+ "clip_ratio/high_mean": 0.002628828883947184,
355
+ "clip_ratio/low_mean": 0.0007913654941755036,
356
+ "clip_ratio/low_min": 0.0007913654941755036,
357
+ "clip_ratio/region_mean": 0.0034201944169277945,
358
+ "completions/clipped_ratio": 0.3888888955116272,
359
+ "completions/max_length": 3000.0,
360
+ "completions/max_terminated_length": 2856.0,
361
+ "completions/mean_length": 2308.111083984375,
362
+ "completions/mean_terminated_length": 1867.8182373046875,
363
+ "completions/min_length": 853.0,
364
+ "completions/min_terminated_length": 853.0,
365
+ "entropy": 0.1780667081475258,
366
+ "epoch": 0.004700854700854701,
367
+ "frac_reward_zero_std": 0.0,
368
+ "grad_norm": 0.06938029080629349,
369
+ "learning_rate": 3.846153846153846e-05,
370
+ "loss": 0.0506,
371
+ "num_tokens": 948171.0,
372
+ "reward": -1.0471482276916504,
373
+ "reward_std": 0.824184238910675,
374
+ "rewards/hint_following/mean": -0.2777777910232544,
375
+ "rewards/hint_following/std": 0.6594851613044739,
376
+ "rewards/length_penalty/mean": -0.7693703770637512,
377
+ "rewards/length_penalty/std": 0.24463628232479095,
378
+ "sampling/importance_sampling_ratio/max": 2.5170931816101074,
379
+ "sampling/importance_sampling_ratio/mean": 0.9934613108634949,
380
+ "sampling/importance_sampling_ratio/min": 0.29526859521865845,
381
+ "sampling/sampling_logp_difference/max": 1.21986985206604,
382
+ "sampling/sampling_logp_difference/mean": 0.021038096398115158,
383
+ "step": 11,
384
+ "step_time": 61.20216279185843
385
+ },
386
+ {
387
+ "clip_ratio/high_max": 0.009717889750997225,
388
+ "clip_ratio/high_mean": 0.009717889750997225,
389
+ "clip_ratio/low_mean": 0.001428690991209199,
390
+ "clip_ratio/low_min": 0.001428690991209199,
391
+ "clip_ratio/region_mean": 0.011146580955634514,
392
+ "completions/clipped_ratio": 0.2777777910232544,
393
+ "completions/max_length": 3000.0,
394
+ "completions/max_terminated_length": 2997.0,
395
+ "completions/mean_length": 2494.944580078125,
396
+ "completions/mean_terminated_length": 2300.6923828125,
397
+ "completions/min_length": 1195.0,
398
+ "completions/min_terminated_length": 1195.0,
399
+ "entropy": 0.21741948276758194,
400
+ "epoch": 0.005128205128205128,
401
+ "frac_reward_zero_std": 0.0,
402
+ "grad_norm": 0.08027122914791107,
403
+ "learning_rate": 4.230769230769231e-05,
404
+ "loss": 0.0271,
405
+ "num_tokens": 1045165.0,
406
+ "reward": -0.6094259023666382,
407
+ "reward_std": 0.9179477691650391,
408
+ "rewards/hint_following/mean": 0.2222222238779068,
409
+ "rewards/hint_following/std": 0.8655670881271362,
410
+ "rewards/length_penalty/mean": -0.8316481113433838,
411
+ "rewards/length_penalty/std": 0.1910414844751358,
412
+ "sampling/importance_sampling_ratio/max": 2.7227704524993896,
413
+ "sampling/importance_sampling_ratio/mean": 0.9923471808433533,
414
+ "sampling/importance_sampling_ratio/min": 0.309160441160202,
415
+ "sampling/sampling_logp_difference/max": 1.1738948822021484,
416
+ "sampling/sampling_logp_difference/mean": 0.02328832633793354,
417
+ "step": 12,
418
+ "step_time": 63.202891704044305
419
+ },
420
+ {
421
+ "clip_ratio/high_max": 0.0020594243348265686,
422
+ "clip_ratio/high_mean": 0.0020594243348265686,
423
+ "clip_ratio/low_mean": 0.0059360921538124485,
424
+ "clip_ratio/low_min": 0.0059360921538124485,
425
+ "clip_ratio/region_mean": 0.007995516527444124,
426
+ "completions/clipped_ratio": 0.2222222238779068,
427
+ "completions/max_length": 3000.0,
428
+ "completions/max_terminated_length": 2981.0,
429
+ "completions/mean_length": 2085.638916015625,
430
+ "completions/mean_terminated_length": 1824.3929443359375,
431
+ "completions/min_length": 956.0,
432
+ "completions/min_terminated_length": 956.0,
433
+ "entropy": 0.18983352681001028,
434
+ "epoch": 0.005555555555555556,
435
+ "frac_reward_zero_std": 0.0,
436
+ "grad_norm": 0.2806412875652313,
437
+ "learning_rate": 4.615384615384616e-05,
438
+ "loss": 0.0294,
439
+ "num_tokens": 1128414.0,
440
+ "reward": -0.6674352288246155,
441
+ "reward_std": 0.8722909092903137,
442
+ "rewards/hint_following/mean": 0.02777777798473835,
443
+ "rewards/hint_following/std": 0.6963624954223633,
444
+ "rewards/length_penalty/mean": -0.6952130198478699,
445
+ "rewards/length_penalty/std": 0.23719921708106995,
446
+ "sampling/importance_sampling_ratio/max": 2.7140581607818604,
447
+ "sampling/importance_sampling_ratio/mean": 0.9935481548309326,
448
+ "sampling/importance_sampling_ratio/min": 0.19567115604877472,
449
+ "sampling/sampling_logp_difference/max": 1.6313197612762451,
450
+ "sampling/sampling_logp_difference/mean": 0.02119005285203457,
451
+ "step": 13,
452
+ "step_time": 61.55426321423147
453
+ },
454
+ {
455
+ "clip_ratio/high_max": 0.0014092368267786999,
456
+ "clip_ratio/high_mean": 0.0014092368267786999,
457
+ "clip_ratio/low_mean": 0.007215868448838592,
458
+ "clip_ratio/low_min": 0.007215868448838592,
459
+ "clip_ratio/region_mean": 0.008625105411435166,
460
+ "completions/clipped_ratio": 0.1944444477558136,
461
+ "completions/max_length": 3000.0,
462
+ "completions/max_terminated_length": 2941.0,
463
+ "completions/mean_length": 1680.3055419921875,
464
+ "completions/mean_terminated_length": 1361.7586669921875,
465
+ "completions/min_length": 398.0,
466
+ "completions/min_terminated_length": 398.0,
467
+ "entropy": 0.20419775694608688,
468
+ "epoch": 0.005982905982905983,
469
+ "frac_reward_zero_std": 0.0,
470
+ "grad_norm": 0.08586689084768295,
471
+ "learning_rate": 5e-05,
472
+ "loss": 0.0681,
473
+ "num_tokens": 1193759.0,
474
+ "reward": -0.5878795981407166,
475
+ "reward_std": 0.8454972505569458,
476
+ "rewards/hint_following/mean": -0.02777777798473835,
477
+ "rewards/hint_following/std": 0.6087979674339294,
478
+ "rewards/length_penalty/mean": -0.5601018071174622,
479
+ "rewards/length_penalty/std": 0.31150683760643005,
480
+ "sampling/importance_sampling_ratio/max": 2.1082680225372314,
481
+ "sampling/importance_sampling_ratio/mean": 0.992834746837616,
482
+ "sampling/importance_sampling_ratio/min": 0.3020838499069214,
483
+ "sampling/sampling_logp_difference/max": 1.1970505714416504,
484
+ "sampling/sampling_logp_difference/mean": 0.02261131815612316,
485
+ "step": 14,
486
+ "step_time": 53.72449496493209
487
+ },
488
+ {
489
+ "clip_ratio/high_max": 0.007441268069669604,
490
+ "clip_ratio/high_mean": 0.007441268069669604,
491
+ "clip_ratio/low_mean": 0.002903422418360909,
492
+ "clip_ratio/low_min": 0.002903422418360909,
493
+ "clip_ratio/region_mean": 0.01034469079847137,
494
+ "completions/clipped_ratio": 0.25,
495
+ "completions/max_length": 3000.0,
496
+ "completions/max_terminated_length": 2906.0,
497
+ "completions/mean_length": 2191.111083984375,
498
+ "completions/mean_terminated_length": 1921.4814453125,
499
+ "completions/min_length": 397.0,
500
+ "completions/min_terminated_length": 397.0,
501
+ "entropy": 0.22065167874097824,
502
+ "epoch": 0.00641025641025641,
503
+ "frac_reward_zero_std": 0.0,
504
+ "grad_norm": 0.07726756483316422,
505
+ "learning_rate": 5e-05,
506
+ "loss": 0.0177,
507
+ "num_tokens": 1279125.0,
508
+ "reward": -0.7859259247779846,
509
+ "reward_std": 0.7896111607551575,
510
+ "rewards/hint_following/mean": -0.0555555559694767,
511
+ "rewards/hint_following/std": 0.6737716794013977,
512
+ "rewards/length_penalty/mean": -0.7303703427314758,
513
+ "rewards/length_penalty/std": 0.2772618532180786,
514
+ "sampling/importance_sampling_ratio/max": 2.1340627670288086,
515
+ "sampling/importance_sampling_ratio/mean": 0.9917824268341064,
516
+ "sampling/importance_sampling_ratio/min": 0.07570598274469376,
517
+ "sampling/sampling_logp_difference/max": 2.5808980464935303,
518
+ "sampling/sampling_logp_difference/mean": 0.023916445672512054,
519
+ "step": 15,
520
+ "step_time": 59.183635055902414
521
+ },
522
+ {
523
+ "clip_ratio/high_max": 0.007959824210653702,
524
+ "clip_ratio/high_mean": 0.007959824210653702,
525
+ "clip_ratio/low_mean": 0.0026605394668877125,
526
+ "clip_ratio/low_min": 0.0026605394668877125,
527
+ "clip_ratio/region_mean": 0.010620363832761845,
528
+ "completions/clipped_ratio": 0.1388888955116272,
529
+ "completions/max_length": 3000.0,
530
+ "completions/max_terminated_length": 2836.0,
531
+ "completions/mean_length": 1801.9722900390625,
532
+ "completions/mean_terminated_length": 1612.0,
533
+ "completions/min_length": 660.0,
534
+ "completions/min_terminated_length": 660.0,
535
+ "entropy": 0.2382892519235611,
536
+ "epoch": 0.006837606837606838,
537
+ "frac_reward_zero_std": 0.0,
538
+ "grad_norm": 0.07441020756959915,
539
+ "learning_rate": 5e-05,
540
+ "loss": 0.1035,
541
+ "num_tokens": 1351580.0,
542
+ "reward": -0.2951018214225769,
543
+ "reward_std": 0.9211878180503845,
544
+ "rewards/hint_following/mean": 0.3055555522441864,
545
+ "rewards/hint_following/std": 0.7099072337150574,
546
+ "rewards/length_penalty/mean": -0.6006573438644409,
547
+ "rewards/length_penalty/std": 0.2538762390613556,
548
+ "sampling/importance_sampling_ratio/max": 2.141491413116455,
549
+ "sampling/importance_sampling_ratio/mean": 0.9912582039833069,
550
+ "sampling/importance_sampling_ratio/min": 0.3961084187030792,
551
+ "sampling/sampling_logp_difference/max": 0.9260673522949219,
552
+ "sampling/sampling_logp_difference/mean": 0.024969136342406273,
553
+ "step": 16,
554
+ "step_time": 58.414970112848096
555
+ },
556
+ {
557
+ "clip_ratio/high_max": 0.00782025900358955,
558
+ "clip_ratio/high_mean": 0.00782025900358955,
559
+ "clip_ratio/low_mean": 0.0016357484661663573,
560
+ "clip_ratio/low_min": 0.0016357484661663573,
561
+ "clip_ratio/region_mean": 0.00945600758617123,
562
+ "completions/clipped_ratio": 0.0833333358168602,
563
+ "completions/max_length": 3000.0,
564
+ "completions/max_terminated_length": 2959.0,
565
+ "completions/mean_length": 1858.02783203125,
566
+ "completions/mean_terminated_length": 1754.212158203125,
567
+ "completions/min_length": 523.0,
568
+ "completions/min_terminated_length": 523.0,
569
+ "entropy": 0.2335476577281952,
570
+ "epoch": 0.007264957264957265,
571
+ "frac_reward_zero_std": 0.0,
572
+ "grad_norm": 0.06890398263931274,
573
+ "learning_rate": 5e-05,
574
+ "loss": 0.0923,
575
+ "num_tokens": 1425669.0,
576
+ "reward": 0.10287962108850479,
577
+ "reward_std": 0.7695233225822449,
578
+ "rewards/hint_following/mean": 0.7222222089767456,
579
+ "rewards/hint_following/std": 0.6146363019943237,
580
+ "rewards/length_penalty/mean": -0.619342565536499,
581
+ "rewards/length_penalty/std": 0.2540966868400574,
582
+ "sampling/importance_sampling_ratio/max": 2.8489930629730225,
583
+ "sampling/importance_sampling_ratio/mean": 0.9912996888160706,
584
+ "sampling/importance_sampling_ratio/min": 0.3177354037761688,
585
+ "sampling/sampling_logp_difference/max": 1.1465363502502441,
586
+ "sampling/sampling_logp_difference/mean": 0.0239370446652174,
587
+ "step": 17,
588
+ "step_time": 56.41279660095461
589
+ },
590
+ {
591
+ "clip_ratio/high_max": 0.008962325053289533,
592
+ "clip_ratio/high_mean": 0.008962325053289533,
593
+ "clip_ratio/low_mean": 0.001638414493451516,
594
+ "clip_ratio/low_min": 0.001638414493451516,
595
+ "clip_ratio/region_mean": 0.010600739469130835,
596
+ "completions/clipped_ratio": 0.2222222238779068,
597
+ "completions/max_length": 3000.0,
598
+ "completions/max_terminated_length": 2681.0,
599
+ "completions/mean_length": 1895.4722900390625,
600
+ "completions/mean_terminated_length": 1579.8929443359375,
601
+ "completions/min_length": 409.0,
602
+ "completions/min_terminated_length": 409.0,
603
+ "entropy": 0.2237540160616239,
604
+ "epoch": 0.007692307692307693,
605
+ "frac_reward_zero_std": 0.0,
606
+ "grad_norm": 0.07782972604036331,
607
+ "learning_rate": 5e-05,
608
+ "loss": 0.0547,
609
+ "num_tokens": 1501340.0,
610
+ "reward": -0.3818241059780121,
611
+ "reward_std": 0.9488492608070374,
612
+ "rewards/hint_following/mean": 0.25,
613
+ "rewards/hint_following/std": 0.8062257766723633,
614
+ "rewards/length_penalty/mean": -0.6318240761756897,
615
+ "rewards/length_penalty/std": 0.2734360098838806,
616
+ "sampling/importance_sampling_ratio/max": 2.2928225994110107,
617
+ "sampling/importance_sampling_ratio/mean": 0.9916783571243286,
618
+ "sampling/importance_sampling_ratio/min": 0.3001212775707245,
619
+ "sampling/sampling_logp_difference/max": 1.203568696975708,
620
+ "sampling/sampling_logp_difference/mean": 0.023737527430057526,
621
+ "step": 18,
622
+ "step_time": 58.57279848307371
623
+ },
624
+ {
625
+ "clip_ratio/high_max": 0.0038812818626562753,
626
+ "clip_ratio/high_mean": 0.0038812818626562753,
627
+ "clip_ratio/low_mean": 0.005094039059864978,
628
+ "clip_ratio/low_min": 0.005094039059864978,
629
+ "clip_ratio/region_mean": 0.00897532096132636,
630
+ "completions/clipped_ratio": 0.1666666716337204,
631
+ "completions/max_length": 3000.0,
632
+ "completions/max_terminated_length": 2937.0,
633
+ "completions/mean_length": 2020.138916015625,
634
+ "completions/mean_terminated_length": 1824.166748046875,
635
+ "completions/min_length": 765.0,
636
+ "completions/min_terminated_length": 765.0,
637
+ "entropy": 0.2477496862411499,
638
+ "epoch": 0.00811965811965812,
639
+ "frac_reward_zero_std": 0.0,
640
+ "grad_norm": 0.07387188822031021,
641
+ "learning_rate": 5e-05,
642
+ "loss": 0.0524,
643
+ "num_tokens": 1581145.0,
644
+ "reward": -0.39560189843177795,
645
+ "reward_std": 0.9154050350189209,
646
+ "rewards/hint_following/mean": 0.2777777910232544,
647
+ "rewards/hint_following/std": 0.7410845756530762,
648
+ "rewards/length_penalty/mean": -0.67337965965271,
649
+ "rewards/length_penalty/std": 0.24809792637825012,
650
+ "sampling/importance_sampling_ratio/max": 2.1414597034454346,
651
+ "sampling/importance_sampling_ratio/mean": 0.9909990429878235,
652
+ "sampling/importance_sampling_ratio/min": 0.250805139541626,
653
+ "sampling/sampling_logp_difference/max": 1.383078932762146,
654
+ "sampling/sampling_logp_difference/mean": 0.025834236294031143,
655
+ "step": 19,
656
+ "step_time": 58.03654656501021
657
+ },
658
+ {
659
+ "clip_ratio/high_max": 0.0034701917708540955,
660
+ "clip_ratio/high_mean": 0.0034701917708540955,
661
+ "clip_ratio/low_mean": 0.004899940686300397,
662
+ "clip_ratio/low_min": 0.004899940686300397,
663
+ "clip_ratio/region_mean": 0.00837013234073917,
664
+ "completions/clipped_ratio": 0.2777777910232544,
665
+ "completions/max_length": 3000.0,
666
+ "completions/max_terminated_length": 2645.0,
667
+ "completions/mean_length": 1926.1666259765625,
668
+ "completions/mean_terminated_length": 1513.1539306640625,
669
+ "completions/min_length": 1111.0,
670
+ "completions/min_terminated_length": 1111.0,
671
+ "entropy": 0.2643888195355733,
672
+ "epoch": 0.008547008547008548,
673
+ "frac_reward_zero_std": 0.0,
674
+ "grad_norm": 0.07248850166797638,
675
+ "learning_rate": 5e-05,
676
+ "loss": 0.0748,
677
+ "num_tokens": 1658131.0,
678
+ "reward": -0.25316667556762695,
679
+ "reward_std": 1.1320863962173462,
680
+ "rewards/hint_following/mean": 0.3888888955116272,
681
+ "rewards/hint_following/std": 0.903256893157959,
682
+ "rewards/length_penalty/mean": -0.6420555114746094,
683
+ "rewards/length_penalty/std": 0.2418719381093979,
684
+ "sampling/importance_sampling_ratio/max": 2.639763593673706,
685
+ "sampling/importance_sampling_ratio/mean": 0.990202009677887,
686
+ "sampling/importance_sampling_ratio/min": 0.3388563394546509,
687
+ "sampling/sampling_logp_difference/max": 1.082179069519043,
688
+ "sampling/sampling_logp_difference/mean": 0.026784788817167282,
689
+ "step": 20,
690
+ "step_time": 59.42642685002647
691
+ },
692
+ {
693
+ "clip_ratio/high_max": 0.0018190126866102219,
694
+ "clip_ratio/high_mean": 0.0018190126866102219,
695
+ "clip_ratio/low_mean": 0.005009224289096892,
696
+ "clip_ratio/low_min": 0.005009224289096892,
697
+ "clip_ratio/region_mean": 0.006828237092122436,
698
+ "completions/clipped_ratio": 0.3611111044883728,
699
+ "completions/max_length": 3000.0,
700
+ "completions/max_terminated_length": 2893.0,
701
+ "completions/mean_length": 2087.0556640625,
702
+ "completions/mean_terminated_length": 1571.04345703125,
703
+ "completions/min_length": 621.0,
704
+ "completions/min_terminated_length": 621.0,
705
+ "entropy": 0.23874077697594961,
706
+ "epoch": 0.008974358974358974,
707
+ "frac_reward_zero_std": 0.0,
708
+ "grad_norm": 0.07083278149366379,
709
+ "learning_rate": 5e-05,
710
+ "loss": 0.0432,
711
+ "num_tokens": 1741971.0,
712
+ "reward": -0.7512407302856445,
713
+ "reward_std": 1.0566543340682983,
714
+ "rewards/hint_following/mean": -0.0555555559694767,
715
+ "rewards/hint_following/std": 0.826159656047821,
716
+ "rewards/length_penalty/mean": -0.6956851482391357,
717
+ "rewards/length_penalty/std": 0.29789718985557556,
718
+ "sampling/importance_sampling_ratio/max": 2.2389557361602783,
719
+ "sampling/importance_sampling_ratio/mean": 0.9909273386001587,
720
+ "sampling/importance_sampling_ratio/min": 0.1903061866760254,
721
+ "sampling/sampling_logp_difference/max": 1.659121036529541,
722
+ "sampling/sampling_logp_difference/mean": 0.024532614275813103,
723
+ "step": 21,
724
+ "step_time": 62.53779820702039
725
+ },
726
+ {
727
+ "clip_ratio/high_max": 0.002690806732668231,
728
+ "clip_ratio/high_mean": 0.002690806732668231,
729
+ "clip_ratio/low_mean": 0.0009356096270494163,
730
+ "clip_ratio/low_min": 0.0009356096270494163,
731
+ "clip_ratio/region_mean": 0.003626416379120201,
732
+ "completions/clipped_ratio": 0.1111111119389534,
733
+ "completions/max_length": 3000.0,
734
+ "completions/max_terminated_length": 2897.0,
735
+ "completions/mean_length": 1785.02783203125,
736
+ "completions/mean_terminated_length": 1633.15625,
737
+ "completions/min_length": 687.0,
738
+ "completions/min_terminated_length": 687.0,
739
+ "entropy": 0.2271458531419436,
740
+ "epoch": 0.009401709401709401,
741
+ "frac_reward_zero_std": 0.0,
742
+ "grad_norm": 0.06376982480287552,
743
+ "learning_rate": 5e-05,
744
+ "loss": 0.0558,
745
+ "num_tokens": 1812208.0,
746
+ "reward": -0.12278702855110168,
747
+ "reward_std": 0.8545697927474976,
748
+ "rewards/hint_following/mean": 0.4722222089767456,
749
+ "rewards/hint_following/std": 0.6963624358177185,
750
+ "rewards/length_penalty/mean": -0.5950092673301697,
751
+ "rewards/length_penalty/std": 0.2556186616420746,
752
+ "sampling/importance_sampling_ratio/max": 2.0840954780578613,
753
+ "sampling/importance_sampling_ratio/mean": 0.9917669892311096,
754
+ "sampling/importance_sampling_ratio/min": 0.4361201226711273,
755
+ "sampling/sampling_logp_difference/max": 0.8298375606536865,
756
+ "sampling/sampling_logp_difference/mean": 0.02324368618428707,
757
+ "step": 22,
758
+ "step_time": 54.28889209416229
759
+ },
760
+ {
761
+ "clip_ratio/high_max": 0.00034443634406973916,
762
+ "clip_ratio/high_mean": 0.00034443634406973916,
763
+ "clip_ratio/low_mean": 0.001613353689511617,
764
+ "clip_ratio/low_min": 0.001613353689511617,
765
+ "clip_ratio/region_mean": 0.001957789994776249,
766
+ "completions/clipped_ratio": 0.0,
767
+ "completions/max_length": 2755.0,
768
+ "completions/max_terminated_length": 2755.0,
769
+ "completions/mean_length": 1515.5555419921875,
770
+ "completions/mean_terminated_length": 1515.5555419921875,
771
+ "completions/min_length": 761.0,
772
+ "completions/min_terminated_length": 761.0,
773
+ "entropy": 0.26160697638988495,
774
+ "epoch": 0.009829059829059829,
775
+ "frac_reward_zero_std": 0.0,
776
+ "grad_norm": 0.05521637201309204,
777
+ "learning_rate": 5e-05,
778
+ "loss": 0.0467,
779
+ "num_tokens": 1873452.0,
780
+ "reward": 0.18925924599170685,
781
+ "reward_std": 0.6213830709457397,
782
+ "rewards/hint_following/mean": 0.6944444179534912,
783
+ "rewards/hint_following/std": 0.467176616191864,
784
+ "rewards/length_penalty/mean": -0.5051851868629456,
785
+ "rewards/length_penalty/std": 0.1804182529449463,
786
+ "sampling/importance_sampling_ratio/max": 2.6960017681121826,
787
+ "sampling/importance_sampling_ratio/mean": 0.9903501272201538,
788
+ "sampling/importance_sampling_ratio/min": 0.2622704803943634,
789
+ "sampling/sampling_logp_difference/max": 1.33837890625,
790
+ "sampling/sampling_logp_difference/mean": 0.026966720819473267,
791
+ "step": 23,
792
+ "step_time": 49.74234241596423
793
+ },
794
+ {
795
+ "clip_ratio/high_max": 0.009169099697222313,
796
+ "clip_ratio/high_mean": 0.009169099697222313,
797
+ "clip_ratio/low_mean": 0.0011019935482181609,
798
+ "clip_ratio/low_min": 0.0011019935482181609,
799
+ "clip_ratio/region_mean": 0.010271093342453241,
800
+ "completions/clipped_ratio": 0.02777777798473835,
801
+ "completions/max_length": 3000.0,
802
+ "completions/max_terminated_length": 2438.0,
803
+ "completions/mean_length": 1354.77783203125,
804
+ "completions/mean_terminated_length": 1307.771484375,
805
+ "completions/min_length": 358.0,
806
+ "completions/min_terminated_length": 358.0,
807
+ "entropy": 0.28231214980284375,
808
+ "epoch": 0.010256410256410256,
809
+ "frac_reward_zero_std": 0.0,
810
+ "grad_norm": 0.061877232044935226,
811
+ "learning_rate": 5e-05,
812
+ "loss": 0.0513,
813
+ "num_tokens": 1928578.0,
814
+ "reward": 0.0761851817369461,
815
+ "reward_std": 0.6364034414291382,
816
+ "rewards/hint_following/mean": 0.5277777910232544,
817
+ "rewards/hint_following/std": 0.559903621673584,
818
+ "rewards/length_penalty/mean": -0.4515925645828247,
819
+ "rewards/length_penalty/std": 0.2236867994070053,
820
+ "sampling/importance_sampling_ratio/max": 2.3720901012420654,
821
+ "sampling/importance_sampling_ratio/mean": 0.9896314144134521,
822
+ "sampling/importance_sampling_ratio/min": 0.3866640329360962,
823
+ "sampling/sampling_logp_difference/max": 0.9501991271972656,
824
+ "sampling/sampling_logp_difference/mean": 0.028320252895355225,
825
+ "step": 24,
826
+ "step_time": 51.51962701487355
827
+ },
828
+ {
829
+ "clip_ratio/high_max": 0.007340530709673961,
830
+ "clip_ratio/high_mean": 0.007340530709673961,
831
+ "clip_ratio/low_mean": 0.002632848802022636,
832
+ "clip_ratio/low_min": 0.002632848802022636,
833
+ "clip_ratio/region_mean": 0.00997337931767106,
834
+ "completions/clipped_ratio": 0.0555555559694767,
835
+ "completions/max_length": 3000.0,
836
+ "completions/max_terminated_length": 2833.0,
837
+ "completions/mean_length": 1369.52783203125,
838
+ "completions/mean_terminated_length": 1273.61767578125,
839
+ "completions/min_length": 489.0,
840
+ "completions/min_terminated_length": 489.0,
841
+ "entropy": 0.24688356121381125,
842
+ "epoch": 0.010683760683760684,
843
+ "frac_reward_zero_std": 0.0,
844
+ "grad_norm": 0.0653451606631279,
845
+ "learning_rate": 5e-05,
846
+ "loss": 0.0562,
847
+ "num_tokens": 1983563.0,
848
+ "reward": -0.15095369517803192,
849
+ "reward_std": 0.692315936088562,
850
+ "rewards/hint_following/mean": 0.3055555522441864,
851
+ "rewards/hint_following/std": 0.576662540435791,
852
+ "rewards/length_penalty/mean": -0.4565092623233795,
853
+ "rewards/length_penalty/std": 0.22569186985492706,
854
+ "sampling/importance_sampling_ratio/max": 2.21685791015625,
855
+ "sampling/importance_sampling_ratio/mean": 0.9907597899436951,
856
+ "sampling/importance_sampling_ratio/min": 0.26015329360961914,
857
+ "sampling/sampling_logp_difference/max": 1.3464841842651367,
858
+ "sampling/sampling_logp_difference/mean": 0.02600344829261303,
859
+ "step": 25,
860
+ "step_time": 53.30105395393912
861
+ },
862
+ {
863
+ "clip_ratio/high_max": 0.005985865951515734,
864
+ "clip_ratio/high_mean": 0.005985865951515734,
865
+ "clip_ratio/low_mean": 0.0026564535413247845,
866
+ "clip_ratio/low_min": 0.0026564535413247845,
867
+ "clip_ratio/region_mean": 0.008642319512243072,
868
+ "completions/clipped_ratio": 0.0833333358168602,
869
+ "completions/max_length": 3000.0,
870
+ "completions/max_terminated_length": 2935.0,
871
+ "completions/mean_length": 1874.6944580078125,
872
+ "completions/mean_terminated_length": 1772.39404296875,
873
+ "completions/min_length": 921.0,
874
+ "completions/min_terminated_length": 921.0,
875
+ "entropy": 0.24719314028819403,
876
+ "epoch": 0.011111111111111112,
877
+ "frac_reward_zero_std": 0.0,
878
+ "grad_norm": 0.07119674980640411,
879
+ "learning_rate": 5e-05,
880
+ "loss": 0.0874,
881
+ "num_tokens": 2058072.0,
882
+ "reward": -0.06934259831905365,
883
+ "reward_std": 0.8091042637825012,
884
+ "rewards/hint_following/mean": 0.5555555820465088,
885
+ "rewards/hint_following/std": 0.6522245407104492,
886
+ "rewards/length_penalty/mean": -0.6248981952667236,
887
+ "rewards/length_penalty/std": 0.21120469272136688,
888
+ "sampling/importance_sampling_ratio/max": 2.3882269859313965,
889
+ "sampling/importance_sampling_ratio/mean": 0.9909408688545227,
890
+ "sampling/importance_sampling_ratio/min": 0.3095887005329132,
891
+ "sampling/sampling_logp_difference/max": 1.1725106239318848,
892
+ "sampling/sampling_logp_difference/mean": 0.024691704660654068,
893
+ "step": 26,
894
+ "step_time": 55.06593968113884
895
+ },
896
+ {
897
+ "clip_ratio/high_max": 0.006906990543939173,
898
+ "clip_ratio/high_mean": 0.006906990543939173,
899
+ "clip_ratio/low_mean": 0.002326829261922588,
900
+ "clip_ratio/low_min": 0.002326829261922588,
901
+ "clip_ratio/region_mean": 0.009233820019289851,
902
+ "completions/clipped_ratio": 0.1388888955116272,
903
+ "completions/max_length": 3000.0,
904
+ "completions/max_terminated_length": 2980.0,
905
+ "completions/mean_length": 1767.5833740234375,
906
+ "completions/mean_terminated_length": 1568.806396484375,
907
+ "completions/min_length": 614.0,
908
+ "completions/min_terminated_length": 614.0,
909
+ "entropy": 0.25413808474938077,
910
+ "epoch": 0.011538461538461539,
911
+ "frac_reward_zero_std": 0.0,
912
+ "grad_norm": 0.06020393222570419,
913
+ "learning_rate": 5e-05,
914
+ "loss": 0.0678,
915
+ "num_tokens": 2126523.0,
916
+ "reward": -0.3114166855812073,
917
+ "reward_std": 0.8995095491409302,
918
+ "rewards/hint_following/mean": 0.2777777910232544,
919
+ "rewards/hint_following/std": 0.7014724016189575,
920
+ "rewards/length_penalty/mean": -0.5891944169998169,
921
+ "rewards/length_penalty/std": 0.24767503142356873,
922
+ "sampling/importance_sampling_ratio/max": 2.9181816577911377,
923
+ "sampling/importance_sampling_ratio/mean": 0.9907821416854858,
924
+ "sampling/importance_sampling_ratio/min": 0.28627336025238037,
925
+ "sampling/sampling_logp_difference/max": 1.2508081197738647,
926
+ "sampling/sampling_logp_difference/mean": 0.025369901210069656,
927
+ "step": 27,
928
+ "step_time": 54.32182147912681
929
+ },
930
+ {
931
+ "clip_ratio/high_max": 0.0,
932
+ "clip_ratio/high_mean": 0.0,
933
+ "clip_ratio/low_mean": 0.0,
934
+ "clip_ratio/low_min": 0.0,
935
+ "clip_ratio/region_mean": 0.0,
936
+ "completions/clipped_ratio": 0.3333333432674408,
937
+ "completions/max_length": 3000.0,
938
+ "completions/max_terminated_length": 1529.0,
939
+ "completions/mean_length": 1610.5833740234375,
940
+ "completions/mean_terminated_length": 915.875,
941
+ "completions/min_length": 478.0,
942
+ "completions/min_terminated_length": 478.0,
943
+ "entropy": 0.19960021724303564,
944
+ "epoch": 0.011965811965811967,
945
+ "frac_reward_zero_std": 1.0,
946
+ "grad_norm": 0.0,
947
+ "learning_rate": 5e-05,
948
+ "loss": 0.0,
949
+ "num_tokens": 2192742.0,
950
+ "reward": -0.8701944351196289,
951
+ "reward_std": 0.8150903582572937,
952
+ "rewards/hint_following/mean": -0.3333333432674408,
953
+ "rewards/hint_following/std": 0.47809144854545593,
954
+ "rewards/length_penalty/mean": -0.5368611216545105,
955
+ "rewards/length_penalty/std": 0.3438311517238617,
956
+ "sampling/importance_sampling_ratio/max": 2.0817480087280273,
957
+ "sampling/importance_sampling_ratio/mean": 0.9929863214492798,
958
+ "sampling/importance_sampling_ratio/min": 0.09024836868047714,
959
+ "sampling/sampling_logp_difference/max": 2.4051897525787354,
960
+ "sampling/sampling_logp_difference/mean": 0.02168331854045391,
961
+ "step": 28,
962
+ "step_time": 56.78273802890908
963
+ },
964
+ {
965
+ "clip_ratio/high_max": 0.001870962364288668,
966
+ "clip_ratio/high_mean": 0.001870962364288668,
967
+ "clip_ratio/low_mean": 0.004373128215471904,
968
+ "clip_ratio/low_min": 0.004373128215471904,
969
+ "clip_ratio/region_mean": 0.006244090540955464,
970
+ "completions/clipped_ratio": 0.1944444477558136,
971
+ "completions/max_length": 3000.0,
972
+ "completions/max_terminated_length": 2978.0,
973
+ "completions/mean_length": 1729.77783203125,
974
+ "completions/mean_terminated_length": 1423.17236328125,
975
+ "completions/min_length": 529.0,
976
+ "completions/min_terminated_length": 529.0,
977
+ "entropy": 0.28223206599553424,
978
+ "epoch": 0.012393162393162393,
979
+ "frac_reward_zero_std": 0.0,
980
+ "grad_norm": 0.07922516763210297,
981
+ "learning_rate": 5e-05,
982
+ "loss": 0.0378,
983
+ "num_tokens": 2262334.0,
984
+ "reward": -0.3265925943851471,
985
+ "reward_std": 0.9465053081512451,
986
+ "rewards/hint_following/mean": 0.25,
987
+ "rewards/hint_following/std": 0.7699722051620483,
988
+ "rewards/length_penalty/mean": -0.5765926241874695,
989
+ "rewards/length_penalty/std": 0.28062164783477783,
990
+ "sampling/importance_sampling_ratio/max": 3.0,
991
+ "sampling/importance_sampling_ratio/mean": 0.9894134402275085,
992
+ "sampling/importance_sampling_ratio/min": 0.29406988620758057,
993
+ "sampling/sampling_logp_difference/max": 1.760861873626709,
994
+ "sampling/sampling_logp_difference/mean": 0.026971129700541496,
995
+ "step": 29,
996
+ "step_time": 55.79818298702594
997
+ },
998
+ {
999
+ "clip_ratio/high_max": 0.001682481961324811,
1000
+ "clip_ratio/high_mean": 0.001682481961324811,
1001
+ "clip_ratio/low_mean": 0.00047758713481016457,
1002
+ "clip_ratio/low_min": 0.00047758713481016457,
1003
+ "clip_ratio/region_mean": 0.002160069086433699,
1004
+ "completions/clipped_ratio": 0.1666666716337204,
1005
+ "completions/max_length": 3000.0,
1006
+ "completions/max_terminated_length": 2747.0,
1007
+ "completions/mean_length": 1684.638916015625,
1008
+ "completions/mean_terminated_length": 1421.5667724609375,
1009
+ "completions/min_length": 805.0,
1010
+ "completions/min_terminated_length": 805.0,
1011
+ "entropy": 0.2508200804392497,
1012
+ "epoch": 0.01282051282051282,
1013
+ "frac_reward_zero_std": 0.0,
1014
+ "grad_norm": 0.06276588886976242,
1015
+ "learning_rate": 5e-05,
1016
+ "loss": 0.0312,
1017
+ "num_tokens": 2331153.0,
1018
+ "reward": 0.02178703434765339,
1019
+ "reward_std": 0.9737120270729065,
1020
+ "rewards/hint_following/mean": 0.5833333134651184,
1021
+ "rewards/hint_following/std": 0.7699722051620483,
1022
+ "rewards/length_penalty/mean": -0.5615463256835938,
1023
+ "rewards/length_penalty/std": 0.233141228556633,
1024
+ "sampling/importance_sampling_ratio/max": 2.9634652137756348,
1025
+ "sampling/importance_sampling_ratio/mean": 0.9905918836593628,
1026
+ "sampling/importance_sampling_ratio/min": 0.2952629625797272,
1027
+ "sampling/sampling_logp_difference/max": 1.2198889255523682,
1028
+ "sampling/sampling_logp_difference/mean": 0.024356598034501076,
1029
+ "step": 30,
1030
+ "step_time": 56.55290222284384
1031
+ },
1032
+ {
1033
+ "clip_ratio/high_max": 0.008109931523601214,
1034
+ "clip_ratio/high_mean": 0.008109931523601214,
1035
+ "clip_ratio/low_mean": 0.0004286583377203594,
1036
+ "clip_ratio/low_min": 0.0004286583377203594,
1037
+ "clip_ratio/region_mean": 0.008538589735204974,
1038
+ "completions/clipped_ratio": 0.0833333358168602,
1039
+ "completions/max_length": 3000.0,
1040
+ "completions/max_terminated_length": 2567.0,
1041
+ "completions/mean_length": 1786.2222900390625,
1042
+ "completions/mean_terminated_length": 1675.8787841796875,
1043
+ "completions/min_length": 468.0,
1044
+ "completions/min_terminated_length": 468.0,
1045
+ "entropy": 0.2908742278814316,
1046
+ "epoch": 0.013247863247863248,
1047
+ "frac_reward_zero_std": 0.0,
1048
+ "grad_norm": 0.07696390151977539,
1049
+ "learning_rate": 5e-05,
1050
+ "loss": 0.0561,
1051
+ "num_tokens": 2403953.0,
1052
+ "reward": -0.0954074040055275,
1053
+ "reward_std": 0.6877580285072327,
1054
+ "rewards/hint_following/mean": 0.5,
1055
+ "rewards/hint_following/std": 0.6546536684036255,
1056
+ "rewards/length_penalty/mean": -0.5954073667526245,
1057
+ "rewards/length_penalty/std": 0.22547654807567596,
1058
+ "sampling/importance_sampling_ratio/max": 2.491806745529175,
1059
+ "sampling/importance_sampling_ratio/mean": 0.9886121153831482,
1060
+ "sampling/importance_sampling_ratio/min": 0.16667605936527252,
1061
+ "sampling/sampling_logp_difference/max": 1.7917031049728394,
1062
+ "sampling/sampling_logp_difference/mean": 0.02771041914820671,
1063
+ "step": 31,
1064
+ "step_time": 58.1784061481012
1065
+ },
1066
+ {
1067
+ "clip_ratio/high_max": 0.0027225150261074305,
1068
+ "clip_ratio/high_mean": 0.0027225150261074305,
1069
+ "clip_ratio/low_mean": 0.003243642975576222,
1070
+ "clip_ratio/low_min": 0.003243642975576222,
1071
+ "clip_ratio/region_mean": 0.00596615804048876,
1072
+ "completions/clipped_ratio": 0.1666666716337204,
1073
+ "completions/max_length": 3000.0,
1074
+ "completions/max_terminated_length": 2917.0,
1075
+ "completions/mean_length": 2085.916748046875,
1076
+ "completions/mean_terminated_length": 1903.10009765625,
1077
+ "completions/min_length": 1014.0,
1078
+ "completions/min_terminated_length": 1014.0,
1079
+ "entropy": 0.278283953666687,
1080
+ "epoch": 0.013675213675213675,
1081
+ "frac_reward_zero_std": 0.0,
1082
+ "grad_norm": 0.06229299306869507,
1083
+ "learning_rate": 5e-05,
1084
+ "loss": 0.0363,
1085
+ "num_tokens": 2486342.0,
1086
+ "reward": -0.36197224259376526,
1087
+ "reward_std": 0.8716595768928528,
1088
+ "rewards/hint_following/mean": 0.3333333432674408,
1089
+ "rewards/hint_following/std": 0.7559289932250977,
1090
+ "rewards/length_penalty/mean": -0.695305585861206,
1091
+ "rewards/length_penalty/std": 0.1993320733308792,
1092
+ "sampling/importance_sampling_ratio/max": 2.49179744720459,
1093
+ "sampling/importance_sampling_ratio/mean": 0.9892889261245728,
1094
+ "sampling/importance_sampling_ratio/min": 0.19078373908996582,
1095
+ "sampling/sampling_logp_difference/max": 1.6566147804260254,
1096
+ "sampling/sampling_logp_difference/mean": 0.02677111327648163,
1097
+ "step": 32,
1098
+ "step_time": 59.783586613833904
1099
+ },
1100
+ {
1101
+ "clip_ratio/high_max": 0.0016072407306637615,
1102
+ "clip_ratio/high_mean": 0.0016072407306637615,
1103
+ "clip_ratio/low_mean": 0.0016927721832568448,
1104
+ "clip_ratio/low_min": 0.0016927721832568448,
1105
+ "clip_ratio/region_mean": 0.003300012855712945,
1106
+ "completions/clipped_ratio": 0.0,
1107
+ "completions/max_length": 2629.0,
1108
+ "completions/max_terminated_length": 2629.0,
1109
+ "completions/mean_length": 1464.5833740234375,
1110
+ "completions/mean_terminated_length": 1464.5833740234375,
1111
+ "completions/min_length": 577.0,
1112
+ "completions/min_terminated_length": 577.0,
1113
+ "entropy": 0.25622066855430603,
1114
+ "epoch": 0.014102564102564103,
1115
+ "frac_reward_zero_std": 0.0,
1116
+ "grad_norm": 0.05397763103246689,
1117
+ "learning_rate": 5e-05,
1118
+ "loss": 0.0059,
1119
+ "num_tokens": 2544023.0,
1120
+ "reward": -0.29375001788139343,
1121
+ "reward_std": 0.45689767599105835,
1122
+ "rewards/hint_following/mean": 0.1944444477558136,
1123
+ "rewards/hint_following/std": 0.4013865292072296,
1124
+ "rewards/length_penalty/mean": -0.48819446563720703,
1125
+ "rewards/length_penalty/std": 0.14601026475429535,
1126
+ "sampling/importance_sampling_ratio/max": 2.081763982772827,
1127
+ "sampling/importance_sampling_ratio/mean": 0.9904037117958069,
1128
+ "sampling/importance_sampling_ratio/min": 0.339209645986557,
1129
+ "sampling/sampling_logp_difference/max": 1.08113694190979,
1130
+ "sampling/sampling_logp_difference/mean": 0.02619028650224209,
1131
+ "step": 33,
1132
+ "step_time": 49.13755757326726
1133
+ },
1134
+ {
1135
+ "clip_ratio/high_max": 0.001959889098846664,
1136
+ "clip_ratio/high_mean": 0.001959889098846664,
1137
+ "clip_ratio/low_mean": 0.0017016992593804996,
1138
+ "clip_ratio/low_min": 0.0017016992593804996,
1139
+ "clip_ratio/region_mean": 0.0036615883776297173,
1140
+ "completions/clipped_ratio": 0.1666666716337204,
1141
+ "completions/max_length": 3000.0,
1142
+ "completions/max_terminated_length": 2814.0,
1143
+ "completions/mean_length": 1581.1666259765625,
1144
+ "completions/mean_terminated_length": 1297.4000244140625,
1145
+ "completions/min_length": 447.0,
1146
+ "completions/min_terminated_length": 447.0,
1147
+ "entropy": 0.25040922313928604,
1148
+ "epoch": 0.01452991452991453,
1149
+ "frac_reward_zero_std": 0.0,
1150
+ "grad_norm": 0.06385543942451477,
1151
+ "learning_rate": 5e-05,
1152
+ "loss": 0.046,
1153
+ "num_tokens": 2606177.0,
1154
+ "reward": -0.0826110988855362,
1155
+ "reward_std": 0.9639437198638916,
1156
+ "rewards/hint_following/mean": 0.4444444477558136,
1157
+ "rewards/hint_following/std": 0.7725447416305542,
1158
+ "rewards/length_penalty/mean": -0.5270555019378662,
1159
+ "rewards/length_penalty/std": 0.2994069457054138,
1160
+ "sampling/importance_sampling_ratio/max": 3.0,
1161
+ "sampling/importance_sampling_ratio/mean": 0.9912077784538269,
1162
+ "sampling/importance_sampling_ratio/min": 0.31699931621551514,
1163
+ "sampling/sampling_logp_difference/max": 1.2634482383728027,
1164
+ "sampling/sampling_logp_difference/mean": 0.02445373497903347,
1165
+ "step": 34,
1166
+ "step_time": 54.58746225386858
1167
+ },
1168
+ {
1169
+ "clip_ratio/high_max": 0.0011685320253794391,
1170
+ "clip_ratio/high_mean": 0.0011685320253794391,
1171
+ "clip_ratio/low_mean": 0.0014686054394890864,
1172
+ "clip_ratio/low_min": 0.0014686054394890864,
1173
+ "clip_ratio/region_mean": 0.0026371375230761864,
1174
+ "completions/clipped_ratio": 0.0,
1175
+ "completions/max_length": 2743.0,
1176
+ "completions/max_terminated_length": 2743.0,
1177
+ "completions/mean_length": 1370.4722900390625,
1178
+ "completions/mean_terminated_length": 1370.4722900390625,
1179
+ "completions/min_length": 650.0,
1180
+ "completions/min_terminated_length": 650.0,
1181
+ "entropy": 0.26418235152959824,
1182
+ "epoch": 0.014957264957264958,
1183
+ "frac_reward_zero_std": 0.0,
1184
+ "grad_norm": 0.057452693581581116,
1185
+ "learning_rate": 5e-05,
1186
+ "loss": 0.01,
1187
+ "num_tokens": 2661580.0,
1188
+ "reward": 0.15428705513477325,
1189
+ "reward_std": 0.5879056453704834,
1190
+ "rewards/hint_following/mean": 0.6111111044883728,
1191
+ "rewards/hint_following/std": 0.49441322684288025,
1192
+ "rewards/length_penalty/mean": -0.45682409405708313,
1193
+ "rewards/length_penalty/std": 0.1867014318704605,
1194
+ "sampling/importance_sampling_ratio/max": 3.0,
1195
+ "sampling/importance_sampling_ratio/mean": 0.9900287985801697,
1196
+ "sampling/importance_sampling_ratio/min": 0.15399949252605438,
1197
+ "sampling/sampling_logp_difference/max": 1.8708059787750244,
1198
+ "sampling/sampling_logp_difference/mean": 0.027402520179748535,
1199
+ "step": 35,
1200
+ "step_time": 47.69401362503413
1201
+ },
1202
+ {
1203
+ "clip_ratio/high_max": 0.0057743463742857175,
1204
+ "clip_ratio/high_mean": 0.0057743463742857175,
1205
+ "clip_ratio/low_mean": 0.0029491251916624606,
1206
+ "clip_ratio/low_min": 0.0029491251916624606,
1207
+ "clip_ratio/region_mean": 0.00872347146893541,
1208
+ "completions/clipped_ratio": 0.1111111119389534,
1209
+ "completions/max_length": 3000.0,
1210
+ "completions/max_terminated_length": 2813.0,
1211
+ "completions/mean_length": 1503.6944580078125,
1212
+ "completions/mean_terminated_length": 1316.65625,
1213
+ "completions/min_length": 456.0,
1214
+ "completions/min_terminated_length": 456.0,
1215
+ "entropy": 0.2601601282755534,
1216
+ "epoch": 0.015384615384615385,
1217
+ "frac_reward_zero_std": 0.0,
1218
+ "grad_norm": 0.06052026152610779,
1219
+ "learning_rate": 5e-05,
1220
+ "loss": 0.0217,
1221
+ "num_tokens": 2721131.0,
1222
+ "reward": -0.44567596912384033,
1223
+ "reward_std": 0.6641168594360352,
1224
+ "rewards/hint_following/mean": 0.0555555559694767,
1225
+ "rewards/hint_following/std": 0.5315446853637695,
1226
+ "rewards/length_penalty/mean": -0.5012314915657043,
1227
+ "rewards/length_penalty/std": 0.2637265622615814,
1228
+ "sampling/importance_sampling_ratio/max": 2.678779125213623,
1229
+ "sampling/importance_sampling_ratio/mean": 0.98993980884552,
1230
+ "sampling/importance_sampling_ratio/min": 0.3602158725261688,
1231
+ "sampling/sampling_logp_difference/max": 1.0210518836975098,
1232
+ "sampling/sampling_logp_difference/mean": 0.02683359384536743,
1233
+ "step": 36,
1234
+ "step_time": 52.9094661710551
1235
+ },
1236
+ {
1237
+ "clip_ratio/high_max": 0.002038249202693502,
1238
+ "clip_ratio/high_mean": 0.002038249202693502,
1239
+ "clip_ratio/low_mean": 0.00521218675809602,
1240
+ "clip_ratio/low_min": 0.00521218675809602,
1241
+ "clip_ratio/region_mean": 0.007250435883179307,
1242
+ "completions/clipped_ratio": 0.0,
1243
+ "completions/max_length": 2391.0,
1244
+ "completions/max_terminated_length": 2391.0,
1245
+ "completions/mean_length": 1279.638916015625,
1246
+ "completions/mean_terminated_length": 1279.638916015625,
1247
+ "completions/min_length": 508.0,
1248
+ "completions/min_terminated_length": 508.0,
1249
+ "entropy": 0.26387400925159454,
1250
+ "epoch": 0.01581196581196581,
1251
+ "frac_reward_zero_std": 0.0,
1252
+ "grad_norm": 0.0660337284207344,
1253
+ "learning_rate": 5e-05,
1254
+ "loss": 0.0072,
1255
+ "num_tokens": 2774056.0,
1256
+ "reward": 0.12900924682617188,
1257
+ "reward_std": 0.5431578159332275,
1258
+ "rewards/hint_following/mean": 0.5555555820465088,
1259
+ "rewards/hint_following/std": 0.5039526224136353,
1260
+ "rewards/length_penalty/mean": -0.4265463054180145,
1261
+ "rewards/length_penalty/std": 0.12839393317699432,
1262
+ "sampling/importance_sampling_ratio/max": 2.7058398723602295,
1263
+ "sampling/importance_sampling_ratio/mean": 0.9903796911239624,
1264
+ "sampling/importance_sampling_ratio/min": 0.1684948354959488,
1265
+ "sampling/sampling_logp_difference/max": 1.7808502912521362,
1266
+ "sampling/sampling_logp_difference/mean": 0.027670899406075478,
1267
+ "step": 37,
1268
+ "step_time": 42.42564612603746
1269
+ },
1270
+ {
1271
+ "clip_ratio/high_max": 0.005249147206389655,
1272
+ "clip_ratio/high_mean": 0.005249147206389655,
1273
+ "clip_ratio/low_mean": 0.0019288610492367297,
1274
+ "clip_ratio/low_min": 0.0019288610492367297,
1275
+ "clip_ratio/region_mean": 0.007178008245925109,
1276
+ "completions/clipped_ratio": 0.1388888955116272,
1277
+ "completions/max_length": 3000.0,
1278
+ "completions/max_terminated_length": 2773.0,
1279
+ "completions/mean_length": 1562.9166259765625,
1280
+ "completions/mean_terminated_length": 1331.1290283203125,
1281
+ "completions/min_length": 508.0,
1282
+ "completions/min_terminated_length": 508.0,
1283
+ "entropy": 0.2739727521936099,
1284
+ "epoch": 0.01623931623931624,
1285
+ "frac_reward_zero_std": 0.0,
1286
+ "grad_norm": 0.05931999906897545,
1287
+ "learning_rate": 5e-05,
1288
+ "loss": 0.0111,
1289
+ "num_tokens": 2836603.0,
1290
+ "reward": 0.034583330154418945,
1291
+ "reward_std": 0.9446364045143127,
1292
+ "rewards/hint_following/mean": 0.5555555820465088,
1293
+ "rewards/hint_following/std": 0.7346308827400208,
1294
+ "rewards/length_penalty/mean": -0.5209722518920898,
1295
+ "rewards/length_penalty/std": 0.2565736174583435,
1296
+ "sampling/importance_sampling_ratio/max": 2.9026551246643066,
1297
+ "sampling/importance_sampling_ratio/mean": 0.9896891117095947,
1298
+ "sampling/importance_sampling_ratio/min": 0.1531003713607788,
1299
+ "sampling/sampling_logp_difference/max": 1.8766615390777588,
1300
+ "sampling/sampling_logp_difference/mean": 0.02661670744419098,
1301
+ "step": 38,
1302
+ "step_time": 54.65668713499326
1303
+ },
1304
+ {
1305
+ "clip_ratio/high_max": 0.0073317300993949175,
1306
+ "clip_ratio/high_mean": 0.0073317300993949175,
1307
+ "clip_ratio/low_mean": 0.0012036147963954136,
1308
+ "clip_ratio/low_min": 0.0012036147963954136,
1309
+ "clip_ratio/region_mean": 0.008535344852134585,
1310
+ "completions/clipped_ratio": 0.0,
1311
+ "completions/max_length": 2373.0,
1312
+ "completions/max_terminated_length": 2373.0,
1313
+ "completions/mean_length": 1587.7222900390625,
1314
+ "completions/mean_terminated_length": 1587.7222900390625,
1315
+ "completions/min_length": 887.0,
1316
+ "completions/min_terminated_length": 887.0,
1317
+ "entropy": 0.2784932255744934,
1318
+ "epoch": 0.016666666666666666,
1319
+ "frac_reward_zero_std": 0.0,
1320
+ "grad_norm": 0.05303090810775757,
1321
+ "learning_rate": 5e-05,
1322
+ "loss": -0.0225,
1323
+ "num_tokens": 2900313.0,
1324
+ "reward": 0.22075927257537842,
1325
+ "reward_std": 0.3945108652114868,
1326
+ "rewards/hint_following/mean": 0.75,
1327
+ "rewards/hint_following/std": 0.43915504217147827,
1328
+ "rewards/length_penalty/mean": -0.5292407274246216,
1329
+ "rewards/length_penalty/std": 0.10378893464803696,
1330
+ "sampling/importance_sampling_ratio/max": 2.234417200088501,
1331
+ "sampling/importance_sampling_ratio/mean": 0.9896702170372009,
1332
+ "sampling/importance_sampling_ratio/min": 0.22993366420269012,
1333
+ "sampling/sampling_logp_difference/max": 1.4699643850326538,
1334
+ "sampling/sampling_logp_difference/mean": 0.027140147984027863,
1335
+ "step": 39,
1336
+ "step_time": 45.66687847697176
1337
+ },
1338
+ {
1339
+ "clip_ratio/high_max": 0.008284316320593158,
1340
+ "clip_ratio/high_mean": 0.008284316320593158,
1341
+ "clip_ratio/low_mean": 0.0006285767303779721,
1342
+ "clip_ratio/low_min": 0.0006285767303779721,
1343
+ "clip_ratio/region_mean": 0.008912893012166023,
1344
+ "completions/clipped_ratio": 0.0,
1345
+ "completions/max_length": 2490.0,
1346
+ "completions/max_terminated_length": 2490.0,
1347
+ "completions/mean_length": 1358.638916015625,
1348
+ "completions/mean_terminated_length": 1358.638916015625,
1349
+ "completions/min_length": 668.0,
1350
+ "completions/min_terminated_length": 668.0,
1351
+ "entropy": 0.2635483493407567,
1352
+ "epoch": 0.017094017094017096,
1353
+ "frac_reward_zero_std": 0.0,
1354
+ "grad_norm": 0.061241693794727325,
1355
+ "learning_rate": 5e-05,
1356
+ "loss": -0.0001,
1357
+ "num_tokens": 2955062.0,
1358
+ "reward": 0.2693426012992859,
1359
+ "reward_std": 0.4795752167701721,
1360
+ "rewards/hint_following/mean": 0.7222222089767456,
1361
+ "rewards/hint_following/std": 0.45425674319267273,
1362
+ "rewards/length_penalty/mean": -0.4528796374797821,
1363
+ "rewards/length_penalty/std": 0.17493696510791779,
1364
+ "sampling/importance_sampling_ratio/max": 2.239511728286743,
1365
+ "sampling/importance_sampling_ratio/mean": 0.9903267621994019,
1366
+ "sampling/importance_sampling_ratio/min": 0.11606565117835999,
1367
+ "sampling/sampling_logp_difference/max": 2.153599262237549,
1368
+ "sampling/sampling_logp_difference/mean": 0.02778884768486023,
1369
+ "step": 40,
1370
+ "step_time": 44.78665240900591
1371
+ },
1372
+ {
1373
+ "clip_ratio/high_max": 0.0024162703969826302,
1374
+ "clip_ratio/high_mean": 0.0024162703969826302,
1375
+ "clip_ratio/low_mean": 0.004380864285243054,
1376
+ "clip_ratio/low_min": 0.004380864285243054,
1377
+ "clip_ratio/region_mean": 0.006797134643420577,
1378
+ "completions/clipped_ratio": 0.02777777798473835,
1379
+ "completions/max_length": 3000.0,
1380
+ "completions/max_terminated_length": 2931.0,
1381
+ "completions/mean_length": 1570.75,
1382
+ "completions/mean_terminated_length": 1529.914306640625,
1383
+ "completions/min_length": 437.0,
1384
+ "completions/min_terminated_length": 437.0,
1385
+ "entropy": 0.2851356367270152,
1386
+ "epoch": 0.01752136752136752,
1387
+ "frac_reward_zero_std": 0.0,
1388
+ "grad_norm": 0.06564202159643173,
1389
+ "learning_rate": 5e-05,
1390
+ "loss": 0.0083,
1391
+ "num_tokens": 3020207.0,
1392
+ "reward": 0.1708611100912094,
1393
+ "reward_std": 0.6461936831474304,
1394
+ "rewards/hint_following/mean": 0.6944444179534912,
1395
+ "rewards/hint_following/std": 0.524782657623291,
1396
+ "rewards/length_penalty/mean": -0.5235832929611206,
1397
+ "rewards/length_penalty/std": 0.23531726002693176,
1398
+ "sampling/importance_sampling_ratio/max": 3.0,
1399
+ "sampling/importance_sampling_ratio/mean": 0.9896803498268127,
1400
+ "sampling/importance_sampling_ratio/min": 0.2426270991563797,
1401
+ "sampling/sampling_logp_difference/max": 1.4162296056747437,
1402
+ "sampling/sampling_logp_difference/mean": 0.02950497530400753,
1403
+ "step": 41,
1404
+ "step_time": 55.908626700984314
1405
+ },
1406
+ {
1407
+ "clip_ratio/high_max": 0.0035200511144163706,
1408
+ "clip_ratio/high_mean": 0.0035200511144163706,
1409
+ "clip_ratio/low_mean": 0.0037633493387450776,
1410
+ "clip_ratio/low_min": 0.0037633493387450776,
1411
+ "clip_ratio/region_mean": 0.007283400433758895,
1412
+ "completions/clipped_ratio": 0.1944444477558136,
1413
+ "completions/max_length": 3000.0,
1414
+ "completions/max_terminated_length": 2916.0,
1415
+ "completions/mean_length": 1708.861083984375,
1416
+ "completions/mean_terminated_length": 1397.2069091796875,
1417
+ "completions/min_length": 601.0,
1418
+ "completions/min_terminated_length": 601.0,
1419
+ "entropy": 0.297603373726209,
1420
+ "epoch": 0.017948717948717947,
1421
+ "frac_reward_zero_std": 0.0,
1422
+ "grad_norm": 0.06374821066856384,
1423
+ "learning_rate": 5e-05,
1424
+ "loss": 0.0258,
1425
+ "num_tokens": 3087486.0,
1426
+ "reward": -0.40295371413230896,
1427
+ "reward_std": 0.9737837910652161,
1428
+ "rewards/hint_following/mean": 0.1666666716337204,
1429
+ "rewards/hint_following/std": 0.7367883920669556,
1430
+ "rewards/length_penalty/mean": -0.5696203708648682,
1431
+ "rewards/length_penalty/std": 0.28024521470069885,
1432
+ "sampling/importance_sampling_ratio/max": 2.632629871368408,
1433
+ "sampling/importance_sampling_ratio/mean": 0.9891336560249329,
1434
+ "sampling/importance_sampling_ratio/min": 0.16689430177211761,
1435
+ "sampling/sampling_logp_difference/max": 1.79039466381073,
1436
+ "sampling/sampling_logp_difference/mean": 0.029687926173210144,
1437
+ "step": 42,
1438
+ "step_time": 55.97696550388355
1439
+ },
1440
+ {
1441
+ "clip_ratio/high_max": 0.005934966184819738,
1442
+ "clip_ratio/high_mean": 0.005934966184819738,
1443
+ "clip_ratio/low_mean": 0.0023504976027955613,
1444
+ "clip_ratio/low_min": 0.0023504976027955613,
1445
+ "clip_ratio/region_mean": 0.008285463865225514,
1446
+ "completions/clipped_ratio": 0.0,
1447
+ "completions/max_length": 2141.0,
1448
+ "completions/max_terminated_length": 2141.0,
1449
+ "completions/mean_length": 1201.638916015625,
1450
+ "completions/mean_terminated_length": 1201.638916015625,
1451
+ "completions/min_length": 517.0,
1452
+ "completions/min_terminated_length": 517.0,
1453
+ "entropy": 0.29462653398513794,
1454
+ "epoch": 0.018376068376068377,
1455
+ "frac_reward_zero_std": 0.0,
1456
+ "grad_norm": 0.05552098527550697,
1457
+ "learning_rate": 5e-05,
1458
+ "loss": 0.0036,
1459
+ "num_tokens": 3135263.0,
1460
+ "reward": 0.12723149359226227,
1461
+ "reward_std": 0.5270769596099854,
1462
+ "rewards/hint_following/mean": 0.5277777910232544,
1463
+ "rewards/hint_following/std": 0.506309449672699,
1464
+ "rewards/length_penalty/mean": -0.40054628252983093,
1465
+ "rewards/length_penalty/std": 0.14752288162708282,
1466
+ "sampling/importance_sampling_ratio/max": 3.0,
1467
+ "sampling/importance_sampling_ratio/mean": 0.9892792701721191,
1468
+ "sampling/importance_sampling_ratio/min": 0.3035065531730652,
1469
+ "sampling/sampling_logp_difference/max": 1.192352056503296,
1470
+ "sampling/sampling_logp_difference/mean": 0.030870094895362854,
1471
+ "step": 43,
1472
+ "step_time": 37.3642987072235
1473
+ },
1474
+ {
1475
+ "clip_ratio/high_max": 0.0028009160111347833,
1476
+ "clip_ratio/high_mean": 0.0028009160111347833,
1477
+ "clip_ratio/low_mean": 0.003960246841112773,
1478
+ "clip_ratio/low_min": 0.003960246841112773,
1479
+ "clip_ratio/region_mean": 0.0067611627746373415,
1480
+ "completions/clipped_ratio": 0.2777777910232544,
1481
+ "completions/max_length": 3000.0,
1482
+ "completions/max_terminated_length": 2767.0,
1483
+ "completions/mean_length": 1834.611083984375,
1484
+ "completions/mean_terminated_length": 1386.3846435546875,
1485
+ "completions/min_length": 439.0,
1486
+ "completions/min_terminated_length": 439.0,
1487
+ "entropy": 0.2577643742163976,
1488
+ "epoch": 0.018803418803418803,
1489
+ "frac_reward_zero_std": 0.0,
1490
+ "grad_norm": 0.06315090507268906,
1491
+ "learning_rate": 5e-05,
1492
+ "loss": 0.0313,
1493
+ "num_tokens": 3208011.0,
1494
+ "reward": -0.22264817357063293,
1495
+ "reward_std": 1.158618688583374,
1496
+ "rewards/hint_following/mean": 0.3888888955116272,
1497
+ "rewards/hint_following/std": 0.903256893157959,
1498
+ "rewards/length_penalty/mean": -0.6115370988845825,
1499
+ "rewards/length_penalty/std": 0.3052543103694916,
1500
+ "sampling/importance_sampling_ratio/max": 2.770780324935913,
1501
+ "sampling/importance_sampling_ratio/mean": 0.9908032417297363,
1502
+ "sampling/importance_sampling_ratio/min": 0.14089906215667725,
1503
+ "sampling/sampling_logp_difference/max": 1.9597115516662598,
1504
+ "sampling/sampling_logp_difference/mean": 0.026477616280317307,
1505
+ "step": 44,
1506
+ "step_time": 57.508703485131264
1507
+ },
1508
+ {
1509
+ "clip_ratio/high_max": 0.0022431810890945294,
1510
+ "clip_ratio/high_mean": 0.0022431810890945294,
1511
+ "clip_ratio/low_mean": 0.0008466135962711027,
1512
+ "clip_ratio/low_min": 0.0008466135962711027,
1513
+ "clip_ratio/region_mean": 0.003089794685365632,
1514
+ "completions/clipped_ratio": 0.0,
1515
+ "completions/max_length": 2567.0,
1516
+ "completions/max_terminated_length": 2567.0,
1517
+ "completions/mean_length": 1403.611083984375,
1518
+ "completions/mean_terminated_length": 1403.611083984375,
1519
+ "completions/min_length": 791.0,
1520
+ "completions/min_terminated_length": 791.0,
1521
+ "entropy": 0.2975164105494817,
1522
+ "epoch": 0.019230769230769232,
1523
+ "frac_reward_zero_std": 0.0,
1524
+ "grad_norm": 0.0667228028178215,
1525
+ "learning_rate": 5e-05,
1526
+ "loss": 0.0231,
1527
+ "num_tokens": 3265633.0,
1528
+ "reward": 0.39324072003364563,
1529
+ "reward_std": 0.3923882842063904,
1530
+ "rewards/hint_following/mean": 0.8611111044883728,
1531
+ "rewards/hint_following/std": 0.35073620080947876,
1532
+ "rewards/length_penalty/mean": -0.4678703546524048,
1533
+ "rewards/length_penalty/std": 0.15880942344665527,
1534
+ "sampling/importance_sampling_ratio/max": 2.800001382827759,
1535
+ "sampling/importance_sampling_ratio/mean": 0.9887976050376892,
1536
+ "sampling/importance_sampling_ratio/min": 0.2656780183315277,
1537
+ "sampling/sampling_logp_difference/max": 1.325470209121704,
1538
+ "sampling/sampling_logp_difference/mean": 0.030811548233032227,
1539
+ "step": 45,
1540
+ "step_time": 45.27160808304325
1541
+ },
1542
+ {
1543
+ "clip_ratio/high_max": 0.007887479305888215,
1544
+ "clip_ratio/high_mean": 0.007887479305888215,
1545
+ "clip_ratio/low_mean": 0.000994703887651364,
1546
+ "clip_ratio/low_min": 0.000994703887651364,
1547
+ "clip_ratio/region_mean": 0.008882183115929365,
1548
+ "completions/clipped_ratio": 0.0,
1549
+ "completions/max_length": 2359.0,
1550
+ "completions/max_terminated_length": 2359.0,
1551
+ "completions/mean_length": 1282.52783203125,
1552
+ "completions/mean_terminated_length": 1282.52783203125,
1553
+ "completions/min_length": 561.0,
1554
+ "completions/min_terminated_length": 561.0,
1555
+ "entropy": 0.2848661094903946,
1556
+ "epoch": 0.019658119658119658,
1557
+ "frac_reward_zero_std": 0.0,
1558
+ "grad_norm": 0.051431454718112946,
1559
+ "learning_rate": 5e-05,
1560
+ "loss": -0.0024,
1561
+ "num_tokens": 3318716.0,
1562
+ "reward": 0.37804630398750305,
1563
+ "reward_std": 0.3656369149684906,
1564
+ "rewards/hint_following/mean": 0.8055555820465088,
1565
+ "rewards/hint_following/std": 0.4013865292072296,
1566
+ "rewards/length_penalty/mean": -0.42750924825668335,
1567
+ "rewards/length_penalty/std": 0.16107404232025146,
1568
+ "sampling/importance_sampling_ratio/max": 3.0,
1569
+ "sampling/importance_sampling_ratio/mean": 0.9896759986877441,
1570
+ "sampling/importance_sampling_ratio/min": 0.09941968321800232,
1571
+ "sampling/sampling_logp_difference/max": 2.3084051609039307,
1572
+ "sampling/sampling_logp_difference/mean": 0.029690390452742577,
1573
+ "step": 46,
1574
+ "step_time": 41.876618986018
1575
+ },
1576
+ {
1577
+ "clip_ratio/high_max": 0.003261159232351929,
1578
+ "clip_ratio/high_mean": 0.003261159232351929,
1579
+ "clip_ratio/low_mean": 0.0030720558910009763,
1580
+ "clip_ratio/low_min": 0.0030720558910009763,
1581
+ "clip_ratio/region_mean": 0.0063332150457426906,
1582
+ "completions/clipped_ratio": 0.0833333358168602,
1583
+ "completions/max_length": 3000.0,
1584
+ "completions/max_terminated_length": 2990.0,
1585
+ "completions/mean_length": 1868.77783203125,
1586
+ "completions/mean_terminated_length": 1765.939453125,
1587
+ "completions/min_length": 947.0,
1588
+ "completions/min_terminated_length": 947.0,
1589
+ "entropy": 0.3000795841217041,
1590
+ "epoch": 0.020085470085470087,
1591
+ "frac_reward_zero_std": 0.0,
1592
+ "grad_norm": 0.07183089107275009,
1593
+ "learning_rate": 5e-05,
1594
+ "loss": 0.0154,
1595
+ "num_tokens": 3395160.0,
1596
+ "reward": -0.12292594462633133,
1597
+ "reward_std": 0.785350501537323,
1598
+ "rewards/hint_following/mean": 0.5,
1599
+ "rewards/hint_following/std": 0.6546536684036255,
1600
+ "rewards/length_penalty/mean": -0.6229259371757507,
1601
+ "rewards/length_penalty/std": 0.2120964676141739,
1602
+ "sampling/importance_sampling_ratio/max": 2.9097461700439453,
1603
+ "sampling/importance_sampling_ratio/mean": 0.988696277141571,
1604
+ "sampling/importance_sampling_ratio/min": 0.10777360945940018,
1605
+ "sampling/sampling_logp_difference/max": 2.227722406387329,
1606
+ "sampling/sampling_logp_difference/mean": 0.030243437737226486,
1607
+ "step": 47,
1608
+ "step_time": 58.67247669992503
1609
+ },
1610
+ {
1611
+ "clip_ratio/high_max": 0.008134961283455292,
1612
+ "clip_ratio/high_mean": 0.008134961283455292,
1613
+ "clip_ratio/low_mean": 0.0004609357565641403,
1614
+ "clip_ratio/low_min": 0.0004609357565641403,
1615
+ "clip_ratio/region_mean": 0.008595897040019432,
1616
+ "completions/clipped_ratio": 0.2222222238779068,
1617
+ "completions/max_length": 3000.0,
1618
+ "completions/max_terminated_length": 2543.0,
1619
+ "completions/mean_length": 1645.8333740234375,
1620
+ "completions/mean_terminated_length": 1258.9285888671875,
1621
+ "completions/min_length": 729.0,
1622
+ "completions/min_terminated_length": 729.0,
1623
+ "entropy": 0.24671275417009988,
1624
+ "epoch": 0.020512820512820513,
1625
+ "frac_reward_zero_std": 0.0,
1626
+ "grad_norm": 0.06476566940546036,
1627
+ "learning_rate": 5e-05,
1628
+ "loss": 0.0513,
1629
+ "num_tokens": 3461970.0,
1630
+ "reward": 0.006944431457668543,
1631
+ "reward_std": 1.0957543849945068,
1632
+ "rewards/hint_following/mean": 0.5555555820465088,
1633
+ "rewards/hint_following/std": 0.8432740569114685,
1634
+ "rewards/length_penalty/mean": -0.5486111044883728,
1635
+ "rewards/length_penalty/std": 0.2772509455680847,
1636
+ "sampling/importance_sampling_ratio/max": 2.0642147064208984,
1637
+ "sampling/importance_sampling_ratio/mean": 0.9907264113426208,
1638
+ "sampling/importance_sampling_ratio/min": 0.04710150137543678,
1639
+ "sampling/sampling_logp_difference/max": 3.055450439453125,
1640
+ "sampling/sampling_logp_difference/mean": 0.025603190064430237,
1641
+ "step": 48,
1642
+ "step_time": 55.27118606586009
1643
+ },
1644
+ {
1645
+ "clip_ratio/high_max": 0.0011131278927981232,
1646
+ "clip_ratio/high_mean": 0.0011131278927981232,
1647
+ "clip_ratio/low_mean": 0.006329516724993785,
1648
+ "clip_ratio/low_min": 0.006329516724993785,
1649
+ "clip_ratio/region_mean": 0.007442644564434886,
1650
+ "completions/clipped_ratio": 0.0,
1651
+ "completions/max_length": 1867.0,
1652
+ "completions/max_terminated_length": 1867.0,
1653
+ "completions/mean_length": 1157.6666259765625,
1654
+ "completions/mean_terminated_length": 1157.6666259765625,
1655
+ "completions/min_length": 722.0,
1656
+ "completions/min_terminated_length": 722.0,
1657
+ "entropy": 0.2628251537680626,
1658
+ "epoch": 0.02094017094017094,
1659
+ "frac_reward_zero_std": 0.0,
1660
+ "grad_norm": 0.06659235060214996,
1661
+ "learning_rate": 5e-05,
1662
+ "loss": 0.0356,
1663
+ "num_tokens": 3510510.0,
1664
+ "reward": 0.5863333344459534,
1665
+ "reward_std": 0.22435788810253143,
1666
+ "rewards/hint_following/mean": 0.9722222089767456,
1667
+ "rewards/hint_following/std": 0.1666666716337204,
1668
+ "rewards/length_penalty/mean": -0.3858889043331146,
1669
+ "rewards/length_penalty/std": 0.10318709164857864,
1670
+ "sampling/importance_sampling_ratio/max": 2.3799936771392822,
1671
+ "sampling/importance_sampling_ratio/mean": 0.9904940128326416,
1672
+ "sampling/importance_sampling_ratio/min": 0.06988711655139923,
1673
+ "sampling/sampling_logp_difference/max": 2.6608738899230957,
1674
+ "sampling/sampling_logp_difference/mean": 0.030541572719812393,
1675
+ "step": 49,
1676
+ "step_time": 33.86789354507346
1677
+ },
1678
+ {
1679
+ "clip_ratio/high_max": 0.0,
1680
+ "clip_ratio/high_mean": 0.0,
1681
+ "clip_ratio/low_mean": 0.0,
1682
+ "clip_ratio/low_min": 0.0,
1683
+ "clip_ratio/region_mean": 0.0,
1684
+ "completions/clipped_ratio": 0.1666666716337204,
1685
+ "completions/max_length": 3000.0,
1686
+ "completions/max_terminated_length": 2940.0,
1687
+ "completions/mean_length": 1683.0,
1688
+ "completions/mean_terminated_length": 1419.60009765625,
1689
+ "completions/min_length": 730.0,
1690
+ "completions/min_terminated_length": 730.0,
1691
+ "entropy": 0.24505582451820374,
1692
+ "epoch": 0.021367521367521368,
1693
+ "frac_reward_zero_std": 1.0,
1694
+ "grad_norm": 0.0,
1695
+ "learning_rate": 5e-05,
1696
+ "loss": 0.0,
1697
+ "num_tokens": 3577332.0,
1698
+ "reward": 0.10566666722297668,
1699
+ "reward_std": 0.9825097322463989,
1700
+ "rewards/hint_following/mean": 0.6666666865348816,
1701
+ "rewards/hint_following/std": 0.7559289932250977,
1702
+ "rewards/length_penalty/mean": -0.5609999895095825,
1703
+ "rewards/length_penalty/std": 0.30474284291267395,
1704
+ "sampling/importance_sampling_ratio/max": 3.0,
1705
+ "sampling/importance_sampling_ratio/mean": 0.9918250441551208,
1706
+ "sampling/importance_sampling_ratio/min": 0.06838197261095047,
1707
+ "sampling/sampling_logp_difference/max": 2.6826460361480713,
1708
+ "sampling/sampling_logp_difference/mean": 0.028820084407925606,
1709
+ "step": 50,
1710
+ "step_time": 54.36680988012813
1711
+ }
1712
+ ],
1713
+ "logging_steps": 1,
1714
+ "max_steps": 250,
1715
+ "num_input_tokens_seen": 3577332,
1716
+ "num_train_epochs": 1,
1717
+ "save_steps": 50,
1718
+ "stateful_callbacks": {
1719
+ "TrainerControl": {
1720
+ "args": {
1721
+ "should_epoch_stop": false,
1722
+ "should_evaluate": false,
1723
+ "should_log": false,
1724
+ "should_save": true,
1725
+ "should_training_stop": false
1726
+ },
1727
+ "attributes": {}
1728
+ }
1729
+ },
1730
+ "total_flos": 0.0,
1731
+ "train_batch_size": 6,
1732
+ "trial_name": null,
1733
+ "trial_params": null
1734
+ }