AutoDecompiler commited on
Commit
7a7ec4d
·
verified ·
1 Parent(s): 76f8a7e

Delete trainer_state.json

Browse files
Files changed (1) hide show
  1. trainer_state.json +0 -546
trainer_state.json DELETED
@@ -1,546 +0,0 @@
1
- {
2
- "best_global_step": null,
3
- "best_metric": null,
4
- "best_model_checkpoint": null,
5
- "epoch": 0.20048019207683074,
6
- "eval_steps": 500,
7
- "global_step": 167,
8
- "is_hyper_param_search": false,
9
- "is_local_process_zero": true,
10
- "is_world_process_zero": true,
11
- "log_history": [
12
- {
13
- "clip_ratio/high_max": 0.0,
14
- "clip_ratio/high_mean": 0.0,
15
- "clip_ratio/low_mean": 0.0,
16
- "clip_ratio/low_min": 0.0,
17
- "clip_ratio/region_mean": 0.0,
18
- "completions/clipped_ratio": 0.01041666679084301,
19
- "completions/max_length": 2827.7,
20
- "completions/max_terminated_length": 2365.2,
21
- "completions/mean_length": 630.6958435058593,
22
- "completions/mean_terminated_length": 594.2115203857422,
23
- "completions/min_length": 132.0,
24
- "completions/min_terminated_length": 132.0,
25
- "entropy": 0.11724737156182527,
26
- "epoch": 0.012004801920768308,
27
- "frac_reward_zero_std": 0.13333333507180214,
28
- "grad_norm": 0.038755763322114944,
29
- "learning_rate": 5.389221556886228e-07,
30
- "loss": 0.0113,
31
- "num_tokens": 799206.0,
32
- "reward": -0.18518302096053957,
33
- "reward_std": 0.20015475898981094,
34
- "rewards/grpo_reward_function/mean": -0.18518302938900888,
35
- "rewards/grpo_reward_function/std": 0.6885311886668205,
36
- "sampling/importance_sampling_ratio/max": 2.1990586280822755,
37
- "sampling/importance_sampling_ratio/mean": 0.4647279143333435,
38
- "sampling/importance_sampling_ratio/min": 0.0005068443759228102,
39
- "sampling/sampling_logp_difference/max": 2.5390082478523253,
40
- "sampling/sampling_logp_difference/mean": 0.013516949955374002,
41
- "step": 10,
42
- "step_time": 569.2043406252749
43
- },
44
- {
45
- "clip_ratio/high_max": 0.0,
46
- "clip_ratio/high_mean": 0.0,
47
- "clip_ratio/low_mean": 0.0,
48
- "clip_ratio/low_min": 0.0,
49
- "clip_ratio/region_mean": 0.0,
50
- "completions/clipped_ratio": 0.010416666977107525,
51
- "completions/max_length": 2742.3,
52
- "completions/max_terminated_length": 2382.7,
53
- "completions/mean_length": 646.3521057128906,
54
- "completions/mean_terminated_length": 610.6208740234375,
55
- "completions/min_length": 131.6,
56
- "completions/min_terminated_length": 131.6,
57
- "entropy": 0.1267015876248479,
58
- "epoch": 0.024009603841536616,
59
- "frac_reward_zero_std": 0.1083333358168602,
60
- "grad_norm": 0.10266362875699997,
61
- "learning_rate": 1.1377245508982037e-06,
62
- "loss": -0.0225,
63
- "num_tokens": 1617099.0,
64
- "reward": 0.01770310625433922,
65
- "reward_std": 0.23654931634664536,
66
- "rewards/grpo_reward_function/mean": 0.0177031047642231,
67
- "rewards/grpo_reward_function/std": 0.8463600814342499,
68
- "sampling/importance_sampling_ratio/max": 1.9834328293800354,
69
- "sampling/importance_sampling_ratio/mean": 0.40317725837230683,
70
- "sampling/importance_sampling_ratio/min": 0.0032052009667828283,
71
- "sampling/sampling_logp_difference/max": 2.1022926926612855,
72
- "sampling/sampling_logp_difference/mean": 0.01352061601355672,
73
- "step": 20,
74
- "step_time": 548.7977518392727
75
- },
76
- {
77
- "clip_ratio/high_max": 0.0,
78
- "clip_ratio/high_mean": 0.0,
79
- "clip_ratio/low_mean": 0.0,
80
- "clip_ratio/low_min": 0.0,
81
- "clip_ratio/region_mean": 0.0,
82
- "completions/clipped_ratio": 0.014583333767950535,
83
- "completions/max_length": 3068.7,
84
- "completions/max_terminated_length": 2151.3,
85
- "completions/mean_length": 682.6271087646485,
86
- "completions/mean_terminated_length": 632.6335662841797,
87
- "completions/min_length": 188.7,
88
- "completions/min_terminated_length": 188.7,
89
- "entropy": 0.125905223749578,
90
- "epoch": 0.03601440576230492,
91
- "frac_reward_zero_std": 0.11666666939854622,
92
- "grad_norm": 0.06394433230161667,
93
- "learning_rate": 1.7365269461077847e-06,
94
- "loss": 0.0229,
95
- "num_tokens": 2465988.0,
96
- "reward": -0.18962360136210918,
97
- "reward_std": 0.19849726594984532,
98
- "rewards/grpo_reward_function/mean": -0.18962358720600606,
99
- "rewards/grpo_reward_function/std": 0.6894359931349754,
100
- "sampling/importance_sampling_ratio/max": 2.460964298248291,
101
- "sampling/importance_sampling_ratio/mean": 0.4253284126520157,
102
- "sampling/importance_sampling_ratio/min": 1.070212653598215e-05,
103
- "sampling/sampling_logp_difference/max": 2.8448187589645384,
104
- "sampling/sampling_logp_difference/mean": 0.013953791093081236,
105
- "step": 30,
106
- "step_time": 554.4699578347615
107
- },
108
- {
109
- "clip_ratio/high_max": 0.0,
110
- "clip_ratio/high_mean": 0.0,
111
- "clip_ratio/low_mean": 0.0,
112
- "clip_ratio/low_min": 0.0,
113
- "clip_ratio/region_mean": 0.0,
114
- "completions/clipped_ratio": 0.01458333358168602,
115
- "completions/max_length": 2789.3,
116
- "completions/max_terminated_length": 1984.1,
117
- "completions/mean_length": 639.958349609375,
118
- "completions/mean_terminated_length": 589.7686553955078,
119
- "completions/min_length": 163.7,
120
- "completions/min_terminated_length": 163.7,
121
- "entropy": 0.11666738856583833,
122
- "epoch": 0.04801920768307323,
123
- "frac_reward_zero_std": 0.1166666679084301,
124
- "grad_norm": 0.08781701326370239,
125
- "learning_rate": 2.3353293413173654e-06,
126
- "loss": -0.0064,
127
- "num_tokens": 3297428.0,
128
- "reward": -0.03914917185902596,
129
- "reward_std": 0.21894535794854164,
130
- "rewards/grpo_reward_function/mean": -0.03914917148649692,
131
- "rewards/grpo_reward_function/std": 0.8605277180671692,
132
- "sampling/importance_sampling_ratio/max": 2.0170334696769716,
133
- "sampling/importance_sampling_ratio/mean": 0.4818507760763168,
134
- "sampling/importance_sampling_ratio/min": 0.0015486635098906688,
135
- "sampling/sampling_logp_difference/max": 2.52269823551178,
136
- "sampling/sampling_logp_difference/mean": 0.012881174683570862,
137
- "step": 40,
138
- "step_time": 541.3490906376392
139
- },
140
- {
141
- "clip_ratio/high_max": 0.0,
142
- "clip_ratio/high_mean": 0.0,
143
- "clip_ratio/low_mean": 4.2163060425082224e-05,
144
- "clip_ratio/low_min": 0.0,
145
- "clip_ratio/region_mean": 4.2163060425082224e-05,
146
- "completions/clipped_ratio": 0.016666667349636555,
147
- "completions/max_length": 2619.1,
148
- "completions/max_terminated_length": 2081.8,
149
- "completions/mean_length": 660.1000122070312,
150
- "completions/mean_terminated_length": 602.3141662597657,
151
- "completions/min_length": 205.5,
152
- "completions/min_terminated_length": 205.5,
153
- "entropy": 0.12803181819617748,
154
- "epoch": 0.060024009603841535,
155
- "frac_reward_zero_std": 0.1416666701436043,
156
- "grad_norm": 0.03966222703456879,
157
- "learning_rate": 2.9341317365269463e-06,
158
- "loss": 0.0112,
159
- "num_tokens": 4129824.0,
160
- "reward": -0.11274411627091467,
161
- "reward_std": 0.2275936236605048,
162
- "rewards/grpo_reward_function/mean": -0.1127441140357405,
163
- "rewards/grpo_reward_function/std": 0.8841595828533173,
164
- "sampling/importance_sampling_ratio/max": 2.008124852180481,
165
- "sampling/importance_sampling_ratio/mean": 0.46366433799266815,
166
- "sampling/importance_sampling_ratio/min": 0.0006394427657710367,
167
- "sampling/sampling_logp_difference/max": 2.58479106426239,
168
- "sampling/sampling_logp_difference/mean": 0.013615725003182888,
169
- "step": 50,
170
- "step_time": 545.8774313618429
171
- },
172
- {
173
- "clip_ratio/high_max": 4.673766961786896e-05,
174
- "clip_ratio/high_mean": 7.7896114817122e-06,
175
- "clip_ratio/low_mean": 0.0,
176
- "clip_ratio/low_min": 0.0,
177
- "clip_ratio/region_mean": 7.7896114817122e-06,
178
- "completions/clipped_ratio": 0.01250000037252903,
179
- "completions/max_length": 2304.9,
180
- "completions/max_terminated_length": 1602.8,
181
- "completions/mean_length": 596.5750213623047,
182
- "completions/mean_terminated_length": 552.8821624755859,
183
- "completions/min_length": 150.0,
184
- "completions/min_terminated_length": 150.0,
185
- "entropy": 0.12017892487347126,
186
- "epoch": 0.07202881152460984,
187
- "frac_reward_zero_std": 0.14166666865348815,
188
- "grad_norm": 0.09313877671957016,
189
- "learning_rate": 3.5329341317365273e-06,
190
- "loss": -0.0307,
191
- "num_tokens": 4936176.0,
192
- "reward": -0.03057028874754906,
193
- "reward_std": 0.2686158835887909,
194
- "rewards/grpo_reward_function/mean": -0.03057028613984585,
195
- "rewards/grpo_reward_function/std": 0.8661522060632706,
196
- "sampling/importance_sampling_ratio/max": 2.2043559432029722,
197
- "sampling/importance_sampling_ratio/mean": 0.4847503274679184,
198
- "sampling/importance_sampling_ratio/min": 7.13271651690217e-05,
199
- "sampling/sampling_logp_difference/max": 2.4851160287857055,
200
- "sampling/sampling_logp_difference/mean": 0.013510057888925075,
201
- "step": 60,
202
- "step_time": 546.4311281181872
203
- },
204
- {
205
- "clip_ratio/high_max": 0.0,
206
- "clip_ratio/high_mean": 0.0,
207
- "clip_ratio/low_mean": 6.860105058876797e-05,
208
- "clip_ratio/low_min": 0.0,
209
- "clip_ratio/region_mean": 6.860105058876797e-05,
210
- "completions/clipped_ratio": 0.018750000558793545,
211
- "completions/max_length": 3205.2,
212
- "completions/max_terminated_length": 2270.1,
213
- "completions/mean_length": 702.3604431152344,
214
- "completions/mean_terminated_length": 636.497216796875,
215
- "completions/min_length": 131.8,
216
- "completions/min_terminated_length": 131.8,
217
- "entropy": 0.11768119670450687,
218
- "epoch": 0.08403361344537816,
219
- "frac_reward_zero_std": 0.1083333358168602,
220
- "grad_norm": 0.0433771014213562,
221
- "learning_rate": 4.131736526946108e-06,
222
- "loss": 0.0553,
223
- "num_tokens": 5841149.0,
224
- "reward": -0.0784481130540371,
225
- "reward_std": 0.23132488708943127,
226
- "rewards/grpo_reward_function/mean": -0.07844811640679836,
227
- "rewards/grpo_reward_function/std": 0.8492624998092652,
228
- "sampling/importance_sampling_ratio/max": 2.233333742618561,
229
- "sampling/importance_sampling_ratio/mean": 0.4808308959007263,
230
- "sampling/importance_sampling_ratio/min": 0.0008302704439188347,
231
- "sampling/sampling_logp_difference/max": 2.9827078700065615,
232
- "sampling/sampling_logp_difference/mean": 0.012630783580243587,
233
- "step": 70,
234
- "step_time": 561.5568902881816
235
- },
236
- {
237
- "clip_ratio/high_max": 0.0,
238
- "clip_ratio/high_mean": 0.0,
239
- "clip_ratio/low_mean": 8.41788569232449e-05,
240
- "clip_ratio/low_min": 0.0,
241
- "clip_ratio/region_mean": 8.41788569232449e-05,
242
- "completions/clipped_ratio": 0.01250000037252903,
243
- "completions/max_length": 2541.9,
244
- "completions/max_terminated_length": 2033.2,
245
- "completions/mean_length": 612.5875213623046,
246
- "completions/mean_terminated_length": 569.556509399414,
247
- "completions/min_length": 149.1,
248
- "completions/min_terminated_length": 149.1,
249
- "entropy": 0.13153507560491562,
250
- "epoch": 0.09603841536614646,
251
- "frac_reward_zero_std": 0.10833333656191826,
252
- "grad_norm": 0.08665835857391357,
253
- "learning_rate": 4.730538922155689e-06,
254
- "loss": 0.0701,
255
- "num_tokens": 6606395.0,
256
- "reward": -0.011540251970291137,
257
- "reward_std": 0.19073452726006507,
258
- "rewards/grpo_reward_function/mean": -0.011540257930755615,
259
- "rewards/grpo_reward_function/std": 0.784630474448204,
260
- "sampling/importance_sampling_ratio/max": 2.1984647274017335,
261
- "sampling/importance_sampling_ratio/mean": 0.5050391256809235,
262
- "sampling/importance_sampling_ratio/min": 0.00014755414913452113,
263
- "sampling/sampling_logp_difference/max": 1.8997669577598573,
264
- "sampling/sampling_logp_difference/mean": 0.013426258694380522,
265
- "step": 80,
266
- "step_time": 551.2647462010384
267
- },
268
- {
269
- "clip_ratio/high_max": 2.2563176753465086e-05,
270
- "clip_ratio/high_mean": 3.760529580176808e-06,
271
- "clip_ratio/low_mean": 1.3224284339230508e-05,
272
- "clip_ratio/low_min": 0.0,
273
- "clip_ratio/region_mean": 1.6984813919407314e-05,
274
- "completions/clipped_ratio": 0.01041666679084301,
275
- "completions/max_length": 2347.6,
276
- "completions/max_terminated_length": 2078.4,
277
- "completions/mean_length": 621.4354370117187,
278
- "completions/mean_terminated_length": 586.118701171875,
279
- "completions/min_length": 163.2,
280
- "completions/min_terminated_length": 163.2,
281
- "entropy": 0.1206895818002522,
282
- "epoch": 0.10804321728691477,
283
- "frac_reward_zero_std": 0.1083333358168602,
284
- "grad_norm": 0.034579165279865265,
285
- "learning_rate": 5.32934131736527e-06,
286
- "loss": 0.0011,
287
- "num_tokens": 7424828.0,
288
- "reward": 0.02708094713743776,
289
- "reward_std": 0.23181376457214356,
290
- "rewards/grpo_reward_function/mean": 0.027080959058366716,
291
- "rewards/grpo_reward_function/std": 0.8183064997196198,
292
- "sampling/importance_sampling_ratio/max": 2.499999237060547,
293
- "sampling/importance_sampling_ratio/mean": 0.486982923746109,
294
- "sampling/importance_sampling_ratio/min": 0.000999147113179788,
295
- "sampling/sampling_logp_difference/max": 2.079863798618317,
296
- "sampling/sampling_logp_difference/mean": 0.012986462097615004,
297
- "step": 90,
298
- "step_time": 550.7672496054322
299
- },
300
- {
301
- "clip_ratio/high_max": 0.00031043787457747386,
302
- "clip_ratio/high_mean": 5.173964618734317e-05,
303
- "clip_ratio/low_mean": 0.0,
304
- "clip_ratio/low_min": 0.0,
305
- "clip_ratio/region_mean": 5.173964618734317e-05,
306
- "completions/clipped_ratio": 0.01666666716337204,
307
- "completions/max_length": 2765.9,
308
- "completions/max_terminated_length": 1842.8,
309
- "completions/mean_length": 675.4458435058593,
310
- "completions/mean_terminated_length": 618.1135345458985,
311
- "completions/min_length": 178.2,
312
- "completions/min_terminated_length": 178.2,
313
- "entropy": 0.149181258212775,
314
- "epoch": 0.12004801920768307,
315
- "frac_reward_zero_std": 0.1416666701436043,
316
- "grad_norm": 0.13291294872760773,
317
- "learning_rate": 5.928143712574851e-06,
318
- "loss": 0.0212,
319
- "num_tokens": 8278282.0,
320
- "reward": 0.0703774506226182,
321
- "reward_std": 0.2336222641170025,
322
- "rewards/grpo_reward_function/mean": 0.07037745183333755,
323
- "rewards/grpo_reward_function/std": 0.8314530551433563,
324
- "sampling/importance_sampling_ratio/max": 2.2870466232299806,
325
- "sampling/importance_sampling_ratio/mean": 0.4643064886331558,
326
- "sampling/importance_sampling_ratio/min": 2.921815394074656e-05,
327
- "sampling/sampling_logp_difference/max": 1.9242668151855469,
328
- "sampling/sampling_logp_difference/mean": 0.014198462665081023,
329
- "step": 100,
330
- "step_time": 547.0480061549694
331
- },
332
- {
333
- "clip_ratio/high_max": 0.0003238706885895226,
334
- "clip_ratio/high_mean": 5.397844997787615e-05,
335
- "clip_ratio/low_mean": 7.069677012623287e-05,
336
- "clip_ratio/low_min": 0.0,
337
- "clip_ratio/region_mean": 0.00012467522010410902,
338
- "completions/clipped_ratio": 0.020833334513008596,
339
- "completions/max_length": 2925.3,
340
- "completions/max_terminated_length": 2297.9,
341
- "completions/mean_length": 680.045849609375,
342
- "completions/mean_terminated_length": 605.8596954345703,
343
- "completions/min_length": 186.9,
344
- "completions/min_terminated_length": 186.9,
345
- "entropy": 0.1511568833142519,
346
- "epoch": 0.13205282112845138,
347
- "frac_reward_zero_std": 0.08333333507180214,
348
- "grad_norm": 0.0421764962375164,
349
- "learning_rate": 6.526946107784432e-06,
350
- "loss": -0.0031,
351
- "num_tokens": 9165800.0,
352
- "reward": 0.04289367534220219,
353
- "reward_std": 0.24053554534912108,
354
- "rewards/grpo_reward_function/mean": 0.042893677949905396,
355
- "rewards/grpo_reward_function/std": 0.835248938202858,
356
- "sampling/importance_sampling_ratio/max": 2.332168984413147,
357
- "sampling/importance_sampling_ratio/mean": 0.4403663039207458,
358
- "sampling/importance_sampling_ratio/min": 0.0001706225667930994,
359
- "sampling/sampling_logp_difference/max": 2.4483426809310913,
360
- "sampling/sampling_logp_difference/mean": 0.014532316662371158,
361
- "step": 110,
362
- "step_time": 548.8019280240871
363
- },
364
- {
365
- "clip_ratio/high_max": 0.00012998266611248256,
366
- "clip_ratio/high_mean": 2.16637781704776e-05,
367
- "clip_ratio/low_mean": 0.0,
368
- "clip_ratio/low_min": 0.0,
369
- "clip_ratio/region_mean": 2.16637781704776e-05,
370
- "completions/clipped_ratio": 0.002083333395421505,
371
- "completions/max_length": 1812.6,
372
- "completions/max_terminated_length": 1592.9,
373
- "completions/mean_length": 602.1333526611328,
374
- "completions/mean_terminated_length": 594.61220703125,
375
- "completions/min_length": 179.7,
376
- "completions/min_terminated_length": 179.7,
377
- "entropy": 0.16710406728088856,
378
- "epoch": 0.14405762304921968,
379
- "frac_reward_zero_std": 0.0833333358168602,
380
- "grad_norm": 0.07664494961500168,
381
- "learning_rate": 7.125748502994012e-06,
382
- "loss": -0.0309,
383
- "num_tokens": 9975204.0,
384
- "reward": 0.0826782912015915,
385
- "reward_std": 0.23934805542230606,
386
- "rewards/grpo_reward_function/mean": 0.0826782874763012,
387
- "rewards/grpo_reward_function/std": 0.8862796187400818,
388
- "sampling/importance_sampling_ratio/max": 2.2503564238548277,
389
- "sampling/importance_sampling_ratio/mean": 0.4635925680398941,
390
- "sampling/importance_sampling_ratio/min": 0.0009416027547558823,
391
- "sampling/sampling_logp_difference/max": 2.073215699195862,
392
- "sampling/sampling_logp_difference/mean": 0.01480921907350421,
393
- "step": 120,
394
- "step_time": 539.0031213279814
395
- },
396
- {
397
- "clip_ratio/high_max": 0.0,
398
- "clip_ratio/high_mean": 0.0,
399
- "clip_ratio/low_mean": 1.9831826648442074e-06,
400
- "clip_ratio/low_min": 0.0,
401
- "clip_ratio/region_mean": 1.9831826648442074e-06,
402
- "completions/clipped_ratio": 0.006250000186264515,
403
- "completions/max_length": 2290.7,
404
- "completions/max_terminated_length": 1880.6,
405
- "completions/mean_length": 599.5916900634766,
406
- "completions/mean_terminated_length": 578.0104217529297,
407
- "completions/min_length": 168.6,
408
- "completions/min_terminated_length": 168.6,
409
- "entropy": 0.16012020353227854,
410
- "epoch": 0.15606242496998798,
411
- "frac_reward_zero_std": 0.11666667014360428,
412
- "grad_norm": 0.05220530927181244,
413
- "learning_rate": 7.724550898203594e-06,
414
- "loss": -0.0377,
415
- "num_tokens": 10768324.0,
416
- "reward": -0.0507307555526495,
417
- "reward_std": 0.18351687043905257,
418
- "rewards/grpo_reward_function/mean": -0.05073075201362372,
419
- "rewards/grpo_reward_function/std": 0.7544578343629837,
420
- "sampling/importance_sampling_ratio/max": 2.375898337364197,
421
- "sampling/importance_sampling_ratio/mean": 0.524286350607872,
422
- "sampling/importance_sampling_ratio/min": 0.00018855740054277704,
423
- "sampling/sampling_logp_difference/max": 2.009746181964874,
424
- "sampling/sampling_logp_difference/mean": 0.01376222250983119,
425
- "step": 130,
426
- "step_time": 548.6409472068772
427
- },
428
- {
429
- "clip_ratio/high_max": 0.00017211703816428782,
430
- "clip_ratio/high_mean": 2.86861730273813e-05,
431
- "clip_ratio/low_mean": 0.0,
432
- "clip_ratio/low_min": 0.0,
433
- "clip_ratio/region_mean": 2.86861730273813e-05,
434
- "completions/clipped_ratio": 0.00833333358168602,
435
- "completions/max_length": 3136.9,
436
- "completions/max_terminated_length": 2221.0,
437
- "completions/mean_length": 684.8500244140625,
438
- "completions/mean_terminated_length": 655.5515625,
439
- "completions/min_length": 168.7,
440
- "completions/min_terminated_length": 168.7,
441
- "entropy": 0.12020768839865922,
442
- "epoch": 0.16806722689075632,
443
- "frac_reward_zero_std": 0.09166666939854622,
444
- "grad_norm": 0.058197326958179474,
445
- "learning_rate": 8.323353293413174e-06,
446
- "loss": -0.0342,
447
- "num_tokens": 11642436.0,
448
- "reward": 0.04102597634773701,
449
- "reward_std": 0.2864942252635956,
450
- "rewards/grpo_reward_function/mean": 0.041025977826211604,
451
- "rewards/grpo_reward_function/std": 0.8844284832477569,
452
- "sampling/importance_sampling_ratio/max": 2.37525737285614,
453
- "sampling/importance_sampling_ratio/mean": 0.46323378682136535,
454
- "sampling/importance_sampling_ratio/min": 2.6157076149502247e-08,
455
- "sampling/sampling_logp_difference/max": 2.5657184720039368,
456
- "sampling/sampling_logp_difference/mean": 0.012760929018259048,
457
- "step": 140,
458
- "step_time": 550.6772611703724
459
- },
460
- {
461
- "clip_ratio/high_max": 0.00029233113455120476,
462
- "clip_ratio/high_mean": 4.872185563726816e-05,
463
- "clip_ratio/low_mean": 4.4254150270717216e-05,
464
- "clip_ratio/low_min": 0.0,
465
- "clip_ratio/region_mean": 9.297600590798538e-05,
466
- "completions/clipped_ratio": 0.01041666679084301,
467
- "completions/max_length": 2211.6,
468
- "completions/max_terminated_length": 1961.7,
469
- "completions/mean_length": 601.2291839599609,
470
- "completions/mean_terminated_length": 565.5705291748047,
471
- "completions/min_length": 143.6,
472
- "completions/min_terminated_length": 143.6,
473
- "entropy": 0.10819828314706684,
474
- "epoch": 0.18007202881152462,
475
- "frac_reward_zero_std": 0.10833333656191826,
476
- "grad_norm": 0.04749957472085953,
477
- "learning_rate": 8.922155688622756e-06,
478
- "loss": -0.0236,
479
- "num_tokens": 12486318.0,
480
- "reward": 0.03778684511780739,
481
- "reward_std": 0.25178585574030876,
482
- "rewards/grpo_reward_function/mean": 0.03778683394193649,
483
- "rewards/grpo_reward_function/std": 0.7447861909866333,
484
- "sampling/importance_sampling_ratio/max": 2.481464517116547,
485
- "sampling/importance_sampling_ratio/mean": 0.5163449585437775,
486
- "sampling/importance_sampling_ratio/min": 3.668112331070006e-05,
487
- "sampling/sampling_logp_difference/max": 2.379316544532776,
488
- "sampling/sampling_logp_difference/mean": 0.012185737490653992,
489
- "step": 150,
490
- "step_time": 551.4488250606694
491
- },
492
- {
493
- "clip_ratio/high_max": 0.0,
494
- "clip_ratio/high_mean": 0.0,
495
- "clip_ratio/low_mean": 5.082125426270068e-06,
496
- "clip_ratio/low_min": 0.0,
497
- "clip_ratio/region_mean": 5.082125426270068e-06,
498
- "completions/clipped_ratio": 0.002083333395421505,
499
- "completions/max_length": 2301.5,
500
- "completions/max_terminated_length": 2174.7,
501
- "completions/mean_length": 625.8979309082031,
502
- "completions/mean_terminated_length": 618.7978820800781,
503
- "completions/min_length": 170.2,
504
- "completions/min_terminated_length": 170.2,
505
- "entropy": 0.10213978644460439,
506
- "epoch": 0.19207683073229292,
507
- "frac_reward_zero_std": 0.08333333507180214,
508
- "grad_norm": 0.05615560710430145,
509
- "learning_rate": 9.520958083832336e-06,
510
- "loss": 0.0043,
511
- "num_tokens": 13325121.0,
512
- "reward": 0.06630225274711847,
513
- "reward_std": 0.18489644899964333,
514
- "rewards/grpo_reward_function/mean": 0.0663022572407499,
515
- "rewards/grpo_reward_function/std": 0.7666326016187668,
516
- "sampling/importance_sampling_ratio/max": 2.1301008343696592,
517
- "sampling/importance_sampling_ratio/mean": 0.44916791915893556,
518
- "sampling/importance_sampling_ratio/min": 7.07070047610614e-05,
519
- "sampling/sampling_logp_difference/max": 2.524372959136963,
520
- "sampling/sampling_logp_difference/mean": 0.013301923777908087,
521
- "step": 160,
522
- "step_time": 538.0738848904148
523
- }
524
- ],
525
- "logging_steps": 10,
526
- "max_steps": 833,
527
- "num_input_tokens_seen": 13935881,
528
- "num_train_epochs": 1,
529
- "save_steps": 167,
530
- "stateful_callbacks": {
531
- "TrainerControl": {
532
- "args": {
533
- "should_epoch_stop": false,
534
- "should_evaluate": false,
535
- "should_log": false,
536
- "should_save": true,
537
- "should_training_stop": false
538
- },
539
- "attributes": {}
540
- }
541
- },
542
- "total_flos": 0.0,
543
- "train_batch_size": 2,
544
- "trial_name": null,
545
- "trial_params": null
546
- }