File size: 32,892 Bytes
d0c2ac6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
{
  "best_global_step": null,
  "best_metric": null,
  "best_model_checkpoint": null,
  "epoch": 2.0,
  "eval_steps": 500,
  "global_step": 1492,
  "is_hyper_param_search": false,
  "is_local_process_zero": true,
  "is_world_process_zero": true,
  "log_history": [
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.05,
      "completions/max_length": 246.3,
      "completions/max_terminated_length": 240.26,
      "completions/mean_length": 195.0275,
      "completions/mean_terminated_length": 191.87969146728517,
      "completions/min_length": 147.76,
      "completions/min_terminated_length": 147.76,
      "entropy": 0.06807037293910981,
      "epoch": 0.06702412868632708,
      "frac_reward_zero_std": 0.4475,
      "grad_norm": 0.1978774070739746,
      "learning_rate": 1e-05,
      "loss": -0.0022,
      "num_tokens": 6268258.0,
      "reward": 12.489985446929932,
      "reward_std": 1.05244723290205,
      "rewards/event_reward_fn/mean": 11.62375,
      "rewards/event_reward_fn/std": 7.598931360244751,
      "rewards/format_reward_fn/mean": 0.8662354218959808,
      "rewards/format_reward_fn/std": 0.24084076710045338,
      "step": 50,
      "step_time": 24.881226640827954
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.043125,
      "completions/max_length": 249.38,
      "completions/max_terminated_length": 244.26,
      "completions/mean_length": 198.4925,
      "completions/mean_terminated_length": 195.8674203491211,
      "completions/min_length": 155.1,
      "completions/min_terminated_length": 155.1,
      "entropy": 0.07096008479595184,
      "epoch": 0.13404825737265416,
      "frac_reward_zero_std": 0.42,
      "grad_norm": 0.31616032123565674,
      "learning_rate": 1e-05,
      "loss": -0.0052,
      "num_tokens": 12603730.0,
      "reward": 11.722552404403686,
      "reward_std": 1.104598103761673,
      "rewards/event_reward_fn/mean": 10.865,
      "rewards/event_reward_fn/std": 7.203483366966248,
      "rewards/format_reward_fn/mean": 0.8575523483753205,
      "rewards/format_reward_fn/std": 0.25920433282852173,
      "step": 100,
      "step_time": 23.881343694739044
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.069375,
      "completions/max_length": 251.04,
      "completions/max_terminated_length": 245.08,
      "completions/mean_length": 201.58125,
      "completions/mean_terminated_length": 197.60445678710937,
      "completions/min_length": 157.96,
      "completions/min_terminated_length": 157.96,
      "entropy": 0.07228697955608368,
      "epoch": 0.20107238605898123,
      "frac_reward_zero_std": 0.41,
      "grad_norm": 0.1767224669456482,
      "learning_rate": 1e-05,
      "loss": 0.002,
      "num_tokens": 19236102.0,
      "reward": 11.989666719436645,
      "reward_std": 1.2850025883316993,
      "rewards/event_reward_fn/mean": 11.1225,
      "rewards/event_reward_fn/std": 7.3152674865722656,
      "rewards/format_reward_fn/mean": 0.8671666479110718,
      "rewards/format_reward_fn/std": 0.24983404949307442,
      "step": 150,
      "step_time": 27.783113366477192
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.068125,
      "completions/max_length": 250.62,
      "completions/max_terminated_length": 244.54,
      "completions/mean_length": 201.1025,
      "completions/mean_terminated_length": 197.25198516845703,
      "completions/min_length": 156.4,
      "completions/min_terminated_length": 156.4,
      "entropy": 0.06773373357951641,
      "epoch": 0.2680965147453083,
      "frac_reward_zero_std": 0.415,
      "grad_norm": 0.13261352479457855,
      "learning_rate": 1e-05,
      "loss": -0.0029,
      "num_tokens": 25426958.0,
      "reward": 12.467143926620484,
      "reward_std": 1.1554639112949372,
      "rewards/event_reward_fn/mean": 11.59875,
      "rewards/event_reward_fn/std": 7.149877543449402,
      "rewards/format_reward_fn/mean": 0.8683938610553742,
      "rewards/format_reward_fn/std": 0.24253679752349855,
      "step": 200,
      "step_time": 24.421198091395198
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.058125,
      "completions/max_length": 250.22,
      "completions/max_terminated_length": 243.8,
      "completions/mean_length": 200.303125,
      "completions/mean_terminated_length": 196.90248321533204,
      "completions/min_length": 162.42,
      "completions/min_terminated_length": 162.42,
      "entropy": 0.06486415289342404,
      "epoch": 0.3351206434316354,
      "frac_reward_zero_std": 0.385,
      "grad_norm": 0.49442073702812195,
      "learning_rate": 1e-05,
      "loss": -0.0036,
      "num_tokens": 31582342.0,
      "reward": 12.355808296203612,
      "reward_std": 1.1142808997631073,
      "rewards/event_reward_fn/mean": 11.48875,
      "rewards/event_reward_fn/std": 7.448825697898865,
      "rewards/format_reward_fn/mean": 0.8670582604408265,
      "rewards/format_reward_fn/std": 0.24978963822126388,
      "step": 250,
      "step_time": 25.453000083304943
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.04875,
      "completions/max_length": 248.68,
      "completions/max_terminated_length": 244.22,
      "completions/mean_length": 198.759375,
      "completions/mean_terminated_length": 196.16592681884765,
      "completions/min_length": 156.2,
      "completions/min_terminated_length": 156.2,
      "entropy": 0.0681518343836069,
      "epoch": 0.40214477211796246,
      "frac_reward_zero_std": 0.39,
      "grad_norm": 0.48775437474250793,
      "learning_rate": 1e-05,
      "loss": -0.0057,
      "num_tokens": 37800719.0,
      "reward": 12.434584522247315,
      "reward_std": 1.183589797616005,
      "rewards/event_reward_fn/mean": 11.56375,
      "rewards/event_reward_fn/std": 7.52141658782959,
      "rewards/format_reward_fn/mean": 0.8708344352245331,
      "rewards/format_reward_fn/std": 0.23306368254125118,
      "step": 300,
      "step_time": 25.360634116120636
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.034375,
      "completions/max_length": 248.34,
      "completions/max_terminated_length": 245.28,
      "completions/mean_length": 203.264375,
      "completions/mean_terminated_length": 201.32774475097656,
      "completions/min_length": 157.54,
      "completions/min_terminated_length": 157.54,
      "entropy": 0.06739457175135613,
      "epoch": 0.4691689008042895,
      "frac_reward_zero_std": 0.3525,
      "grad_norm": 0.33356958627700806,
      "learning_rate": 1e-05,
      "loss": -0.004,
      "num_tokens": 44150011.0,
      "reward": 13.173797435760498,
      "reward_std": 1.2946509444713592,
      "rewards/event_reward_fn/mean": 12.28875,
      "rewards/event_reward_fn/std": 7.145490102767944,
      "rewards/format_reward_fn/mean": 0.885047378540039,
      "rewards/format_reward_fn/std": 0.22108205765485764,
      "step": 350,
      "step_time": 26.940150288008155
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.064375,
      "completions/max_length": 252.9,
      "completions/max_terminated_length": 247.8,
      "completions/mean_length": 203.064375,
      "completions/mean_terminated_length": 199.5350747680664,
      "completions/min_length": 158.26,
      "completions/min_terminated_length": 158.26,
      "entropy": 0.0657703248411417,
      "epoch": 0.5361930294906166,
      "frac_reward_zero_std": 0.435,
      "grad_norm": 0.26359474658966064,
      "learning_rate": 1e-05,
      "loss": -0.0021,
      "num_tokens": 50384400.0,
      "reward": 12.238037357330322,
      "reward_std": 1.057584773004055,
      "rewards/event_reward_fn/mean": 11.37,
      "rewards/event_reward_fn/std": 7.154304637908935,
      "rewards/format_reward_fn/mean": 0.8680373668670655,
      "rewards/format_reward_fn/std": 0.26109003871679304,
      "step": 400,
      "step_time": 25.59800311360508
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.049375,
      "completions/max_length": 249.06,
      "completions/max_terminated_length": 244.9,
      "completions/mean_length": 203.706875,
      "completions/mean_terminated_length": 200.99220581054686,
      "completions/min_length": 161.06,
      "completions/min_terminated_length": 161.06,
      "entropy": 0.06626586891710758,
      "epoch": 0.6032171581769437,
      "frac_reward_zero_std": 0.3775,
      "grad_norm": 0.48660293221473694,
      "learning_rate": 1e-05,
      "loss": -0.004,
      "num_tokens": 56771056.0,
      "reward": 13.009743461608887,
      "reward_std": 1.2429037857055665,
      "rewards/event_reward_fn/mean": 12.130625,
      "rewards/event_reward_fn/std": 7.234463820457458,
      "rewards/format_reward_fn/mean": 0.8791184043884277,
      "rewards/format_reward_fn/std": 0.23800445690751076,
      "step": 450,
      "step_time": 25.550446799769997
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.066875,
      "completions/max_length": 251.92,
      "completions/max_terminated_length": 246.12,
      "completions/mean_length": 204.07625,
      "completions/mean_terminated_length": 200.35590240478516,
      "completions/min_length": 160.74,
      "completions/min_terminated_length": 160.74,
      "entropy": 0.06663089752197265,
      "epoch": 0.6702412868632708,
      "frac_reward_zero_std": 0.4025,
      "grad_norm": 0.6319305300712585,
      "learning_rate": 1e-05,
      "loss": -0.0042,
      "num_tokens": 63078757.0,
      "reward": 12.313038005828858,
      "reward_std": 1.1368902394175529,
      "rewards/event_reward_fn/mean": 11.4575,
      "rewards/event_reward_fn/std": 6.7143393945693965,
      "rewards/format_reward_fn/mean": 0.8555380630493165,
      "rewards/format_reward_fn/std": 0.2657873314619064,
      "step": 500,
      "step_time": 26.24973841637373
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.091875,
      "completions/max_length": 252.82,
      "completions/max_terminated_length": 246.88,
      "completions/mean_length": 203.815,
      "completions/mean_terminated_length": 198.69242126464843,
      "completions/min_length": 161.16,
      "completions/min_terminated_length": 161.16,
      "entropy": 0.06187104433774948,
      "epoch": 0.7372654155495979,
      "frac_reward_zero_std": 0.425,
      "grad_norm": 0.40395304560661316,
      "learning_rate": 1e-05,
      "loss": -0.0025,
      "num_tokens": 69170452.0,
      "reward": 12.482298536300659,
      "reward_std": 1.0457301473617553,
      "rewards/event_reward_fn/mean": 11.64625,
      "rewards/event_reward_fn/std": 7.317771224975586,
      "rewards/format_reward_fn/mean": 0.8360484623908997,
      "rewards/format_reward_fn/std": 0.2895883430540562,
      "step": 550,
      "step_time": 24.193240740820766
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.110625,
      "completions/max_length": 252.62,
      "completions/max_terminated_length": 246.8,
      "completions/mean_length": 208.275625,
      "completions/mean_terminated_length": 202.49910614013672,
      "completions/min_length": 165.54,
      "completions/min_terminated_length": 165.54,
      "entropy": 0.0649487990140915,
      "epoch": 0.8042895442359249,
      "frac_reward_zero_std": 0.38,
      "grad_norm": 0.37119486927986145,
      "learning_rate": 1e-05,
      "loss": 0.0006,
      "num_tokens": 75499314.0,
      "reward": 12.80059557914734,
      "reward_std": 1.1889909988641738,
      "rewards/event_reward_fn/mean": 11.97375,
      "rewards/event_reward_fn/std": 7.475857477188111,
      "rewards/format_reward_fn/mean": 0.8268455564975739,
      "rewards/format_reward_fn/std": 0.29714462146162984,
      "step": 600,
      "step_time": 24.3176869976148
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.05625,
      "completions/max_length": 249.28,
      "completions/max_terminated_length": 244.8,
      "completions/mean_length": 202.789375,
      "completions/mean_terminated_length": 199.91522064208985,
      "completions/min_length": 161.74,
      "completions/min_terminated_length": 161.74,
      "entropy": 0.06481640346348286,
      "epoch": 0.871313672922252,
      "frac_reward_zero_std": 0.3975,
      "grad_norm": 0.08866075426340103,
      "learning_rate": 1e-05,
      "loss": -0.0023,
      "num_tokens": 81673001.0,
      "reward": 12.689926280975342,
      "reward_std": 1.2458794575929641,
      "rewards/event_reward_fn/mean": 11.815625,
      "rewards/event_reward_fn/std": 7.275726590156555,
      "rewards/format_reward_fn/mean": 0.8743013119697571,
      "rewards/format_reward_fn/std": 0.23756251022219657,
      "step": 650,
      "step_time": 25.04028965227306
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.100625,
      "completions/max_length": 253.72,
      "completions/max_terminated_length": 248.28,
      "completions/mean_length": 205.536875,
      "completions/mean_terminated_length": 200.1349432373047,
      "completions/min_length": 162.16,
      "completions/min_terminated_length": 162.16,
      "entropy": 0.0658975774794817,
      "epoch": 0.938337801608579,
      "frac_reward_zero_std": 0.3975,
      "grad_norm": 0.2268964648246765,
      "learning_rate": 1e-05,
      "loss": -0.0008,
      "num_tokens": 87934795.0,
      "reward": 12.72035478591919,
      "reward_std": 1.1722034803032875,
      "rewards/event_reward_fn/mean": 11.888125,
      "rewards/event_reward_fn/std": 7.583159003257752,
      "rewards/format_reward_fn/mean": 0.8322297859191895,
      "rewards/format_reward_fn/std": 0.29026631206274034,
      "step": 700,
      "step_time": 24.744350045956672
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.110625,
      "completions/max_length": 253.56,
      "completions/max_terminated_length": 248.12,
      "completions/mean_length": 205.06125,
      "completions/mean_terminated_length": 198.67494079589844,
      "completions/min_length": 155.32,
      "completions/min_terminated_length": 155.32,
      "entropy": 0.06425789251923561,
      "epoch": 1.0053619302949062,
      "frac_reward_zero_std": 0.4325,
      "grad_norm": 0.6108524799346924,
      "learning_rate": 1e-05,
      "loss": -0.0016,
      "num_tokens": 94007519.0,
      "reward": 12.451499423980714,
      "reward_std": 1.17638818860054,
      "rewards/event_reward_fn/mean": 11.630625,
      "rewards/event_reward_fn/std": 7.498000311851501,
      "rewards/format_reward_fn/mean": 0.8208743929862976,
      "rewards/format_reward_fn/std": 0.3162671920657158,
      "step": 750,
      "step_time": 25.037732787020506
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.11625,
      "completions/max_length": 252.56,
      "completions/max_terminated_length": 246.16,
      "completions/mean_length": 205.586875,
      "completions/mean_terminated_length": 199.2539434814453,
      "completions/min_length": 159.14,
      "completions/min_terminated_length": 159.14,
      "entropy": 0.06738013252615929,
      "epoch": 1.0723860589812333,
      "frac_reward_zero_std": 0.4275,
      "grad_norm": 0.1556655317544937,
      "learning_rate": 1e-05,
      "loss": -0.0027,
      "num_tokens": 100291983.0,
      "reward": 12.259195594787597,
      "reward_std": 1.0736777836084366,
      "rewards/event_reward_fn/mean": 11.440625,
      "rewards/event_reward_fn/std": 7.4294485759735105,
      "rewards/format_reward_fn/mean": 0.8185705220699311,
      "rewards/format_reward_fn/std": 0.30390245616436007,
      "step": 800,
      "step_time": 25.602326317727567
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.124375,
      "completions/max_length": 253.96,
      "completions/max_terminated_length": 248.22,
      "completions/mean_length": 207.960625,
      "completions/mean_terminated_length": 201.18785888671874,
      "completions/min_length": 163.16,
      "completions/min_terminated_length": 163.16,
      "entropy": 0.06615202344954013,
      "epoch": 1.1394101876675604,
      "frac_reward_zero_std": 0.4075,
      "grad_norm": 0.34817561507225037,
      "learning_rate": 1e-05,
      "loss": -0.0008,
      "num_tokens": 106570401.0,
      "reward": 13.199323978424072,
      "reward_std": 1.0555983385443688,
      "rewards/event_reward_fn/mean": 12.388125,
      "rewards/event_reward_fn/std": 7.955943965911866,
      "rewards/format_reward_fn/mean": 0.8111989140510559,
      "rewards/format_reward_fn/std": 0.318571160286665,
      "step": 850,
      "step_time": 25.24974968288094
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.091875,
      "completions/max_length": 254.2,
      "completions/max_terminated_length": 248.06,
      "completions/mean_length": 204.960625,
      "completions/mean_terminated_length": 199.68820373535155,
      "completions/min_length": 160.64,
      "completions/min_terminated_length": 160.64,
      "entropy": 0.06304333973675966,
      "epoch": 1.2064343163538873,
      "frac_reward_zero_std": 0.4475,
      "grad_norm": 0.70001620054245,
      "learning_rate": 1e-05,
      "loss": -0.0014,
      "num_tokens": 112801556.0,
      "reward": 12.732025756835938,
      "reward_std": 1.1594133710861205,
      "rewards/event_reward_fn/mean": 11.89,
      "rewards/event_reward_fn/std": 7.514157109260559,
      "rewards/format_reward_fn/mean": 0.8420258605480194,
      "rewards/format_reward_fn/std": 0.28636473283171654,
      "step": 900,
      "step_time": 26.33290139209479
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.093125,
      "completions/max_length": 253.24,
      "completions/max_terminated_length": 248.26,
      "completions/mean_length": 208.87875,
      "completions/mean_terminated_length": 204.2489434814453,
      "completions/min_length": 162.6,
      "completions/min_terminated_length": 162.6,
      "entropy": 0.06733124047517776,
      "epoch": 1.2734584450402144,
      "frac_reward_zero_std": 0.3875,
      "grad_norm": 0.42013463377952576,
      "learning_rate": 1e-05,
      "loss": -0.0016,
      "num_tokens": 119241388.0,
      "reward": 13.07864158630371,
      "reward_std": 1.2291508412361145,
      "rewards/event_reward_fn/mean": 12.24375,
      "rewards/event_reward_fn/std": 7.096493782997132,
      "rewards/format_reward_fn/mean": 0.8348916172981262,
      "rewards/format_reward_fn/std": 0.29221967339515686,
      "step": 950,
      "step_time": 26.795288713425396
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.115,
      "completions/max_length": 254.4,
      "completions/max_terminated_length": 248.28,
      "completions/mean_length": 208.54375,
      "completions/mean_terminated_length": 202.61700103759765,
      "completions/min_length": 164.5,
      "completions/min_terminated_length": 164.5,
      "entropy": 0.0674970293045044,
      "epoch": 1.3404825737265416,
      "frac_reward_zero_std": 0.4125,
      "grad_norm": 0.22458799183368683,
      "learning_rate": 1e-05,
      "loss": -0.001,
      "num_tokens": 125595072.0,
      "reward": 13.164627075195312,
      "reward_std": 1.1512372946739198,
      "rewards/event_reward_fn/mean": 12.338125,
      "rewards/event_reward_fn/std": 7.617123994827271,
      "rewards/format_reward_fn/mean": 0.8265021049976349,
      "rewards/format_reward_fn/std": 0.3076726086437702,
      "step": 1000,
      "step_time": 26.641401502527298
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.1175,
      "completions/max_length": 254.24,
      "completions/max_terminated_length": 247.84,
      "completions/mean_length": 208.166875,
      "completions/mean_terminated_length": 202.1624597167969,
      "completions/min_length": 165.56,
      "completions/min_terminated_length": 165.56,
      "entropy": 0.06616773374378682,
      "epoch": 1.4075067024128687,
      "frac_reward_zero_std": 0.37,
      "grad_norm": 0.34188446402549744,
      "learning_rate": 1e-05,
      "loss": -0.0015,
      "num_tokens": 131697642.0,
      "reward": 12.04176643371582,
      "reward_std": 1.14944515645504,
      "rewards/event_reward_fn/mean": 11.2275,
      "rewards/event_reward_fn/std": 6.602086658477783,
      "rewards/format_reward_fn/mean": 0.8142664790153503,
      "rewards/format_reward_fn/std": 0.31338023476302623,
      "step": 1050,
      "step_time": 23.740581118687988
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.14625,
      "completions/max_length": 254.62,
      "completions/max_terminated_length": 246.4,
      "completions/mean_length": 210.695,
      "completions/mean_terminated_length": 203.1378680419922,
      "completions/min_length": 165.5,
      "completions/min_terminated_length": 165.5,
      "entropy": 0.06749342102557421,
      "epoch": 1.4745308310991958,
      "frac_reward_zero_std": 0.405,
      "grad_norm": 0.052085984498262405,
      "learning_rate": 1e-05,
      "loss": 0.0009,
      "num_tokens": 137994889.0,
      "reward": 12.919311103820801,
      "reward_std": 1.1956829646229743,
      "rewards/event_reward_fn/mean": 12.114375,
      "rewards/event_reward_fn/std": 7.290353746414184,
      "rewards/format_reward_fn/mean": 0.8049361062049866,
      "rewards/format_reward_fn/std": 0.34213557571172715,
      "step": 1100,
      "step_time": 25.770162526927887
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.089375,
      "completions/max_length": 253.1,
      "completions/max_terminated_length": 248.18,
      "completions/mean_length": 208.190625,
      "completions/mean_terminated_length": 203.62595764160156,
      "completions/min_length": 165.04,
      "completions/min_terminated_length": 165.04,
      "entropy": 0.06642140716314315,
      "epoch": 1.5415549597855227,
      "frac_reward_zero_std": 0.4025,
      "grad_norm": 0.2945052981376648,
      "learning_rate": 1e-05,
      "loss": -0.0024,
      "num_tokens": 144185904.0,
      "reward": 12.36389796257019,
      "reward_std": 1.0678041917830705,
      "rewards/event_reward_fn/mean": 11.530625,
      "rewards/event_reward_fn/std": 6.954381022453308,
      "rewards/format_reward_fn/mean": 0.8332730257511138,
      "rewards/format_reward_fn/std": 0.2778134834766388,
      "step": 1150,
      "step_time": 25.22423570729792
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.08375,
      "completions/max_length": 251.84,
      "completions/max_terminated_length": 248.02,
      "completions/mean_length": 207.27625,
      "completions/mean_terminated_length": 202.8922332763672,
      "completions/min_length": 162.06,
      "completions/min_terminated_length": 162.06,
      "entropy": 0.06358764633536339,
      "epoch": 1.6085790884718498,
      "frac_reward_zero_std": 0.4425,
      "grad_norm": 0.18330521881580353,
      "learning_rate": 1e-05,
      "loss": 0.0013,
      "num_tokens": 150418096.0,
      "reward": 13.024293642044068,
      "reward_std": 1.0176034840941428,
      "rewards/event_reward_fn/mean": 12.1875,
      "rewards/event_reward_fn/std": 7.625103950500488,
      "rewards/format_reward_fn/mean": 0.8367935848236084,
      "rewards/format_reward_fn/std": 0.2789352157711983,
      "step": 1200,
      "step_time": 25.085855303443967
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.061875,
      "completions/max_length": 252.08,
      "completions/max_terminated_length": 246.72,
      "completions/mean_length": 203.59125,
      "completions/mean_terminated_length": 200.21913879394532,
      "completions/min_length": 163.32,
      "completions/min_terminated_length": 163.32,
      "entropy": 0.062210914567112925,
      "epoch": 1.675603217158177,
      "frac_reward_zero_std": 0.47,
      "grad_norm": 0.40231508016586304,
      "learning_rate": 1e-05,
      "loss": -0.0028,
      "num_tokens": 156497077.0,
      "reward": 13.417193460464478,
      "reward_std": 0.9512623021006584,
      "rewards/event_reward_fn/mean": 12.55625,
      "rewards/event_reward_fn/std": 7.378204255104065,
      "rewards/format_reward_fn/mean": 0.8609435141086579,
      "rewards/format_reward_fn/std": 0.2591093431413174,
      "step": 1250,
      "step_time": 24.335955093875526
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.083125,
      "completions/max_length": 252.32,
      "completions/max_terminated_length": 248.26,
      "completions/mean_length": 208.861875,
      "completions/mean_terminated_length": 204.65392211914062,
      "completions/min_length": 167.5,
      "completions/min_terminated_length": 167.5,
      "entropy": 0.06511906914412975,
      "epoch": 1.742627345844504,
      "frac_reward_zero_std": 0.4125,
      "grad_norm": 0.5480676889419556,
      "learning_rate": 1e-05,
      "loss": 0.0038,
      "num_tokens": 162925421.0,
      "reward": 13.32497908592224,
      "reward_std": 1.179861811459996,
      "rewards/event_reward_fn/mean": 12.480625,
      "rewards/event_reward_fn/std": 7.050848822593689,
      "rewards/format_reward_fn/mean": 0.8443540966510773,
      "rewards/format_reward_fn/std": 0.27325084805488586,
      "step": 1300,
      "step_time": 25.89649803057313
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.09,
      "completions/max_length": 253.5,
      "completions/max_terminated_length": 248.1,
      "completions/mean_length": 208.64,
      "completions/mean_terminated_length": 204.13995147705077,
      "completions/min_length": 161.3,
      "completions/min_terminated_length": 161.3,
      "entropy": 0.06706013955175877,
      "epoch": 1.8096514745308312,
      "frac_reward_zero_std": 0.375,
      "grad_norm": 0.2488502711057663,
      "learning_rate": 1e-05,
      "loss": -0.0035,
      "num_tokens": 169272835.0,
      "reward": 12.847478866577148,
      "reward_std": 1.1458918780088425,
      "rewards/event_reward_fn/mean": 11.995,
      "rewards/event_reward_fn/std": 7.542719440460205,
      "rewards/format_reward_fn/mean": 0.852478951215744,
      "rewards/format_reward_fn/std": 0.28421372681856155,
      "step": 1350,
      "step_time": 25.208012702837586
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.10375,
      "completions/max_length": 253.5,
      "completions/max_terminated_length": 248.62,
      "completions/mean_length": 209.326875,
      "completions/mean_terminated_length": 204.015185546875,
      "completions/min_length": 164.82,
      "completions/min_terminated_length": 164.82,
      "entropy": 0.06814022108912468,
      "epoch": 1.876675603217158,
      "frac_reward_zero_std": 0.3525,
      "grad_norm": 0.16522294282913208,
      "learning_rate": 1e-05,
      "loss": -0.0006,
      "num_tokens": 175702725.0,
      "reward": 13.004317455291748,
      "reward_std": 1.250138430595398,
      "rewards/event_reward_fn/mean": 12.18,
      "rewards/event_reward_fn/std": 7.555402121543884,
      "rewards/format_reward_fn/mean": 0.8243174004554749,
      "rewards/format_reward_fn/std": 0.2965914273262024,
      "step": 1400,
      "step_time": 26.651869118101896
    },
    {
      "clip_ratio/high_max": 0.0,
      "clip_ratio/high_mean": 0.0,
      "clip_ratio/low_mean": 0.0,
      "clip_ratio/low_min": 0.0,
      "clip_ratio/region_mean": 0.0,
      "completions/clipped_ratio": 0.066875,
      "completions/max_length": 252.98,
      "completions/max_terminated_length": 249.98,
      "completions/mean_length": 208.22625,
      "completions/mean_terminated_length": 204.81213348388673,
      "completions/min_length": 164.02,
      "completions/min_terminated_length": 164.02,
      "entropy": 0.0681792851537466,
      "epoch": 1.9436997319034852,
      "frac_reward_zero_std": 0.3775,
      "grad_norm": 0.3131242096424103,
      "learning_rate": 1e-05,
      "loss": -0.0045,
      "num_tokens": 181968342.0,
      "reward": 12.955870590209962,
      "reward_std": 1.2048768895864486,
      "rewards/event_reward_fn/mean": 12.0975,
      "rewards/event_reward_fn/std": 7.502232184410095,
      "rewards/format_reward_fn/mean": 0.8583705246448516,
      "rewards/format_reward_fn/std": 0.2481384778022766,
      "step": 1450,
      "step_time": 23.915284326784313
    }
  ],
  "logging_steps": 50,
  "max_steps": 7460,
  "num_input_tokens_seen": 187096042,
  "num_train_epochs": 10,
  "save_steps": 500,
  "stateful_callbacks": {
    "TrainerControl": {
      "args": {
        "should_epoch_stop": false,
        "should_evaluate": false,
        "should_log": false,
        "should_save": true,
        "should_training_stop": false
      },
      "attributes": {}
    }
  },
  "total_flos": 0.0,
  "train_batch_size": 4,
  "trial_name": null,
  "trial_params": null
}