{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0133333333333334, "eval_steps": 500, "global_step": 76, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.1675567626953125, "epoch": 0.013333333333333334, "frac_reward_zero_std": 0.75, "grad_norm": 27.889081954956055, "learning_rate": 0.0, "loss": -0.0, "num_tokens": 57308.0, "reward": 0.671875, "reward_std": 0.28125, "rewards/category_reward_func/mean": 0.671875, "rewards/category_reward_func/std": 1.273835301399231, "step": 1 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.1007843017578125, "epoch": 0.02666666666666667, "frac_reward_zero_std": 0.75, "grad_norm": 19.59115982055664, "learning_rate": 2.153382790366965e-07, "loss": -0.0, "num_tokens": 99008.0, "reward": 0.671875, "reward_std": 0.28125, "rewards/category_reward_func/mean": 0.671875, "rewards/category_reward_func/std": 1.273835301399231, "step": 2 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.2137451171875, "epoch": 0.04, "frac_reward_zero_std": 0.75, "grad_norm": 15.223323822021484, "learning_rate": 3.4130309724299266e-07, "loss": -0.0, "num_tokens": 124196.0, "reward": 1.359375, "reward_std": 0.28125, "rewards/category_reward_func/mean": 1.359375, "rewards/category_reward_func/std": 0.5625, "step": 3 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.1277008056640625, "epoch": 0.05333333333333334, "frac_reward_zero_std": 0.75, "grad_norm": 17.724227905273438, "learning_rate": 4.30676558073393e-07, "loss": -0.0, "num_tokens": 165020.0, "reward": 1.359375, "reward_std": 0.28125, "rewards/category_reward_func/mean": 1.359375, "rewards/category_reward_func/std": 0.5625, "step": 4 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.198822021484375, "epoch": 0.06666666666666667, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5e-07, "loss": 0.0, "num_tokens": 241172.0, "reward": 1.5, "reward_std": 0.0, "rewards/category_reward_func/mean": 1.5, "rewards/category_reward_func/std": 0.0, "step": 5 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.175628662109375, "epoch": 0.08, "frac_reward_zero_std": 0.75, "grad_norm": 19.26618003845215, "learning_rate": 5.566413762796892e-07, "loss": 0.0, "num_tokens": 293140.0, "reward": 0.421875, "reward_std": 0.34375, "rewards/category_reward_func/mean": 0.421875, "rewards/category_reward_func/std": 1.273835301399231, "step": 6 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.175262451171875, "epoch": 0.09333333333333334, "frac_reward_zero_std": 0.75, "grad_norm": 16.38401985168457, "learning_rate": 6.045309775610838e-07, "loss": 0.0, "num_tokens": 352196.0, "reward": 1.21875, "reward_std": 0.32475951313972473, "rewards/category_reward_func/mean": 1.21875, "rewards/category_reward_func/std": 0.7685213088989258, "step": 7 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.221466064453125, "epoch": 0.10666666666666667, "frac_reward_zero_std": 0.75, "grad_norm": 21.485958099365234, "learning_rate": 6.460148371100895e-07, "loss": 0.0, "num_tokens": 396284.0, "reward": 1.21875, "reward_std": 0.32475951313972473, "rewards/category_reward_func/mean": 1.21875, "rewards/category_reward_func/std": 0.7685213088989258, "step": 8 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.14752197265625, "epoch": 0.12, "frac_reward_zero_std": 0.5, "grad_norm": 38.9441032409668, "learning_rate": 6.826061944859853e-07, "loss": -0.0, "num_tokens": 430332.0, "reward": 1.1875, "reward_std": 0.625, "rewards/category_reward_func/mean": 1.1875, "rewards/category_reward_func/std": 0.8587782382965088, "step": 9 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.26763916015625, "epoch": 0.13333333333333333, "frac_reward_zero_std": 0.25, "grad_norm": 48.07648849487305, "learning_rate": 7.153382790366966e-07, "loss": 0.0, "num_tokens": 468928.0, "reward": 0.328125, "reward_std": 0.65625, "rewards/category_reward_func/mean": 0.328125, "rewards/category_reward_func/std": 1.2237876653671265, "step": 10 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.091400146484375, "epoch": 0.14666666666666667, "frac_reward_zero_std": 0.5, "grad_norm": 63.72185516357422, "learning_rate": 7.449480512024891e-07, "loss": 0.0, "num_tokens": 499832.0, "reward": 0.65625, "reward_std": 0.5625, "rewards/category_reward_func/mean": 0.65625, "rewards/category_reward_func/std": 1.125, "step": 11 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.2340087890625, "epoch": 0.16, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.719796553163857e-07, "loss": 0.0, "num_tokens": 550096.0, "reward": 1.5, "reward_std": 0.0, "rewards/category_reward_func/mean": 1.5, "rewards/category_reward_func/std": 0.0, "step": 12 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.1494903564453125, "epoch": 0.17333333333333334, "frac_reward_zero_std": 0.25, "grad_norm": 49.42058563232422, "learning_rate": 7.968463205835412e-07, "loss": 0.0, "num_tokens": 589668.0, "reward": 0.46875, "reward_std": 0.9925506114959717, "rewards/category_reward_func/mean": 0.46875, "rewards/category_reward_func/std": 1.2209115028381348, "step": 13 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.169189453125, "epoch": 0.18666666666666668, "frac_reward_zero_std": 0.25, "grad_norm": 48.9026985168457, "learning_rate": 8.198692565977803e-07, "loss": 0.0, "num_tokens": 619500.0, "reward": 0.515625, "reward_std": 0.9307689666748047, "rewards/category_reward_func/mean": 0.515625, "rewards/category_reward_func/std": 1.1527819633483887, "step": 14 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.201019287109375, "epoch": 0.2, "frac_reward_zero_std": 0.75, "grad_norm": 44.559425354003906, "learning_rate": 8.413030972429927e-07, "loss": 0.0, "num_tokens": 665544.0, "reward": -0.046875, "reward_std": 0.28125, "rewards/category_reward_func/mean": -0.046875, "rewards/category_reward_func/std": 1.0771055221557617, "step": 15 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.22216796875, "epoch": 0.21333333333333335, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 8.61353116146786e-07, "loss": 0.0, "num_tokens": 711584.0, "reward": 1.5, "reward_std": 0.0, "rewards/category_reward_func/mean": 1.5, "rewards/category_reward_func/std": 0.0, "step": 16 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.0984649658203125, "epoch": 0.22666666666666666, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 8.801872138612941e-07, "loss": 0.0, "num_tokens": 761108.0, "reward": 0.9375, "reward_std": 0.0, "rewards/category_reward_func/mean": 0.9375, "rewards/category_reward_func/std": 1.0062305927276611, "step": 17 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.23809814453125, "epoch": 0.24, "frac_reward_zero_std": 0.75, "grad_norm": 29.925472259521484, "learning_rate": 8.979444735226818e-07, "loss": 0.0, "num_tokens": 800464.0, "reward": 1.21875, "reward_std": 0.32475951313972473, "rewards/category_reward_func/mean": 1.21875, "rewards/category_reward_func/std": 0.7685213088989258, "step": 18 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.1519927978515625, "epoch": 0.25333333333333335, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 9.147414002175752e-07, "loss": 0.0, "num_tokens": 858768.0, "reward": 1.5, "reward_std": 0.0, "rewards/category_reward_func/mean": 1.5, "rewards/category_reward_func/std": 0.0, "step": 19 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.2058868408203125, "epoch": 0.26666666666666666, "frac_reward_zero_std": 0.75, "grad_norm": 23.14743423461914, "learning_rate": 9.306765580733931e-07, "loss": 0.0, "num_tokens": 918644.0, "reward": 1.328125, "reward_std": 0.34375, "rewards/category_reward_func/mean": 1.328125, "rewards/category_reward_func/std": 0.6875, "step": 20 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.1350860595703125, "epoch": 0.28, "frac_reward_zero_std": 0.25, "grad_norm": 36.214595794677734, "learning_rate": 9.458340748040765e-07, "loss": -0.0, "num_tokens": 954748.0, "reward": 0.796875, "reward_std": 0.84375, "rewards/category_reward_func/mean": 0.796875, "rewards/category_reward_func/std": 1.0771055221557617, "step": 21 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.178436279296875, "epoch": 0.29333333333333333, "frac_reward_zero_std": 0.5, "grad_norm": 31.41033363342285, "learning_rate": 9.602863302391857e-07, "loss": 0.0, "num_tokens": 987760.0, "reward": 0.515625, "reward_std": 0.65625, "rewards/category_reward_func/mean": 0.515625, "rewards/category_reward_func/std": 1.333756446838379, "step": 22 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.157470703125, "epoch": 0.30666666666666664, "frac_reward_zero_std": 0.5, "grad_norm": 29.780487060546875, "learning_rate": 9.740960467331898e-07, "loss": 0.0, "num_tokens": 1042116.0, "reward": 0.609375, "reward_std": 0.6060094833374023, "rewards/category_reward_func/mean": 0.609375, "rewards/category_reward_func/std": 1.2005858421325684, "step": 23 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.099609375, "epoch": 0.32, "frac_reward_zero_std": 0.5, "grad_norm": 20.054729461669922, "learning_rate": 9.873179343530824e-07, "loss": -0.0, "num_tokens": 1083348.0, "reward": 1.21875, "reward_std": 0.5625, "rewards/category_reward_func/mean": 1.21875, "rewards/category_reward_func/std": 0.7685213088989258, "step": 24 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.133087158203125, "epoch": 0.3333333333333333, "frac_reward_zero_std": 0.5, "grad_norm": 49.41580581665039, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1133176.0, "reward": 0.9375, "reward_std": 0.5625, "rewards/category_reward_func/mean": 0.9375, "rewards/category_reward_func/std": 1.0062305927276611, "step": 25 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.0871124267578125, "epoch": 0.3466666666666667, "frac_reward_zero_std": 0.75, "grad_norm": 29.15717124938965, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1164928.0, "reward": 0.65625, "reward_std": 0.32475951313972473, "rewards/category_reward_func/mean": 0.65625, "rewards/category_reward_func/std": 1.125, "step": 26 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.075164794921875, "epoch": 0.36, "frac_reward_zero_std": 0.5, "grad_norm": 46.123504638671875, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1199224.0, "reward": 0.25, "reward_std": 0.6495190262794495, "rewards/category_reward_func/mean": 0.25, "rewards/category_reward_func/std": 1.3038405179977417, "step": 27 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.289825439453125, "epoch": 0.37333333333333335, "frac_reward_zero_std": 0.5, "grad_norm": 76.34833526611328, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 1221156.0, "reward": 0.515625, "reward_std": 0.6060094833374023, "rewards/category_reward_func/mean": 0.515625, "rewards/category_reward_func/std": 1.1527819633483887, "step": 28 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.135528564453125, "epoch": 0.38666666666666666, "frac_reward_zero_std": 0.5, "grad_norm": 52.0339469909668, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1254016.0, "reward": 0.796875, "reward_std": 0.6060094833374023, "rewards/category_reward_func/mean": 0.796875, "rewards/category_reward_func/std": 1.0771055221557617, "step": 29 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.202178955078125, "epoch": 0.4, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1321252.0, "reward": 1.5, "reward_std": 0.0, "rewards/category_reward_func/mean": 1.5, "rewards/category_reward_func/std": 0.0, "step": 30 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.174652099609375, "epoch": 0.41333333333333333, "frac_reward_zero_std": 0.75, "grad_norm": 18.99894142150879, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1357452.0, "reward": 1.21875, "reward_std": 0.32475951313972473, "rewards/category_reward_func/mean": 1.21875, "rewards/category_reward_func/std": 0.7685213088989258, "step": 31 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.221099853515625, "epoch": 0.4266666666666667, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1411864.0, "reward": 1.5, "reward_std": 0.0, "rewards/category_reward_func/mean": 1.5, "rewards/category_reward_func/std": 0.0, "step": 32 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.216094970703125, "epoch": 0.44, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1454212.0, "reward": 0.9375, "reward_std": 0.0, "rewards/category_reward_func/mean": 0.9375, "rewards/category_reward_func/std": 1.0062305927276611, "step": 33 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.24285888671875, "epoch": 0.4533333333333333, "frac_reward_zero_std": 0.5, "grad_norm": 24.658159255981445, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1491820.0, "reward": 0.109375, "reward_std": 0.6060094833374023, "rewards/category_reward_func/mean": 0.109375, "rewards/category_reward_func/std": 1.281173825263977, "step": 34 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.271026611328125, "epoch": 0.4666666666666667, "frac_reward_zero_std": 0.75, "grad_norm": 29.675661087036133, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 1543328.0, "reward": 1.359375, "reward_std": 0.28125, "rewards/category_reward_func/mean": 1.359375, "rewards/category_reward_func/std": 0.5625, "step": 35 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.20477294921875, "epoch": 0.48, "frac_reward_zero_std": 0.5, "grad_norm": 34.019290924072266, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1581208.0, "reward": 0.9375, "reward_std": 0.5625, "rewards/category_reward_func/mean": 0.9375, "rewards/category_reward_func/std": 1.0062305927276611, "step": 36 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.163177490234375, "epoch": 0.49333333333333335, "frac_reward_zero_std": 0.75, "grad_norm": 27.8812255859375, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 1614672.0, "reward": 1.359375, "reward_std": 0.28125, "rewards/category_reward_func/mean": 1.359375, "rewards/category_reward_func/std": 0.5625, "step": 37 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.0792388916015625, "epoch": 0.5066666666666667, "frac_reward_zero_std": 0.25, "grad_norm": 30.810190200805664, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1656416.0, "reward": 0.65625, "reward_std": 0.8872594833374023, "rewards/category_reward_func/mean": 0.65625, "rewards/category_reward_func/std": 1.125, "step": 38 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.2446746826171875, "epoch": 0.52, "frac_reward_zero_std": 0.75, "grad_norm": 21.632761001586914, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1709100.0, "reward": 1.078125, "reward_std": 0.28125, "rewards/category_reward_func/mean": 1.078125, "rewards/category_reward_func/std": 0.9070039987564087, "step": 39 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.181365966796875, "epoch": 0.5333333333333333, "frac_reward_zero_std": 0.5, "grad_norm": 21.632286071777344, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 1755020.0, "reward": 1.21875, "reward_std": 0.5625, "rewards/category_reward_func/mean": 1.21875, "rewards/category_reward_func/std": 0.7685213088989258, "step": 40 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.157623291015625, "epoch": 0.5466666666666666, "frac_reward_zero_std": 0.5, "grad_norm": 28.650922775268555, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1786396.0, "reward": 0.796875, "reward_std": 0.6060094833374023, "rewards/category_reward_func/mean": 0.796875, "rewards/category_reward_func/std": 1.0771055221557617, "step": 41 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.1880950927734375, "epoch": 0.56, "frac_reward_zero_std": 0.25, "grad_norm": 46.74673080444336, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1805832.0, "reward": 0.375, "reward_std": 0.8872594833374023, "rewards/category_reward_func/mean": 0.375, "rewards/category_reward_func/std": 1.1618950366973877, "step": 42 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.205352783203125, "epoch": 0.5733333333333334, "frac_reward_zero_std": 0.0, "grad_norm": 57.66422653198242, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1828832.0, "reward": 0.234375, "reward_std": 1.1685094833374023, "rewards/category_reward_func/mean": 0.234375, "rewards/category_reward_func/std": 1.1527819633483887, "step": 43 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.43341064453125, "epoch": 0.5866666666666667, "frac_reward_zero_std": 0.75, "grad_norm": 22.34069061279297, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1882088.0, "reward": 1.125, "reward_std": 0.4330126941204071, "rewards/category_reward_func/mean": 1.125, "rewards/category_reward_func/std": 1.0246951580047607, "step": 44 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.317779541015625, "epoch": 0.6, "frac_reward_zero_std": 0.5, "grad_norm": 20.323429107666016, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 1935288.0, "reward": 0.9375, "reward_std": 0.5625, "rewards/category_reward_func/mean": 0.9375, "rewards/category_reward_func/std": 1.0062305927276611, "step": 45 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.321075439453125, "epoch": 0.6133333333333333, "frac_reward_zero_std": 0.25, "grad_norm": 143.27099609375, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 1960884.0, "reward": 0.796875, "reward_std": 0.9307689666748047, "rewards/category_reward_func/mean": 0.796875, "rewards/category_reward_func/std": 1.0771055221557617, "step": 46 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.469451904296875, "epoch": 0.6266666666666667, "frac_reward_zero_std": 0.5, "grad_norm": 34.52668762207031, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 2022248.0, "reward": 0.984375, "reward_std": 0.7142627239227295, "rewards/category_reward_func/mean": 0.984375, "rewards/category_reward_func/std": 1.1197795867919922, "step": 47 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.20635986328125, "epoch": 0.64, "frac_reward_zero_std": 0.75, "grad_norm": 14.806098937988281, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 2073196.0, "reward": 1.359375, "reward_std": 0.28125, "rewards/category_reward_func/mean": 1.359375, "rewards/category_reward_func/std": 0.5625, "step": 48 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.25537109375, "epoch": 0.6533333333333333, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2147716.0, "reward": 1.5, "reward_std": 0.0, "rewards/category_reward_func/mean": 1.5, "rewards/category_reward_func/std": 0.0, "step": 49 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.28509521484375, "epoch": 0.6666666666666666, "frac_reward_zero_std": 0.5, "grad_norm": 30.238126754760742, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2198480.0, "reward": 1.03125, "reward_std": 0.6997594833374023, "rewards/category_reward_func/mean": 1.03125, "rewards/category_reward_func/std": 1.0201103687286377, "step": 50 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.242279052734375, "epoch": 0.68, "frac_reward_zero_std": 0.75, "grad_norm": 23.85097312927246, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2254876.0, "reward": 1.125, "reward_std": 0.4330126941204071, "rewards/category_reward_func/mean": 1.125, "rewards/category_reward_func/std": 1.0246951580047607, "step": 51 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.44366455078125, "epoch": 0.6933333333333334, "frac_reward_zero_std": 0.75, "grad_norm": 18.61939811706543, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 2291896.0, "reward": 0.234375, "reward_std": 0.28125, "rewards/category_reward_func/mean": 0.234375, "rewards/category_reward_func/std": 1.1527819633483887, "step": 52 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.39569091796875, "epoch": 0.7066666666666667, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2345988.0, "reward": 1.5, "reward_std": 0.0, "rewards/category_reward_func/mean": 1.5, "rewards/category_reward_func/std": 0.0, "step": 53 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.429656982421875, "epoch": 0.72, "frac_reward_zero_std": 0.75, "grad_norm": 19.174150466918945, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2382980.0, "reward": 1.21875, "reward_std": 0.32475951313972473, "rewards/category_reward_func/mean": 1.21875, "rewards/category_reward_func/std": 0.7685213088989258, "step": 54 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.2694091796875, "epoch": 0.7333333333333333, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2433172.0, "reward": 1.5, "reward_std": 0.0, "rewards/category_reward_func/mean": 1.5, "rewards/category_reward_func/std": 0.0, "step": 55 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.37933349609375, "epoch": 0.7466666666666667, "frac_reward_zero_std": 0.75, "grad_norm": 11.9332857131958, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 2472316.0, "reward": 1.359375, "reward_std": 0.28125, "rewards/category_reward_func/mean": 1.359375, "rewards/category_reward_func/std": 0.5625, "step": 56 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.4208984375, "epoch": 0.76, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2524856.0, "reward": 0.8125, "reward_std": 0.0, "rewards/category_reward_func/mean": 0.8125, "rewards/category_reward_func/std": 1.229837417602539, "step": 57 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.561767578125, "epoch": 0.7733333333333333, "frac_reward_zero_std": 0.75, "grad_norm": 11.59165096282959, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2571776.0, "reward": 1.078125, "reward_std": 0.28125, "rewards/category_reward_func/mean": 1.078125, "rewards/category_reward_func/std": 0.9070039987564087, "step": 58 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.3621826171875, "epoch": 0.7866666666666666, "frac_reward_zero_std": 0.5, "grad_norm": 19.301597595214844, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 2606296.0, "reward": 1.21875, "reward_std": 0.5625, "rewards/category_reward_func/mean": 1.21875, "rewards/category_reward_func/std": 0.7685213088989258, "step": 59 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.36553955078125, "epoch": 0.8, "frac_reward_zero_std": 0.75, "grad_norm": 10.53289794921875, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2654164.0, "reward": 1.078125, "reward_std": 0.28125, "rewards/category_reward_func/mean": 1.078125, "rewards/category_reward_func/std": 0.9070039987564087, "step": 60 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.2923583984375, "epoch": 0.8133333333333334, "frac_reward_zero_std": 0.5, "grad_norm": 99.41712188720703, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 2698972.0, "reward": 1.078125, "reward_std": 0.6060094833374023, "rewards/category_reward_func/mean": 1.078125, "rewards/category_reward_func/std": 0.9070039987564087, "step": 61 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.400146484375, "epoch": 0.8266666666666667, "frac_reward_zero_std": 0.5, "grad_norm": 21.47633934020996, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 2734224.0, "reward": 0.65625, "reward_std": 0.5625, "rewards/category_reward_func/mean": 0.65625, "rewards/category_reward_func/std": 1.125, "step": 62 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.38702392578125, "epoch": 0.84, "frac_reward_zero_std": 0.5, "grad_norm": 29.996910095214844, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 2774124.0, "reward": 1.078125, "reward_std": 0.6060094833374023, "rewards/category_reward_func/mean": 1.078125, "rewards/category_reward_func/std": 0.9070039987564087, "step": 63 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.38385009765625, "epoch": 0.8533333333333334, "frac_reward_zero_std": 0.75, "grad_norm": 15.907926559448242, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2833568.0, "reward": 1.21875, "reward_std": 0.32475951313972473, "rewards/category_reward_func/mean": 1.21875, "rewards/category_reward_func/std": 0.7685213088989258, "step": 64 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.3819580078125, "epoch": 0.8666666666666667, "frac_reward_zero_std": 0.0, "grad_norm": 46.19581604003906, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2860380.0, "reward": 0.234375, "reward_std": 1.1685094833374023, "rewards/category_reward_func/mean": 0.234375, "rewards/category_reward_func/std": 1.1527819633483887, "step": 65 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.35260009765625, "epoch": 0.88, "frac_reward_zero_std": 0.5, "grad_norm": 22.587528228759766, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 2909180.0, "reward": 1.21875, "reward_std": 0.5625, "rewards/category_reward_func/mean": 1.21875, "rewards/category_reward_func/std": 0.7685213088989258, "step": 66 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.392333984375, "epoch": 0.8933333333333333, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2947732.0, "reward": 1.5, "reward_std": 0.0, "rewards/category_reward_func/mean": 1.5, "rewards/category_reward_func/std": 0.0, "step": 67 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.422607421875, "epoch": 0.9066666666666666, "frac_reward_zero_std": 0.5, "grad_norm": 20.867935180664062, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 2960580.0, "reward": 0.515625, "reward_std": 0.6060094833374023, "rewards/category_reward_func/mean": 0.515625, "rewards/category_reward_func/std": 1.1527819633483887, "step": 68 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.3450927734375, "epoch": 0.92, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 2997376.0, "reward": 1.5, "reward_std": 0.0, "rewards/category_reward_func/mean": 1.5, "rewards/category_reward_func/std": 0.0, "step": 69 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.4351806640625, "epoch": 0.9333333333333333, "frac_reward_zero_std": 0.75, "grad_norm": 21.277936935424805, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 3071648.0, "reward": 1.3125, "reward_std": 0.375, "rewards/category_reward_func/mean": 1.3125, "rewards/category_reward_func/std": 0.75, "step": 70 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.454345703125, "epoch": 0.9466666666666667, "frac_reward_zero_std": 0.5, "grad_norm": 11.43299674987793, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 3131384.0, "reward": 0.65625, "reward_std": 0.5625, "rewards/category_reward_func/mean": 0.65625, "rewards/category_reward_func/std": 1.125, "step": 71 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.323486328125, "epoch": 0.96, "frac_reward_zero_std": 0.75, "grad_norm": 11.89018726348877, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 3172224.0, "reward": 0.671875, "reward_std": 0.28125, "rewards/category_reward_func/mean": 0.671875, "rewards/category_reward_func/std": 1.273835301399231, "step": 72 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.457763671875, "epoch": 0.9733333333333334, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 3207712.0, "reward": 0.125, "reward_std": 0.0, "rewards/category_reward_func/mean": 0.125, "rewards/category_reward_func/std": 1.4200940132141113, "step": 73 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.28741455078125, "epoch": 0.9866666666666667, "frac_reward_zero_std": 0.5, "grad_norm": 23.053607940673828, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 3234472.0, "reward": 1.21875, "reward_std": 0.5625, "rewards/category_reward_func/mean": 1.21875, "rewards/category_reward_func/std": 0.7685213088989258, "step": 74 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.346923828125, "epoch": 1.0, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1e-06, "loss": 0.0, "num_tokens": 3272428.0, "reward": 1.5, "reward_std": 0.0, "rewards/category_reward_func/mean": 1.5, "rewards/category_reward_func/std": 0.0, "step": 75 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 2.0, "completions/max_terminated_length": 2.0, "completions/mean_length": 2.0, "completions/mean_terminated_length": 2.0, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 0.29266357421875, "epoch": 1.0133333333333334, "frac_reward_zero_std": 0.75, "grad_norm": 11.529069900512695, "learning_rate": 1e-06, "loss": -0.0, "num_tokens": 3302836.0, "reward": 0.796875, "reward_std": 0.28125, "rewards/category_reward_func/mean": 0.796875, "rewards/category_reward_func/std": 1.0771055221557617, "step": 76 } ], "logging_steps": 1.0, "max_steps": 225, "num_input_tokens_seen": 3302836, "num_train_epochs": 3, "save_steps": 38, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 1, "trial_name": null, "trial_params": null }