{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 2000.0, "global_step": 3595, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.140625, "completions/max_length": 513.0, "completions/mean_length": 397.359375, "completions/min_length": 244.0, "epoch": 0.0002781641168289291, "frac_reward_zero_std": 0.0, "grad_norm": 0.7349766492843628, "kl": 0.0, "learning_rate": 5.555555555555555e-09, "loss": 6.970367394387722e-09, "reward": 1.9015536308288574, "reward_std": 0.8292676210403442, "rewards/IngredientFormatReward/mean": 0.748828113079071, "rewards/IngredientFormatReward/std": 0.4224153757095337, "rewards/IngredientMatchReward/mean": 0.3902685046195984, "rewards/IngredientMatchReward/std": 0.28992533683776855, "rewards/IngredientQuantityMatchReward/mean": 0.5593319535255432, "rewards/IngredientQuantityMatchReward/std": 0.4461570084095001, "rewards/TotalKcalExactMatchReward/mean": 0.203125, "rewards/TotalKcalExactMatchReward/std": 0.40390563011169434, "step": 1 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 513.0, "completions/mean_length": 403.083984375, "completions/min_length": 247.75, "epoch": 0.0013908205841446453, "frac_reward_zero_std": 0.015625, "grad_norm": 0.7494612336158752, "kl": 0.00029059583584967186, "learning_rate": 2.7777777777777774e-08, "loss": 1.1643016478046775e-05, "reward": 2.084854483604431, "reward_std": 0.7374771535396576, "rewards/IngredientFormatReward/mean": 0.8349934965372086, "rewards/IngredientFormatReward/std": 0.35871507972478867, "rewards/IngredientMatchReward/mean": 0.45674292743206024, "rewards/IngredientMatchReward/std": 0.3166300803422928, "rewards/IngredientQuantityMatchReward/mean": 0.5743680000305176, "rewards/IngredientQuantityMatchReward/std": 0.450081929564476, "rewards/TotalKcalExactMatchReward/mean": 0.21875, "rewards/TotalKcalExactMatchReward/std": 0.41126810014247894, "step": 5 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 513.0, "completions/mean_length": 395.8078125, "completions/min_length": 222.2, "epoch": 0.0027816411682892906, "frac_reward_zero_std": 0.0, "grad_norm": 0.7679515480995178, "kl": 0.00040537232052884065, "learning_rate": 5.555555555555555e-08, "loss": 1.6196584329009057e-05, "reward": 2.1525042057037354, "reward_std": 0.6553369760513306, "rewards/IngredientFormatReward/mean": 0.85098956823349, "rewards/IngredientFormatReward/std": 0.3357125997543335, "rewards/IngredientMatchReward/mean": 0.46298115253448485, "rewards/IngredientMatchReward/std": 0.2870068609714508, "rewards/IngredientQuantityMatchReward/mean": 0.5979084730148315, "rewards/IngredientQuantityMatchReward/std": 0.43828884363174436, "rewards/TotalKcalExactMatchReward/mean": 0.240625, "rewards/TotalKcalExactMatchReward/std": 0.42608442306518557, "step": 10 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08125, "completions/max_length": 513.0, "completions/mean_length": 394.2984375, "completions/min_length": 217.4, "epoch": 0.004172461752433936, "frac_reward_zero_std": 0.0, "grad_norm": 0.7247325778007507, "kl": 0.00041362347001268064, "learning_rate": 8.333333333333333e-08, "loss": 1.6542989760637285e-05, "reward": 1.9850705146789551, "reward_std": 0.6654590249061585, "rewards/IngredientFormatReward/mean": 0.8356510400772095, "rewards/IngredientFormatReward/std": 0.3531181126832962, "rewards/IngredientMatchReward/mean": 0.460903400182724, "rewards/IngredientMatchReward/std": 0.30696226954460143, "rewards/IngredientQuantityMatchReward/mean": 0.4963285565376282, "rewards/IngredientQuantityMatchReward/std": 0.43468194007873534, "rewards/TotalKcalExactMatchReward/mean": 0.1921875, "rewards/TotalKcalExactMatchReward/std": 0.39231924414634706, "step": 15 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0671875, "completions/max_length": 513.0, "completions/mean_length": 402.853125, "completions/min_length": 269.8, "epoch": 0.005563282336578581, "frac_reward_zero_std": 0.0, "grad_norm": 0.725919246673584, "kl": 0.00042192289620288646, "learning_rate": 1.111111111111111e-07, "loss": 1.6887777019292115e-05, "reward": 2.064033532142639, "reward_std": 0.7137743353843689, "rewards/IngredientFormatReward/mean": 0.8258333444595337, "rewards/IngredientFormatReward/std": 0.36316843032836915, "rewards/IngredientMatchReward/mean": 0.43542484641075135, "rewards/IngredientMatchReward/std": 0.28290987610816953, "rewards/IngredientQuantityMatchReward/mean": 0.5527753114700318, "rewards/IngredientQuantityMatchReward/std": 0.4324473083019257, "rewards/TotalKcalExactMatchReward/mean": 0.25, "rewards/TotalKcalExactMatchReward/std": 0.43261595964431765, "step": 20 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.078125, "completions/max_length": 513.0, "completions/mean_length": 403.4921875, "completions/min_length": 247.4, "epoch": 0.006954102920723227, "frac_reward_zero_std": 0.0, "grad_norm": 0.7491269707679749, "kl": 0.00040630993462400513, "learning_rate": 1.3888888888888888e-07, "loss": 1.6278773546218873e-05, "reward": 2.092119598388672, "reward_std": 0.7215837955474853, "rewards/IngredientFormatReward/mean": 0.8207626461982727, "rewards/IngredientFormatReward/std": 0.3697893679141998, "rewards/IngredientMatchReward/mean": 0.4614254832267761, "rewards/IngredientMatchReward/std": 0.31788976192474366, "rewards/IngredientQuantityMatchReward/mean": 0.5443065166473389, "rewards/IngredientQuantityMatchReward/std": 0.4438231110572815, "rewards/TotalKcalExactMatchReward/mean": 0.265625, "rewards/TotalKcalExactMatchReward/std": 0.4396721661090851, "step": 25 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.065625, "completions/max_length": 513.0, "completions/mean_length": 398.553125, "completions/min_length": 237.8, "epoch": 0.008344923504867872, "frac_reward_zero_std": 0.0, "grad_norm": 0.7440645694732666, "kl": 0.00041069387589232065, "learning_rate": 1.6666666666666665e-07, "loss": 1.6446597874164583e-05, "reward": 2.1207164525985718, "reward_std": 0.6690902352333069, "rewards/IngredientFormatReward/mean": 0.8505319833755494, "rewards/IngredientFormatReward/std": 0.3448954880237579, "rewards/IngredientMatchReward/mean": 0.49686943292617797, "rewards/IngredientMatchReward/std": 0.3237551271915436, "rewards/IngredientQuantityMatchReward/mean": 0.5108150601387024, "rewards/IngredientQuantityMatchReward/std": 0.45325875878334043, "rewards/TotalKcalExactMatchReward/mean": 0.2625, "rewards/TotalKcalExactMatchReward/std": 0.4397946834564209, "step": 30 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.084375, "completions/max_length": 513.0, "completions/mean_length": 404.7203125, "completions/min_length": 257.4, "epoch": 0.009735744089012517, "frac_reward_zero_std": 0.0, "grad_norm": 0.7201761603355408, "kl": 0.00041231723917007913, "learning_rate": 1.9444444444444445e-07, "loss": 1.6507774125784637e-05, "reward": 2.106699252128601, "reward_std": 0.7597081899642945, "rewards/IngredientFormatReward/mean": 0.8413802146911621, "rewards/IngredientFormatReward/std": 0.34706807136535645, "rewards/IngredientMatchReward/mean": 0.43627233505249025, "rewards/IngredientMatchReward/std": 0.2960783183574677, "rewards/IngredientQuantityMatchReward/mean": 0.5540466845035553, "rewards/IngredientQuantityMatchReward/std": 0.4453916549682617, "rewards/TotalKcalExactMatchReward/mean": 0.275, "rewards/TotalKcalExactMatchReward/std": 0.4379255294799805, "step": 35 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.084375, "completions/max_length": 513.0, "completions/mean_length": 402.0609375, "completions/min_length": 247.4, "epoch": 0.011126564673157162, "frac_reward_zero_std": 0.0, "grad_norm": 0.74068683385849, "kl": 0.00042258204157405996, "learning_rate": 2.222222222222222e-07, "loss": 1.691014040261507e-05, "reward": 2.0093788385391234, "reward_std": 0.7664336800575257, "rewards/IngredientFormatReward/mean": 0.8254426956176758, "rewards/IngredientFormatReward/std": 0.3664437234401703, "rewards/IngredientMatchReward/mean": 0.4059288203716278, "rewards/IngredientMatchReward/std": 0.28713943660259245, "rewards/IngredientQuantityMatchReward/mean": 0.5623823046684265, "rewards/IngredientQuantityMatchReward/std": 0.45745145678520205, "rewards/TotalKcalExactMatchReward/mean": 0.215625, "rewards/TotalKcalExactMatchReward/std": 0.40993704795837405, "step": 40 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0734375, "completions/max_length": 513.0, "completions/mean_length": 396.8671875, "completions/min_length": 220.4, "epoch": 0.012517385257301807, "frac_reward_zero_std": 0.0, "grad_norm": 0.7788537740707397, "kl": 0.0004178027720627142, "learning_rate": 2.5e-07, "loss": 1.670618075877428e-05, "reward": 1.9563416957855224, "reward_std": 0.6972125172615051, "rewards/IngredientFormatReward/mean": 0.822872006893158, "rewards/IngredientFormatReward/std": 0.3712240636348724, "rewards/IngredientMatchReward/mean": 0.44173184037208557, "rewards/IngredientMatchReward/std": 0.3116217702627182, "rewards/IngredientQuantityMatchReward/mean": 0.4729879081249237, "rewards/IngredientQuantityMatchReward/std": 0.45757158994674685, "rewards/TotalKcalExactMatchReward/mean": 0.21875, "rewards/TotalKcalExactMatchReward/std": 0.4113896369934082, "step": 45 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 513.0, "completions/mean_length": 398.4078125, "completions/min_length": 218.6, "epoch": 0.013908205841446454, "frac_reward_zero_std": 0.0, "grad_norm": 0.8167436122894287, "kl": 0.0004406511514389422, "learning_rate": 2.7777777777777776e-07, "loss": 1.766979112289846e-05, "reward": 2.054394245147705, "reward_std": 0.6666451692581177, "rewards/IngredientFormatReward/mean": 0.840610122680664, "rewards/IngredientFormatReward/std": 0.3461251139640808, "rewards/IngredientMatchReward/mean": 0.46625000834465025, "rewards/IngredientMatchReward/std": 0.3063514709472656, "rewards/IngredientQuantityMatchReward/mean": 0.5053465902805329, "rewards/IngredientQuantityMatchReward/std": 0.45414825081825255, "rewards/TotalKcalExactMatchReward/mean": 0.2421875, "rewards/TotalKcalExactMatchReward/std": 0.4245078921318054, "step": 50 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0515625, "completions/max_length": 513.0, "completions/mean_length": 388.9828125, "completions/min_length": 222.0, "epoch": 0.015299026425591099, "frac_reward_zero_std": 0.0, "grad_norm": 0.8138473629951477, "kl": 0.00042133941533393224, "learning_rate": 3.055555555555556e-07, "loss": 1.6886042430996895e-05, "reward": 2.1590343952178954, "reward_std": 0.6529965162277221, "rewards/IngredientFormatReward/mean": 0.8830208301544189, "rewards/IngredientFormatReward/std": 0.3005488455295563, "rewards/IngredientMatchReward/mean": 0.47440544366836546, "rewards/IngredientMatchReward/std": 0.30408360362052916, "rewards/IngredientQuantityMatchReward/mean": 0.5719206094741821, "rewards/IngredientQuantityMatchReward/std": 0.4585339307785034, "rewards/TotalKcalExactMatchReward/mean": 0.2296875, "rewards/TotalKcalExactMatchReward/std": 0.41347005367279055, "step": 55 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0796875, "completions/max_length": 513.0, "completions/mean_length": 398.1203125, "completions/min_length": 222.4, "epoch": 0.016689847009735744, "frac_reward_zero_std": 0.0, "grad_norm": 0.7178951501846313, "kl": 0.0004636022935301298, "learning_rate": 3.333333333333333e-07, "loss": 1.8570698739495128e-05, "reward": 2.014821100234985, "reward_std": 0.7525708317756653, "rewards/IngredientFormatReward/mean": 0.8299218893051148, "rewards/IngredientFormatReward/std": 0.36488322615623475, "rewards/IngredientMatchReward/mean": 0.41852368116378785, "rewards/IngredientMatchReward/std": 0.2902585655450821, "rewards/IngredientQuantityMatchReward/mean": 0.5179380416870117, "rewards/IngredientQuantityMatchReward/std": 0.45164860486984254, "rewards/TotalKcalExactMatchReward/mean": 0.2484375, "rewards/TotalKcalExactMatchReward/std": 0.42951870560646055, "step": 60 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 513.0, "completions/mean_length": 396.471875, "completions/min_length": 257.8, "epoch": 0.01808066759388039, "frac_reward_zero_std": 0.0, "grad_norm": 0.7024106383323669, "kl": 0.00044692111259792, "learning_rate": 3.6111111111111107e-07, "loss": 1.784815249266103e-05, "reward": 2.1446351766586305, "reward_std": 0.6606986165046692, "rewards/IngredientFormatReward/mean": 0.8456249952316284, "rewards/IngredientFormatReward/std": 0.34710876941680907, "rewards/IngredientMatchReward/mean": 0.4586080014705658, "rewards/IngredientMatchReward/std": 0.2785313993692398, "rewards/IngredientQuantityMatchReward/mean": 0.5747771084308624, "rewards/IngredientQuantityMatchReward/std": 0.44896321892738345, "rewards/TotalKcalExactMatchReward/mean": 0.265625, "rewards/TotalKcalExactMatchReward/std": 0.44074881076812744, "step": 65 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.040625, "completions/max_length": 513.0, "completions/mean_length": 394.8890625, "completions/min_length": 239.4, "epoch": 0.019471488178025034, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7436627745628357, "kl": 0.0004441993540240219, "learning_rate": 3.888888888888889e-07, "loss": 1.7779297195374966e-05, "reward": 2.118137836456299, "reward_std": 0.7043254613876343, "rewards/IngredientFormatReward/mean": 0.8741369247436523, "rewards/IngredientFormatReward/std": 0.30888078510761263, "rewards/IngredientMatchReward/mean": 0.4107558369636536, "rewards/IngredientMatchReward/std": 0.28572845458984375, "rewards/IngredientQuantityMatchReward/mean": 0.5613700985908509, "rewards/IngredientQuantityMatchReward/std": 0.4577866315841675, "rewards/TotalKcalExactMatchReward/mean": 0.271875, "rewards/TotalKcalExactMatchReward/std": 0.44533634185791016, "step": 70 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.065625, "completions/max_length": 513.0, "completions/mean_length": 396.0078125, "completions/min_length": 223.4, "epoch": 0.02086230876216968, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7756255865097046, "kl": 0.0005053242901340127, "learning_rate": 4.1666666666666667e-07, "loss": 2.017030492424965e-05, "reward": 2.087790060043335, "reward_std": 0.711814534664154, "rewards/IngredientFormatReward/mean": 0.8524739623069764, "rewards/IngredientFormatReward/std": 0.3421242475509644, "rewards/IngredientMatchReward/mean": 0.46218688488006593, "rewards/IngredientMatchReward/std": 0.30402821898460386, "rewards/IngredientQuantityMatchReward/mean": 0.5246916890144349, "rewards/IngredientQuantityMatchReward/std": 0.4510571718215942, "rewards/TotalKcalExactMatchReward/mean": 0.2484375, "rewards/TotalKcalExactMatchReward/std": 0.43120152950286866, "step": 75 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0765625, "completions/max_length": 513.0, "completions/mean_length": 400.8671875, "completions/min_length": 235.0, "epoch": 0.022253129346314324, "frac_reward_zero_std": 0.0, "grad_norm": 0.770311713218689, "kl": 0.0005777321712230332, "learning_rate": 4.444444444444444e-07, "loss": 2.3143680300563575e-05, "reward": 2.0575687885284424, "reward_std": 0.77638738155365, "rewards/IngredientFormatReward/mean": 0.8024814009666443, "rewards/IngredientFormatReward/std": 0.38416465520858767, "rewards/IngredientMatchReward/mean": 0.4254823923110962, "rewards/IngredientMatchReward/std": 0.3045427978038788, "rewards/IngredientQuantityMatchReward/mean": 0.563979983329773, "rewards/IngredientQuantityMatchReward/std": 0.4503722727298737, "rewards/TotalKcalExactMatchReward/mean": 0.265625, "rewards/TotalKcalExactMatchReward/std": 0.4396756887435913, "step": 80 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0796875, "completions/max_length": 513.0, "completions/mean_length": 408.1390625, "completions/min_length": 238.0, "epoch": 0.02364394993045897, "frac_reward_zero_std": 0.0, "grad_norm": 0.7720329761505127, "kl": 0.0006533111230965005, "learning_rate": 4.722222222222222e-07, "loss": 2.6147160679101943e-05, "reward": 2.0888258457183837, "reward_std": 0.7742647767066956, "rewards/IngredientFormatReward/mean": 0.8335677027702332, "rewards/IngredientFormatReward/std": 0.36205244064331055, "rewards/IngredientMatchReward/mean": 0.47457787990570066, "rewards/IngredientMatchReward/std": 0.3164846241474152, "rewards/IngredientQuantityMatchReward/mean": 0.4963052570819855, "rewards/IngredientQuantityMatchReward/std": 0.4384254515171051, "rewards/TotalKcalExactMatchReward/mean": 0.284375, "rewards/TotalKcalExactMatchReward/std": 0.45113744139671325, "step": 85 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0578125, "completions/max_length": 513.0, "completions/mean_length": 389.85, "completions/min_length": 225.6, "epoch": 0.025034770514603615, "frac_reward_zero_std": 0.0, "grad_norm": 0.7358577251434326, "kl": 0.0007881541740061948, "learning_rate": 5e-07, "loss": 3.152899444103241e-05, "reward": 2.196377229690552, "reward_std": 0.6824372291564942, "rewards/IngredientFormatReward/mean": 0.8830841064453125, "rewards/IngredientFormatReward/std": 0.3072942852973938, "rewards/IngredientMatchReward/mean": 0.4858600676059723, "rewards/IngredientMatchReward/std": 0.3125761985778809, "rewards/IngredientQuantityMatchReward/mean": 0.5524331212043763, "rewards/IngredientQuantityMatchReward/std": 0.44864551424980165, "rewards/TotalKcalExactMatchReward/mean": 0.275, "rewards/TotalKcalExactMatchReward/std": 0.4422182202339172, "step": 90 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 513.0, "completions/mean_length": 393.0734375, "completions/min_length": 249.6, "epoch": 0.02642559109874826, "frac_reward_zero_std": 0.0, "grad_norm": 0.8563322424888611, "kl": 0.0014499979191896274, "learning_rate": 5.277777777777777e-07, "loss": 5.795812467113137e-05, "reward": 2.0926523447036742, "reward_std": 0.6976558685302734, "rewards/IngredientFormatReward/mean": 0.8504129528999329, "rewards/IngredientFormatReward/std": 0.3291980028152466, "rewards/IngredientMatchReward/mean": 0.47219123840332033, "rewards/IngredientMatchReward/std": 0.2914651095867157, "rewards/IngredientQuantityMatchReward/mean": 0.5294231832027435, "rewards/IngredientQuantityMatchReward/std": 0.4433592915534973, "rewards/TotalKcalExactMatchReward/mean": 0.240625, "rewards/TotalKcalExactMatchReward/std": 0.427162367105484, "step": 95 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.096875, "completions/max_length": 513.0, "completions/mean_length": 407.1578125, "completions/min_length": 259.8, "epoch": 0.027816411682892908, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7285707592964172, "kl": 0.0023988128137716557, "learning_rate": 5.555555555555555e-07, "loss": 9.596748277544976e-05, "reward": 2.165234994888306, "reward_std": 0.7257884621620179, "rewards/IngredientFormatReward/mean": 0.8481119632720947, "rewards/IngredientFormatReward/std": 0.3413813829421997, "rewards/IngredientMatchReward/mean": 0.4407334983348846, "rewards/IngredientMatchReward/std": 0.29101337790489196, "rewards/IngredientQuantityMatchReward/mean": 0.5388895452022553, "rewards/IngredientQuantityMatchReward/std": 0.4558054804801941, "rewards/TotalKcalExactMatchReward/mean": 0.3375, "rewards/TotalKcalExactMatchReward/std": 0.4702342689037323, "step": 100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.08125, "completions/max_length": 513.0, "completions/mean_length": 406.5203125, "completions/min_length": 240.4, "epoch": 0.02920723226703755, "frac_reward_zero_std": 0.0, "grad_norm": 0.7217776775360107, "kl": 0.0016849263662152224, "learning_rate": 5.833333333333334e-07, "loss": 6.740582175552845e-05, "reward": 2.108404779434204, "reward_std": 0.7473301291465759, "rewards/IngredientFormatReward/mean": 0.8337239503860474, "rewards/IngredientFormatReward/std": 0.36305957436561587, "rewards/IngredientMatchReward/mean": 0.43829986453056335, "rewards/IngredientMatchReward/std": 0.2977118492126465, "rewards/IngredientQuantityMatchReward/mean": 0.520756047964096, "rewards/IngredientQuantityMatchReward/std": 0.44096989631652833, "rewards/TotalKcalExactMatchReward/mean": 0.315625, "rewards/TotalKcalExactMatchReward/std": 0.4639370679855347, "step": 105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0671875, "completions/max_length": 513.0, "completions/mean_length": 401.4265625, "completions/min_length": 245.2, "epoch": 0.030598052851182198, "frac_reward_zero_std": 0.0, "grad_norm": 0.736385703086853, "kl": 0.003075429912132677, "learning_rate": 6.111111111111112e-07, "loss": 0.00012296217028051615, "reward": 2.197610855102539, "reward_std": 0.7173490405082703, "rewards/IngredientFormatReward/mean": 0.8491666674613952, "rewards/IngredientFormatReward/std": 0.3398062169551849, "rewards/IngredientMatchReward/mean": 0.41518911719322205, "rewards/IngredientMatchReward/std": 0.2829656690359116, "rewards/IngredientQuantityMatchReward/mean": 0.5645050764083862, "rewards/IngredientQuantityMatchReward/std": 0.45311381816864016, "rewards/TotalKcalExactMatchReward/mean": 0.36875, "rewards/TotalKcalExactMatchReward/std": 0.4792756617069244, "step": 110 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0640625, "completions/max_length": 513.0, "completions/mean_length": 394.53125, "completions/min_length": 230.2, "epoch": 0.031988873435326845, "frac_reward_zero_std": 0.0, "grad_norm": 0.7664805054664612, "kl": 0.004706994362641126, "learning_rate": 6.388888888888888e-07, "loss": 0.00018819000106304884, "reward": 2.175532102584839, "reward_std": 0.7320732951164246, "rewards/IngredientFormatReward/mean": 0.8788281440734863, "rewards/IngredientFormatReward/std": 0.3098617732524872, "rewards/IngredientMatchReward/mean": 0.4196726202964783, "rewards/IngredientMatchReward/std": 0.2729521691799164, "rewards/IngredientQuantityMatchReward/mean": 0.5520314157009125, "rewards/IngredientQuantityMatchReward/std": 0.4562692940235138, "rewards/TotalKcalExactMatchReward/mean": 0.325, "rewards/TotalKcalExactMatchReward/std": 0.4657438635826111, "step": 115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 513.0, "completions/mean_length": 394.9921875, "completions/min_length": 246.6, "epoch": 0.03337969401947149, "frac_reward_zero_std": 0.0, "grad_norm": 0.8188135623931885, "kl": 0.004332279106165515, "learning_rate": 6.666666666666666e-07, "loss": 0.00017332370625808836, "reward": 2.2424280643463135, "reward_std": 0.6947030186653137, "rewards/IngredientFormatReward/mean": 0.8772433042526245, "rewards/IngredientFormatReward/std": 0.3062496542930603, "rewards/IngredientMatchReward/mean": 0.45630409717559817, "rewards/IngredientMatchReward/std": 0.3024283528327942, "rewards/IngredientQuantityMatchReward/mean": 0.5870055854320526, "rewards/IngredientQuantityMatchReward/std": 0.42272945046424865, "rewards/TotalKcalExactMatchReward/mean": 0.321875, "rewards/TotalKcalExactMatchReward/std": 0.4648774802684784, "step": 120 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.040625, "completions/max_length": 513.0, "completions/mean_length": 390.7203125, "completions/min_length": 249.4, "epoch": 0.03477051460361613, "frac_reward_zero_std": 0.0125, "grad_norm": 0.816830575466156, "kl": 0.003736836545431288, "learning_rate": 6.944444444444444e-07, "loss": 0.0001495170406997204, "reward": 2.2428334236145018, "reward_std": 0.638411796092987, "rewards/IngredientFormatReward/mean": 0.9001041531562806, "rewards/IngredientFormatReward/std": 0.2765121221542358, "rewards/IngredientMatchReward/mean": 0.4665631353855133, "rewards/IngredientMatchReward/std": 0.30025785565376284, "rewards/IngredientQuantityMatchReward/mean": 0.5074161112308502, "rewards/IngredientQuantityMatchReward/std": 0.44353713989257815, "rewards/TotalKcalExactMatchReward/mean": 0.36875, "rewards/TotalKcalExactMatchReward/std": 0.475113707780838, "step": 125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 513.0, "completions/mean_length": 397.80625, "completions/min_length": 251.8, "epoch": 0.03616133518776078, "frac_reward_zero_std": 0.0, "grad_norm": 0.8476716876029968, "kl": 0.004359716525505064, "learning_rate": 7.222222222222221e-07, "loss": 0.00017424174584448337, "reward": 2.316778326034546, "reward_std": 0.6683953762054443, "rewards/IngredientFormatReward/mean": 0.9030468583106994, "rewards/IngredientFormatReward/std": 0.2746707767248154, "rewards/IngredientMatchReward/mean": 0.44811282157897947, "rewards/IngredientMatchReward/std": 0.28702212870121, "rewards/IngredientQuantityMatchReward/mean": 0.5578061103820801, "rewards/IngredientQuantityMatchReward/std": 0.44357584714889525, "rewards/TotalKcalExactMatchReward/mean": 0.4078125, "rewards/TotalKcalExactMatchReward/std": 0.49025474190711976, "step": 130 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.075, "completions/max_length": 513.0, "completions/mean_length": 401.109375, "completions/min_length": 244.4, "epoch": 0.037552155771905425, "frac_reward_zero_std": 0.0, "grad_norm": 0.774059534072876, "kl": 0.005080446207284694, "learning_rate": 7.5e-07, "loss": 0.00020329309627413749, "reward": 2.2883750915527346, "reward_std": 0.7694447875022888, "rewards/IngredientFormatReward/mean": 0.867638885974884, "rewards/IngredientFormatReward/std": 0.31823705434799193, "rewards/IngredientMatchReward/mean": 0.45627238750457766, "rewards/IngredientMatchReward/std": 0.30123019218444824, "rewards/IngredientQuantityMatchReward/mean": 0.5004013001918792, "rewards/IngredientQuantityMatchReward/std": 0.4554529309272766, "rewards/TotalKcalExactMatchReward/mean": 0.4640625, "rewards/TotalKcalExactMatchReward/std": 0.4989950478076935, "step": 135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 513.0, "completions/mean_length": 399.2828125, "completions/min_length": 229.2, "epoch": 0.03894297635605007, "frac_reward_zero_std": 0.0, "grad_norm": 0.7574522495269775, "kl": 0.006042543252988253, "learning_rate": 7.777777777777778e-07, "loss": 0.000241729780100286, "reward": 2.141042113304138, "reward_std": 0.749863576889038, "rewards/IngredientFormatReward/mean": 0.83494793176651, "rewards/IngredientFormatReward/std": 0.340142023563385, "rewards/IngredientMatchReward/mean": 0.4368985652923584, "rewards/IngredientMatchReward/std": 0.3157567739486694, "rewards/IngredientQuantityMatchReward/mean": 0.4723206400871277, "rewards/IngredientQuantityMatchReward/std": 0.4440523386001587, "rewards/TotalKcalExactMatchReward/mean": 0.396875, "rewards/TotalKcalExactMatchReward/std": 0.47984519600868225, "step": 140 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0609375, "completions/max_length": 513.0, "completions/mean_length": 403.6734375, "completions/min_length": 246.8, "epoch": 0.04033379694019471, "frac_reward_zero_std": 0.0, "grad_norm": 0.7979903817176819, "kl": 0.004112960568454583, "learning_rate": 8.055555555555556e-07, "loss": 0.00016456012381240725, "reward": 2.3432874202728273, "reward_std": 0.7342286825180053, "rewards/IngredientFormatReward/mean": 0.8814843773841858, "rewards/IngredientFormatReward/std": 0.31216220259666444, "rewards/IngredientMatchReward/mean": 0.4461422383785248, "rewards/IngredientMatchReward/std": 0.3000922918319702, "rewards/IngredientQuantityMatchReward/mean": 0.5172233939170837, "rewards/IngredientQuantityMatchReward/std": 0.4469065308570862, "rewards/TotalKcalExactMatchReward/mean": 0.4984375, "rewards/TotalKcalExactMatchReward/std": 0.497554749250412, "step": 145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0671875, "completions/max_length": 513.0, "completions/mean_length": 394.4578125, "completions/min_length": 206.8, "epoch": 0.04172461752433936, "frac_reward_zero_std": 0.0, "grad_norm": 0.7423848509788513, "kl": 0.004441153233346995, "learning_rate": 8.333333333333333e-07, "loss": 0.0001776395714841783, "reward": 2.220476579666138, "reward_std": 0.7300424218177796, "rewards/IngredientFormatReward/mean": 0.8623697757720947, "rewards/IngredientFormatReward/std": 0.31136985719203947, "rewards/IngredientMatchReward/mean": 0.4273505866527557, "rewards/IngredientMatchReward/std": 0.30395570397377014, "rewards/IngredientQuantityMatchReward/mean": 0.4557562291622162, "rewards/IngredientQuantityMatchReward/std": 0.4484573841094971, "rewards/TotalKcalExactMatchReward/mean": 0.475, "rewards/TotalKcalExactMatchReward/std": 0.49919587969779966, "step": 150 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05, "completions/max_length": 513.0, "completions/mean_length": 397.7890625, "completions/min_length": 233.8, "epoch": 0.043115438108484005, "frac_reward_zero_std": 0.0, "grad_norm": 0.7154651880264282, "kl": 0.004370697992271744, "learning_rate": 8.611111111111111e-07, "loss": 0.0001748448354192078, "reward": 2.4821091651916505, "reward_std": 0.7230484843254089, "rewards/IngredientFormatReward/mean": 0.908965790271759, "rewards/IngredientFormatReward/std": 0.27234546542167665, "rewards/IngredientMatchReward/mean": 0.480882340669632, "rewards/IngredientMatchReward/std": 0.29070753455162046, "rewards/IngredientQuantityMatchReward/mean": 0.5750735938549042, "rewards/IngredientQuantityMatchReward/std": 0.4402975857257843, "rewards/TotalKcalExactMatchReward/mean": 0.5171875, "rewards/TotalKcalExactMatchReward/std": 0.49578348994255067, "step": 155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0796875, "completions/max_length": 513.0, "completions/mean_length": 403.9, "completions/min_length": 233.2, "epoch": 0.04450625869262865, "frac_reward_zero_std": 0.0125, "grad_norm": 0.8095347285270691, "kl": 0.006063173009897583, "learning_rate": 8.888888888888888e-07, "loss": 0.00024237227626144886, "reward": 2.404101753234863, "reward_std": 0.7797430753707886, "rewards/IngredientFormatReward/mean": 0.8703645825386047, "rewards/IngredientFormatReward/std": 0.31925941705703736, "rewards/IngredientMatchReward/mean": 0.4635703802108765, "rewards/IngredientMatchReward/std": 0.3105102479457855, "rewards/IngredientQuantityMatchReward/mean": 0.5482918679714203, "rewards/IngredientQuantityMatchReward/std": 0.44624740481376646, "rewards/TotalKcalExactMatchReward/mean": 0.521875, "rewards/TotalKcalExactMatchReward/std": 0.49667310118675234, "step": 160 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05625, "completions/max_length": 513.0, "completions/mean_length": 401.25625, "completions/min_length": 241.0, "epoch": 0.0458970792767733, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7412997484207153, "kl": 0.006450092940940522, "learning_rate": 9.166666666666665e-07, "loss": 0.0002578888554126024, "reward": 2.505335474014282, "reward_std": 0.7264118909835815, "rewards/IngredientFormatReward/mean": 0.9001525521278382, "rewards/IngredientFormatReward/std": 0.2870828151702881, "rewards/IngredientMatchReward/mean": 0.4542801439762115, "rewards/IngredientMatchReward/std": 0.2940641105175018, "rewards/IngredientQuantityMatchReward/mean": 0.6071528434753418, "rewards/IngredientQuantityMatchReward/std": 0.44424931406974794, "rewards/TotalKcalExactMatchReward/mean": 0.54375, "rewards/TotalKcalExactMatchReward/std": 0.4922543168067932, "step": 165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0484375, "completions/max_length": 513.0, "completions/mean_length": 391.915625, "completions/min_length": 234.8, "epoch": 0.04728789986091794, "frac_reward_zero_std": 0.0125, "grad_norm": 0.8398517966270447, "kl": 0.006277643307112157, "learning_rate": 9.444444444444444e-07, "loss": 0.000251129362732172, "reward": 2.5534721851348876, "reward_std": 0.7245315790176392, "rewards/IngredientFormatReward/mean": 0.9078869223594666, "rewards/IngredientFormatReward/std": 0.2684895843267441, "rewards/IngredientMatchReward/mean": 0.48150365352630614, "rewards/IngredientMatchReward/std": 0.31550759077072144, "rewards/IngredientQuantityMatchReward/mean": 0.5640815615653991, "rewards/IngredientQuantityMatchReward/std": 0.42716373801231383, "rewards/TotalKcalExactMatchReward/mean": 0.6, "rewards/TotalKcalExactMatchReward/std": 0.48972519040107726, "step": 170 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05, "completions/max_length": 513.0, "completions/mean_length": 386.38125, "completions/min_length": 216.6, "epoch": 0.048678720445062586, "frac_reward_zero_std": 0.0, "grad_norm": 0.7972651720046997, "kl": 0.007527178205782548, "learning_rate": 9.722222222222222e-07, "loss": 0.00030102948658168317, "reward": 2.422546720504761, "reward_std": 0.6932518959045411, "rewards/IngredientFormatReward/mean": 0.9198958396911621, "rewards/IngredientFormatReward/std": 0.24611102044582367, "rewards/IngredientMatchReward/mean": 0.4539143145084381, "rewards/IngredientMatchReward/std": 0.30238354206085205, "rewards/IngredientQuantityMatchReward/mean": 0.473736572265625, "rewards/IngredientQuantityMatchReward/std": 0.44805272221565245, "rewards/TotalKcalExactMatchReward/mean": 0.575, "rewards/TotalKcalExactMatchReward/std": 0.48392468094825747, "step": 175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 513.0, "completions/mean_length": 391.946875, "completions/min_length": 210.4, "epoch": 0.05006954102920723, "frac_reward_zero_std": 0.025, "grad_norm": 0.7244554758071899, "kl": 0.006808792043011635, "learning_rate": 1e-06, "loss": 0.00027234144508838656, "reward": 2.585988426208496, "reward_std": 0.6814924240112304, "rewards/IngredientFormatReward/mean": 0.8969047546386719, "rewards/IngredientFormatReward/std": 0.28515351116657256, "rewards/IngredientMatchReward/mean": 0.4871840238571167, "rewards/IngredientMatchReward/std": 0.30355110168457033, "rewards/IngredientQuantityMatchReward/mean": 0.5706496775150299, "rewards/IngredientQuantityMatchReward/std": 0.4206697165966034, "rewards/TotalKcalExactMatchReward/mean": 0.63125, "rewards/TotalKcalExactMatchReward/std": 0.47843066453933714, "step": 180 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 513.0, "completions/mean_length": 394.35, "completions/min_length": 258.6, "epoch": 0.05146036161335188, "frac_reward_zero_std": 0.0, "grad_norm": 0.7314364910125732, "kl": 0.007314420084003359, "learning_rate": 9.999947107075408e-07, "loss": 0.00029256632551550866, "reward": 2.662673282623291, "reward_std": 0.5554831802845002, "rewards/IngredientFormatReward/mean": 0.9673437595367431, "rewards/IngredientFormatReward/std": 0.15578847080469133, "rewards/IngredientMatchReward/mean": 0.5206299781799316, "rewards/IngredientMatchReward/std": 0.3142448902130127, "rewards/IngredientQuantityMatchReward/mean": 0.5731370329856873, "rewards/IngredientQuantityMatchReward/std": 0.43153069615364076, "rewards/TotalKcalExactMatchReward/mean": 0.6015625, "rewards/TotalKcalExactMatchReward/std": 0.4841706693172455, "step": 185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 513.0, "completions/mean_length": 385.6, "completions/min_length": 228.6, "epoch": 0.05285118219749652, "frac_reward_zero_std": 0.0, "grad_norm": 0.7836043238639832, "kl": 0.008130288516986184, "learning_rate": 9.999788429420697e-07, "loss": 0.0003252896945923567, "reward": 2.6136358737945558, "reward_std": 0.6777990698814392, "rewards/IngredientFormatReward/mean": 0.9325000047683716, "rewards/IngredientFormatReward/std": 0.2264118403196335, "rewards/IngredientMatchReward/mean": 0.48915613889694215, "rewards/IngredientMatchReward/std": 0.30804571509361267, "rewards/IngredientQuantityMatchReward/mean": 0.5607298374176025, "rewards/IngredientQuantityMatchReward/std": 0.4353516399860382, "rewards/TotalKcalExactMatchReward/mean": 0.63125, "rewards/TotalKcalExactMatchReward/std": 0.47783060669898986, "step": 190 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 513.0, "completions/mean_length": 400.6265625, "completions/min_length": 265.0, "epoch": 0.054242002781641166, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7703161239624023, "kl": 0.010244119248818606, "learning_rate": 9.999523970393036e-07, "loss": 0.0004097242839634418, "reward": 2.4793437004089354, "reward_std": 0.6277800083160401, "rewards/IngredientFormatReward/mean": 0.9137834787368775, "rewards/IngredientFormatReward/std": 0.25849959552288054, "rewards/IngredientMatchReward/mean": 0.45915488600730897, "rewards/IngredientMatchReward/std": 0.2793452024459839, "rewards/IngredientQuantityMatchReward/mean": 0.5282802939414978, "rewards/IngredientQuantityMatchReward/std": 0.45662105083465576, "rewards/TotalKcalExactMatchReward/mean": 0.578125, "rewards/TotalKcalExactMatchReward/std": 0.49128650426864623, "step": 195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0453125, "completions/max_length": 513.0, "completions/mean_length": 395.09375, "completions/min_length": 236.0, "epoch": 0.055632823365785816, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7537059783935547, "kl": 0.009191960797761568, "learning_rate": 9.999153735587632e-07, "loss": 0.00036758126225322485, "reward": 2.4740325927734377, "reward_std": 0.6981268048286438, "rewards/IngredientFormatReward/mean": 0.9252864718437195, "rewards/IngredientFormatReward/std": 0.24678438305854797, "rewards/IngredientMatchReward/mean": 0.45877419114112855, "rewards/IngredientMatchReward/std": 0.29859951734542844, "rewards/IngredientQuantityMatchReward/mean": 0.46965948939323426, "rewards/IngredientQuantityMatchReward/std": 0.44375336170196533, "rewards/TotalKcalExactMatchReward/mean": 0.6203125, "rewards/TotalKcalExactMatchReward/std": 0.4775801539421082, "step": 200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0421875, "completions/max_length": 513.0, "completions/mean_length": 396.365625, "completions/min_length": 241.8, "epoch": 0.05702364394993046, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7670131325721741, "kl": 0.008866281888913363, "learning_rate": 9.9986777328376e-07, "loss": 0.00035456912592053414, "reward": 2.679036331176758, "reward_std": 0.57293900847435, "rewards/IngredientFormatReward/mean": 0.9371874928474426, "rewards/IngredientFormatReward/std": 0.23141905665397644, "rewards/IngredientMatchReward/mean": 0.5184425234794616, "rewards/IngredientMatchReward/std": 0.30044159293174744, "rewards/IngredientQuantityMatchReward/mean": 0.5343438982963562, "rewards/IngredientQuantityMatchReward/std": 0.450323224067688, "rewards/TotalKcalExactMatchReward/mean": 0.6890625, "rewards/TotalKcalExactMatchReward/std": 0.4618146061897278, "step": 205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0609375, "completions/max_length": 512.8, "completions/mean_length": 397.0625, "completions/min_length": 240.8, "epoch": 0.0584144645340751, "frac_reward_zero_std": 0.025, "grad_norm": 0.7266873121261597, "kl": 0.009309542400296777, "learning_rate": 9.998095972213817e-07, "loss": 0.0003722626715898514, "reward": 2.597199583053589, "reward_std": 0.6256489753723145, "rewards/IngredientFormatReward/mean": 0.91268230676651, "rewards/IngredientFormatReward/std": 0.25630738735198977, "rewards/IngredientMatchReward/mean": 0.47620717287063596, "rewards/IngredientMatchReward/std": 0.30730987787246705, "rewards/IngredientQuantityMatchReward/mean": 0.5395601093769073, "rewards/IngredientQuantityMatchReward/std": 0.44395468235015867, "rewards/TotalKcalExactMatchReward/mean": 0.66875, "rewards/TotalKcalExactMatchReward/std": 0.4664685130119324, "step": 210 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.059375, "completions/max_length": 513.0, "completions/mean_length": 397.6953125, "completions/min_length": 250.6, "epoch": 0.059805285118219746, "frac_reward_zero_std": 0.05, "grad_norm": 0.7109785079956055, "kl": 0.009796890115831047, "learning_rate": 9.99740846602469e-07, "loss": 0.00039182775653898714, "reward": 2.7128491401672363, "reward_std": 0.5502410054206848, "rewards/IngredientFormatReward/mean": 0.9221875071525574, "rewards/IngredientFormatReward/std": 0.2499729573726654, "rewards/IngredientMatchReward/mean": 0.49059213399887086, "rewards/IngredientMatchReward/std": 0.3133565247058868, "rewards/IngredientQuantityMatchReward/mean": 0.625069534778595, "rewards/IngredientQuantityMatchReward/std": 0.4262022256851196, "rewards/TotalKcalExactMatchReward/mean": 0.675, "rewards/TotalKcalExactMatchReward/std": 0.4658839821815491, "step": 215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 513.0, "completions/mean_length": 377.0, "completions/min_length": 219.4, "epoch": 0.061196105702364396, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7904135584831238, "kl": 0.00917984291445464, "learning_rate": 9.996615228815905e-07, "loss": 0.0003672625869512558, "reward": 2.5808473587036134, "reward_std": 0.6128181099891663, "rewards/IngredientFormatReward/mean": 0.9564843654632569, "rewards/IngredientFormatReward/std": 0.18436979204416276, "rewards/IngredientMatchReward/mean": 0.44942651987075805, "rewards/IngredientMatchReward/std": 0.3052241802215576, "rewards/IngredientQuantityMatchReward/mean": 0.5499364614486695, "rewards/IngredientQuantityMatchReward/std": 0.45513352155685427, "rewards/TotalKcalExactMatchReward/mean": 0.625, "rewards/TotalKcalExactMatchReward/std": 0.47820115089416504, "step": 220 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0359375, "completions/max_length": 513.0, "completions/mean_length": 392.703125, "completions/min_length": 261.0, "epoch": 0.06258692628650904, "frac_reward_zero_std": 0.0125, "grad_norm": 0.8294149041175842, "kl": 0.00947985349339433, "learning_rate": 9.995716277370112e-07, "loss": 0.00037920954637229445, "reward": 2.6771074295043946, "reward_std": 0.6180533349514008, "rewards/IngredientFormatReward/mean": 0.9501804351806641, "rewards/IngredientFormatReward/std": 0.19291509985923766, "rewards/IngredientMatchReward/mean": 0.5157575666904449, "rewards/IngredientMatchReward/std": 0.32128278613090516, "rewards/IngredientQuantityMatchReward/mean": 0.5752319931983948, "rewards/IngredientQuantityMatchReward/std": 0.4331511855125427, "rewards/TotalKcalExactMatchReward/mean": 0.6359375, "rewards/TotalKcalExactMatchReward/std": 0.48184582591056824, "step": 225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0515625, "completions/max_length": 513.0, "completions/mean_length": 396.38125, "completions/min_length": 256.2, "epoch": 0.06397774687065369, "frac_reward_zero_std": 0.0, "grad_norm": 0.8006162643432617, "kl": 0.013353917776839808, "learning_rate": 9.994711630706585e-07, "loss": 0.0005340860225260258, "reward": 2.593950796127319, "reward_std": 0.6663724660873414, "rewards/IngredientFormatReward/mean": 0.9172916650772095, "rewards/IngredientFormatReward/std": 0.2557332932949066, "rewards/IngredientMatchReward/mean": 0.5012678146362305, "rewards/IngredientMatchReward/std": 0.3088660776615143, "rewards/IngredientQuantityMatchReward/mean": 0.5988288640975952, "rewards/IngredientQuantityMatchReward/std": 0.43793771862983705, "rewards/TotalKcalExactMatchReward/mean": 0.5765625, "rewards/TotalKcalExactMatchReward/std": 0.4930992126464844, "step": 230 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 509.8, "completions/mean_length": 388.740625, "completions/min_length": 247.4, "epoch": 0.06536856745479833, "frac_reward_zero_std": 0.0, "grad_norm": 0.728783905506134, "kl": 0.010416155745042488, "learning_rate": 9.993601310080799e-07, "loss": 0.0004166138358414173, "reward": 2.6682600498199465, "reward_std": 0.5333548069000245, "rewards/IngredientFormatReward/mean": 0.9703906297683715, "rewards/IngredientFormatReward/std": 0.145054429769516, "rewards/IngredientMatchReward/mean": 0.521271800994873, "rewards/IngredientMatchReward/std": 0.2949683487415314, "rewards/IngredientQuantityMatchReward/mean": 0.5734726548194885, "rewards/IngredientQuantityMatchReward/std": 0.44596282243728635, "rewards/TotalKcalExactMatchReward/mean": 0.603125, "rewards/TotalKcalExactMatchReward/std": 0.4815952718257904, "step": 235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 513.0, "completions/mean_length": 387.5265625, "completions/min_length": 242.2, "epoch": 0.06675938803894298, "frac_reward_zero_std": 0.0, "grad_norm": 0.7192296385765076, "kl": 0.010648410249268636, "learning_rate": 9.992385338983998e-07, "loss": 0.00042591681703925134, "reward": 2.7542993068695067, "reward_std": 0.5644841074943543, "rewards/IngredientFormatReward/mean": 0.9609635472297668, "rewards/IngredientFormatReward/std": 0.17166336327791215, "rewards/IngredientMatchReward/mean": 0.5329110741615295, "rewards/IngredientMatchReward/std": 0.2976407825946808, "rewards/IngredientQuantityMatchReward/mean": 0.593237179517746, "rewards/IngredientQuantityMatchReward/std": 0.4195017755031586, "rewards/TotalKcalExactMatchReward/mean": 0.6671875, "rewards/TotalKcalExactMatchReward/std": 0.4707712590694427, "step": 240 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0625, "completions/max_length": 513.0, "completions/mean_length": 396.48125, "completions/min_length": 243.2, "epoch": 0.06815020862308763, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7437554597854614, "kl": 0.014695871755247936, "learning_rate": 9.99106374314269e-07, "loss": 0.0005878821946680545, "reward": 2.5849650382995604, "reward_std": 0.6724656701087952, "rewards/IngredientFormatReward/mean": 0.9083029747009277, "rewards/IngredientFormatReward/std": 0.27130608856678007, "rewards/IngredientMatchReward/mean": 0.5053019642829895, "rewards/IngredientMatchReward/std": 0.30541407465934756, "rewards/IngredientQuantityMatchReward/mean": 0.585422694683075, "rewards/IngredientQuantityMatchReward/std": 0.43134939670562744, "rewards/TotalKcalExactMatchReward/mean": 0.5859375, "rewards/TotalKcalExactMatchReward/std": 0.48427985310554506, "step": 245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04375, "completions/max_length": 513.0, "completions/mean_length": 393.321875, "completions/min_length": 228.0, "epoch": 0.06954102920723226, "frac_reward_zero_std": 0.0, "grad_norm": 0.6927481889724731, "kl": 0.012796262564370409, "learning_rate": 9.989636550518103e-07, "loss": 0.0005117684602737426, "reward": 2.5995935440063476, "reward_std": 0.6334937334060669, "rewards/IngredientFormatReward/mean": 0.9379650235176087, "rewards/IngredientFormatReward/std": 0.22491171956062317, "rewards/IngredientMatchReward/mean": 0.5036893606185913, "rewards/IngredientMatchReward/std": 0.3012221992015839, "rewards/IngredientQuantityMatchReward/mean": 0.5235642194747925, "rewards/IngredientQuantityMatchReward/std": 0.43460506200790405, "rewards/TotalKcalExactMatchReward/mean": 0.634375, "rewards/TotalKcalExactMatchReward/std": 0.4736035823822021, "step": 250 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 510.8, "completions/mean_length": 385.353125, "completions/min_length": 243.6, "epoch": 0.07093184979137691, "frac_reward_zero_std": 0.025, "grad_norm": 0.7734192609786987, "kl": 0.012177031306782738, "learning_rate": 9.988103791305593e-07, "loss": 0.0004871044307947159, "reward": 2.7546799659729, "reward_std": 0.5723294258117676, "rewards/IngredientFormatReward/mean": 0.9548437476158143, "rewards/IngredientFormatReward/std": 0.1730358988046646, "rewards/IngredientMatchReward/mean": 0.5104142010211945, "rewards/IngredientMatchReward/std": 0.3049011051654816, "rewards/IngredientQuantityMatchReward/mean": 0.6144220471382141, "rewards/IngredientQuantityMatchReward/std": 0.4470917284488678, "rewards/TotalKcalExactMatchReward/mean": 0.675, "rewards/TotalKcalExactMatchReward/std": 0.4676419675350189, "step": 255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.053125, "completions/max_length": 513.0, "completions/mean_length": 394.6890625, "completions/min_length": 265.2, "epoch": 0.07232267037552156, "frac_reward_zero_std": 0.0, "grad_norm": 0.6882739067077637, "kl": 0.013909409387269988, "learning_rate": 9.986465497934006e-07, "loss": 0.0005562674719840288, "reward": 2.643060731887817, "reward_std": 0.60300612449646, "rewards/IngredientFormatReward/mean": 0.9370833277702332, "rewards/IngredientFormatReward/std": 0.2204873889684677, "rewards/IngredientMatchReward/mean": 0.4994661629199982, "rewards/IngredientMatchReward/std": 0.28945026993751527, "rewards/IngredientQuantityMatchReward/mean": 0.5690113306045532, "rewards/IngredientQuantityMatchReward/std": 0.42663033604621886, "rewards/TotalKcalExactMatchReward/mean": 0.6375, "rewards/TotalKcalExactMatchReward/std": 0.4768159568309784, "step": 260 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0609375, "completions/max_length": 513.0, "completions/mean_length": 394.25, "completions/min_length": 261.6, "epoch": 0.0737134909596662, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7281216979026794, "kl": 0.01698645217111334, "learning_rate": 9.984721705064993e-07, "loss": 0.0006792577914893627, "reward": 2.5794155597686768, "reward_std": 0.6507645726203919, "rewards/IngredientFormatReward/mean": 0.9251153349876404, "rewards/IngredientFormatReward/std": 0.24796229004859924, "rewards/IngredientMatchReward/mean": 0.4920498549938202, "rewards/IngredientMatchReward/std": 0.31408061981201174, "rewards/IngredientQuantityMatchReward/mean": 0.5185003936290741, "rewards/IngredientQuantityMatchReward/std": 0.44893131852149964, "rewards/TotalKcalExactMatchReward/mean": 0.64375, "rewards/TotalKcalExactMatchReward/std": 0.4780986487865448, "step": 265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0375, "completions/max_length": 513.0, "completions/mean_length": 394.65625, "completions/min_length": 219.6, "epoch": 0.07510431154381085, "frac_reward_zero_std": 0.0125, "grad_norm": 0.899786114692688, "kl": 0.01553950141533278, "learning_rate": 9.98287244959228e-07, "loss": 0.0006215041503310204, "reward": 2.607704925537109, "reward_std": 0.5998050808906555, "rewards/IngredientFormatReward/mean": 0.9451153397560119, "rewards/IngredientFormatReward/std": 0.20488475263118744, "rewards/IngredientMatchReward/mean": 0.5260571718215943, "rewards/IngredientMatchReward/std": 0.3056329846382141, "rewards/IngredientQuantityMatchReward/mean": 0.4787199139595032, "rewards/IngredientQuantityMatchReward/std": 0.4413680613040924, "rewards/TotalKcalExactMatchReward/mean": 0.6578125, "rewards/TotalKcalExactMatchReward/std": 0.46880462765693665, "step": 270 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0359375, "completions/max_length": 513.0, "completions/mean_length": 394.5984375, "completions/min_length": 262.0, "epoch": 0.07649513212795549, "frac_reward_zero_std": 0.0, "grad_norm": 0.7272493243217468, "kl": 0.012873871141346171, "learning_rate": 9.980917770640873e-07, "loss": 0.0005149764940142632, "reward": 2.81902494430542, "reward_std": 0.5218873679637909, "rewards/IngredientFormatReward/mean": 0.9535379409790039, "rewards/IngredientFormatReward/std": 0.1974748343229294, "rewards/IngredientMatchReward/mean": 0.5688200712203979, "rewards/IngredientMatchReward/std": 0.31162298321723936, "rewards/IngredientQuantityMatchReward/mean": 0.5747918665409089, "rewards/IngredientQuantityMatchReward/std": 0.42267643809318545, "rewards/TotalKcalExactMatchReward/mean": 0.721875, "rewards/TotalKcalExactMatchReward/std": 0.4450909078121185, "step": 275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.034375, "completions/max_length": 513.0, "completions/mean_length": 394.384375, "completions/min_length": 252.6, "epoch": 0.07788595271210014, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7545354962348938, "kl": 0.013803366833599284, "learning_rate": 9.978857709566249e-07, "loss": 0.0005521467886865139, "reward": 2.590224504470825, "reward_std": 0.5547021985054016, "rewards/IngredientFormatReward/mean": 0.9383593797683716, "rewards/IngredientFormatReward/std": 0.21592361330986024, "rewards/IngredientMatchReward/mean": 0.5045319378376008, "rewards/IngredientMatchReward/std": 0.2803065776824951, "rewards/IngredientQuantityMatchReward/mean": 0.5004582226276397, "rewards/IngredientQuantityMatchReward/std": 0.43875272274017335, "rewards/TotalKcalExactMatchReward/mean": 0.646875, "rewards/TotalKcalExactMatchReward/std": 0.4710656404495239, "step": 280 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.8, "completions/mean_length": 389.828125, "completions/min_length": 236.4, "epoch": 0.07927677329624479, "frac_reward_zero_std": 0.0, "grad_norm": 1.1000856161117554, "kl": 0.015478781308047473, "learning_rate": 9.976692309953471e-07, "loss": 0.0006189643405377865, "reward": 2.813587999343872, "reward_std": 0.5521968245506287, "rewards/IngredientFormatReward/mean": 0.9665885329246521, "rewards/IngredientFormatReward/std": 0.15655259266495705, "rewards/IngredientMatchReward/mean": 0.5620684623718262, "rewards/IngredientMatchReward/std": 0.32483131885528566, "rewards/IngredientQuantityMatchReward/mean": 0.5911810338497162, "rewards/IngredientQuantityMatchReward/std": 0.44112812876701357, "rewards/TotalKcalExactMatchReward/mean": 0.69375, "rewards/TotalKcalExactMatchReward/std": 0.45805424451828003, "step": 285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0484375, "completions/max_length": 513.0, "completions/mean_length": 392.559375, "completions/min_length": 230.8, "epoch": 0.08066759388038942, "frac_reward_zero_std": 0.0125, "grad_norm": 0.8232524394989014, "kl": 0.015272686962271109, "learning_rate": 9.974421617616267e-07, "loss": 0.0006192252039909363, "reward": 2.5914158821105957, "reward_std": 0.6719521403312683, "rewards/IngredientFormatReward/mean": 0.9345498442649841, "rewards/IngredientFormatReward/std": 0.23133745789527893, "rewards/IngredientMatchReward/mean": 0.5013374328613281, "rewards/IngredientMatchReward/std": 0.31415660977363585, "rewards/IngredientQuantityMatchReward/mean": 0.4836536109447479, "rewards/IngredientQuantityMatchReward/std": 0.4528293967247009, "rewards/TotalKcalExactMatchReward/mean": 0.671875, "rewards/TotalKcalExactMatchReward/std": 0.4693707346916199, "step": 290 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 508.6, "completions/mean_length": 384.0015625, "completions/min_length": 226.6, "epoch": 0.08205841446453407, "frac_reward_zero_std": 0.0, "grad_norm": 0.7752309441566467, "kl": 0.014664123253896832, "learning_rate": 9.97204568059606e-07, "loss": 0.0005867013707756997, "reward": 2.727179002761841, "reward_std": 0.5595415353775024, "rewards/IngredientFormatReward/mean": 0.9712499976158142, "rewards/IngredientFormatReward/std": 0.1503540590405464, "rewards/IngredientMatchReward/mean": 0.5309282064437866, "rewards/IngredientMatchReward/std": 0.29940311312675477, "rewards/IngredientQuantityMatchReward/mean": 0.5359383165836334, "rewards/IngredientQuantityMatchReward/std": 0.42868558764457704, "rewards/TotalKcalExactMatchReward/mean": 0.6890625, "rewards/TotalKcalExactMatchReward/std": 0.4502646565437317, "step": 295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.034375, "completions/max_length": 510.4, "completions/mean_length": 396.6265625, "completions/min_length": 260.4, "epoch": 0.08344923504867872, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7676609754562378, "kl": 0.014426854299381375, "learning_rate": 9.96956454916095e-07, "loss": 0.0005770421121269464, "reward": 2.58949761390686, "reward_std": 0.6066441535949707, "rewards/IngredientFormatReward/mean": 0.9593489646911622, "rewards/IngredientFormatReward/std": 0.17723449170589448, "rewards/IngredientMatchReward/mean": 0.5475229501724244, "rewards/IngredientMatchReward/std": 0.32086688876152036, "rewards/IngredientQuantityMatchReward/mean": 0.44668819308280944, "rewards/IngredientQuantityMatchReward/std": 0.4429606318473816, "rewards/TotalKcalExactMatchReward/mean": 0.6359375, "rewards/TotalKcalExactMatchReward/std": 0.4722549319267273, "step": 300 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0296875, "completions/max_length": 513.0, "completions/mean_length": 391.1828125, "completions/min_length": 230.0, "epoch": 0.08484005563282336, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7131373882293701, "kl": 0.014628965401789174, "learning_rate": 9.96697827580466e-07, "loss": 0.0005851372610777616, "reward": 2.770707702636719, "reward_std": 0.6018207550048829, "rewards/IngredientFormatReward/mean": 0.9516406297683716, "rewards/IngredientFormatReward/std": 0.20158994495868682, "rewards/IngredientMatchReward/mean": 0.5231913506984711, "rewards/IngredientMatchReward/std": 0.31127086877822874, "rewards/IngredientQuantityMatchReward/mean": 0.6005632758140564, "rewards/IngredientQuantityMatchReward/std": 0.4462136089801788, "rewards/TotalKcalExactMatchReward/mean": 0.6953125, "rewards/TotalKcalExactMatchReward/std": 0.4513925492763519, "step": 305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 509.6, "completions/mean_length": 391.6234375, "completions/min_length": 258.0, "epoch": 0.08623087621696801, "frac_reward_zero_std": 0.025, "grad_norm": 0.7294876575469971, "kl": 0.015699848416261375, "learning_rate": 9.964286915245411e-07, "loss": 0.0006280275993049145, "reward": 2.776950168609619, "reward_std": 0.5536986947059631, "rewards/IngredientFormatReward/mean": 0.977734375, "rewards/IngredientFormatReward/std": 0.1404439002275467, "rewards/IngredientMatchReward/mean": 0.514191472530365, "rewards/IngredientMatchReward/std": 0.31364077925682066, "rewards/IngredientQuantityMatchReward/mean": 0.5928367972373962, "rewards/IngredientQuantityMatchReward/std": 0.44084068536758425, "rewards/TotalKcalExactMatchReward/mean": 0.6921875, "rewards/TotalKcalExactMatchReward/std": 0.45822349190711975, "step": 310 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 511.8, "completions/mean_length": 388.1203125, "completions/min_length": 233.6, "epoch": 0.08762169680111266, "frac_reward_zero_std": 0.0, "grad_norm": 0.780403196811676, "kl": 0.016697630251292138, "learning_rate": 9.96149052442478e-07, "loss": 0.0006678056437522173, "reward": 2.6895166397094727, "reward_std": 0.5929653167724609, "rewards/IngredientFormatReward/mean": 0.9600744128227234, "rewards/IngredientFormatReward/std": 0.18856671154499055, "rewards/IngredientMatchReward/mean": 0.5147532820701599, "rewards/IngredientMatchReward/std": 0.28225860595703123, "rewards/IngredientQuantityMatchReward/mean": 0.5459389805793762, "rewards/IngredientQuantityMatchReward/std": 0.4366516828536987, "rewards/TotalKcalExactMatchReward/mean": 0.66875, "rewards/TotalKcalExactMatchReward/std": 0.4671382009983063, "step": 315 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 504.8, "completions/mean_length": 386.215625, "completions/min_length": 266.8, "epoch": 0.0890125173852573, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7417155504226685, "kl": 0.015790742193348705, "learning_rate": 9.95858916250648e-07, "loss": 0.0006316334940493107, "reward": 2.8779314041137694, "reward_std": 0.5287329256534576, "rewards/IngredientFormatReward/mean": 0.9809635519981384, "rewards/IngredientFormatReward/std": 0.103397686034441, "rewards/IngredientMatchReward/mean": 0.5704042553901673, "rewards/IngredientMatchReward/std": 0.2879896104335785, "rewards/IngredientQuantityMatchReward/mean": 0.6140635967254638, "rewards/IngredientQuantityMatchReward/std": 0.41456308364868166, "rewards/TotalKcalExactMatchReward/mean": 0.7125, "rewards/TotalKcalExactMatchReward/std": 0.44940419793128966, "step": 320 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0328125, "completions/max_length": 512.8, "completions/mean_length": 398.8328125, "completions/min_length": 265.2, "epoch": 0.09040333796940195, "frac_reward_zero_std": 0.0, "grad_norm": 0.7298070192337036, "kl": 0.017347930185496806, "learning_rate": 9.955582890875117e-07, "loss": 0.0006938629318028689, "reward": 2.8600261211395264, "reward_std": 0.5498974919319153, "rewards/IngredientFormatReward/mean": 0.9581212997436523, "rewards/IngredientFormatReward/std": 0.17986520379781723, "rewards/IngredientMatchReward/mean": 0.58367520570755, "rewards/IngredientMatchReward/std": 0.3089649319648743, "rewards/IngredientQuantityMatchReward/mean": 0.6135420680046082, "rewards/IngredientQuantityMatchReward/std": 0.41773806810379027, "rewards/TotalKcalExactMatchReward/mean": 0.7046875, "rewards/TotalKcalExactMatchReward/std": 0.4514980256557465, "step": 325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 512.0, "completions/mean_length": 382.915625, "completions/min_length": 248.2, "epoch": 0.0917941585535466, "frac_reward_zero_std": 0.025, "grad_norm": 0.7022362351417542, "kl": 0.016667363268788905, "learning_rate": 9.952471773134892e-07, "loss": 0.0006667527835816145, "reward": 2.7937798500061035, "reward_std": 0.521451336145401, "rewards/IngredientFormatReward/mean": 0.9814843893051147, "rewards/IngredientFormatReward/std": 0.1265960529446602, "rewards/IngredientMatchReward/mean": 0.5398667573928833, "rewards/IngredientMatchReward/std": 0.3047673046588898, "rewards/IngredientQuantityMatchReward/mean": 0.5927411913871765, "rewards/IngredientQuantityMatchReward/std": 0.40651952624320986, "rewards/TotalKcalExactMatchReward/mean": 0.6796875, "rewards/TotalKcalExactMatchReward/std": 0.46350181102752686, "step": 330 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.028125, "completions/max_length": 512.0, "completions/mean_length": 390.434375, "completions/min_length": 266.6, "epoch": 0.09318497913769123, "frac_reward_zero_std": 0.05, "grad_norm": 0.8118611574172974, "kl": 0.01669728810666129, "learning_rate": 9.949255875108252e-07, "loss": 0.000667879544198513, "reward": 2.800106716156006, "reward_std": 0.5328549861907959, "rewards/IngredientFormatReward/mean": 0.9609114646911621, "rewards/IngredientFormatReward/std": 0.16961526293307544, "rewards/IngredientMatchReward/mean": 0.5768297553062439, "rewards/IngredientMatchReward/std": 0.30959353446960447, "rewards/IngredientQuantityMatchReward/mean": 0.5498654782772064, "rewards/IngredientQuantityMatchReward/std": 0.43173102736473085, "rewards/TotalKcalExactMatchReward/mean": 0.7125, "rewards/TotalKcalExactMatchReward/std": 0.45088860392570496, "step": 335 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 506.0, "completions/mean_length": 386.9359375, "completions/min_length": 251.8, "epoch": 0.09457579972183588, "frac_reward_zero_std": 0.0, "grad_norm": 0.7292128801345825, "kl": 0.017025439580902458, "learning_rate": 9.945935264834493e-07, "loss": 0.0006810775958001614, "reward": 2.8443037033081056, "reward_std": 0.5523277521133423, "rewards/IngredientFormatReward/mean": 0.96328125, "rewards/IngredientFormatReward/std": 0.17349836379289627, "rewards/IngredientMatchReward/mean": 0.5613864183425903, "rewards/IngredientMatchReward/std": 0.30622108578681945, "rewards/IngredientQuantityMatchReward/mean": 0.6008860230445862, "rewards/IngredientQuantityMatchReward/std": 0.4337588012218475, "rewards/TotalKcalExactMatchReward/mean": 0.71875, "rewards/TotalKcalExactMatchReward/std": 0.4478823184967041, "step": 340 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 507.4, "completions/mean_length": 391.43125, "completions/min_length": 240.6, "epoch": 0.09596662030598054, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7324079871177673, "kl": 0.01828371067531407, "learning_rate": 9.942510012568337e-07, "loss": 0.0007314646616578102, "reward": 2.748303699493408, "reward_std": 0.538184404373169, "rewards/IngredientFormatReward/mean": 0.9805208444595337, "rewards/IngredientFormatReward/std": 0.1241001047194004, "rewards/IngredientMatchReward/mean": 0.5349696159362793, "rewards/IngredientMatchReward/std": 0.27805001139640806, "rewards/IngredientQuantityMatchReward/mean": 0.5281257867813111, "rewards/IngredientQuantityMatchReward/std": 0.44162182211875917, "rewards/TotalKcalExactMatchReward/mean": 0.7046875, "rewards/TotalKcalExactMatchReward/std": 0.45519277453422546, "step": 345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 513.0, "completions/mean_length": 389.78125, "completions/min_length": 255.4, "epoch": 0.09735744089012517, "frac_reward_zero_std": 0.0, "grad_norm": 0.7865349650382996, "kl": 0.018346749048214407, "learning_rate": 9.938980190778424e-07, "loss": 0.0007338568568229676, "reward": 2.782780981063843, "reward_std": 0.5469257652759552, "rewards/IngredientFormatReward/mean": 0.9700781226158142, "rewards/IngredientFormatReward/std": 0.16300422549247742, "rewards/IngredientMatchReward/mean": 0.5200992047786712, "rewards/IngredientMatchReward/std": 0.30364001989364625, "rewards/IngredientQuantityMatchReward/mean": 0.545728611946106, "rewards/IngredientQuantityMatchReward/std": 0.4233491599559784, "rewards/TotalKcalExactMatchReward/mean": 0.746875, "rewards/TotalKcalExactMatchReward/std": 0.43145933747291565, "step": 350 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 508.2, "completions/mean_length": 394.4890625, "completions/min_length": 241.0, "epoch": 0.09874826147426982, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7335918545722961, "kl": 0.01754337821621448, "learning_rate": 9.935345874145795e-07, "loss": 0.0007017217110842466, "reward": 2.850579833984375, "reward_std": 0.4709952771663666, "rewards/IngredientFormatReward/mean": 0.978515625, "rewards/IngredientFormatReward/std": 0.12574003208428622, "rewards/IngredientMatchReward/mean": 0.5792435526847839, "rewards/IngredientMatchReward/std": 0.2941656231880188, "rewards/IngredientQuantityMatchReward/mean": 0.6225082039833069, "rewards/IngredientQuantityMatchReward/std": 0.4228747606277466, "rewards/TotalKcalExactMatchReward/mean": 0.6703125, "rewards/TotalKcalExactMatchReward/std": 0.4670146644115448, "step": 355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 513.0, "completions/mean_length": 386.7578125, "completions/min_length": 233.4, "epoch": 0.10013908205841446, "frac_reward_zero_std": 0.0, "grad_norm": 0.7823643088340759, "kl": 0.019979644613340498, "learning_rate": 9.931607139562303e-07, "loss": 0.0007737993262708187, "reward": 2.771283721923828, "reward_std": 0.48842650055885317, "rewards/IngredientFormatReward/mean": 0.9694271087646484, "rewards/IngredientFormatReward/std": 0.16611113548278808, "rewards/IngredientMatchReward/mean": 0.549196434020996, "rewards/IngredientMatchReward/std": 0.3055261015892029, "rewards/IngredientQuantityMatchReward/mean": 0.5510977387428284, "rewards/IngredientQuantityMatchReward/std": 0.4223670423030853, "rewards/TotalKcalExactMatchReward/mean": 0.7015625, "rewards/TotalKcalExactMatchReward/std": 0.4430360376834869, "step": 360 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 510.4, "completions/mean_length": 386.5203125, "completions/min_length": 231.8, "epoch": 0.10152990264255911, "frac_reward_zero_std": 0.0125, "grad_norm": 0.725286066532135, "kl": 0.018332467472646385, "learning_rate": 9.92776406612899e-07, "loss": 0.0007331996224820614, "reward": 2.7579694747924806, "reward_std": 0.5351399898529052, "rewards/IngredientFormatReward/mean": 0.954687488079071, "rewards/IngredientFormatReward/std": 0.18398547321557998, "rewards/IngredientMatchReward/mean": 0.5577244639396668, "rewards/IngredientMatchReward/std": 0.28916783928871154, "rewards/IngredientQuantityMatchReward/mean": 0.5721200585365296, "rewards/IngredientQuantityMatchReward/std": 0.4316247820854187, "rewards/TotalKcalExactMatchReward/mean": 0.6734375, "rewards/TotalKcalExactMatchReward/std": 0.46652968525886535, "step": 365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 510.6, "completions/mean_length": 385.4703125, "completions/min_length": 254.2, "epoch": 0.10292072322670376, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7083786129951477, "kl": 0.018818658008240164, "learning_rate": 9.923816735154415e-07, "loss": 0.0007529781199991703, "reward": 2.7624123096466064, "reward_std": 0.4925862789154053, "rewards/IngredientFormatReward/mean": 0.9794531226158142, "rewards/IngredientFormatReward/std": 0.1064148060977459, "rewards/IngredientMatchReward/mean": 0.5263802230358123, "rewards/IngredientMatchReward/std": 0.2988487720489502, "rewards/IngredientQuantityMatchReward/mean": 0.5237663745880127, "rewards/IngredientQuantityMatchReward/std": 0.4354355216026306, "rewards/TotalKcalExactMatchReward/mean": 0.7328125, "rewards/TotalKcalExactMatchReward/std": 0.4356215178966522, "step": 370 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 508.8, "completions/mean_length": 393.2140625, "completions/min_length": 260.0, "epoch": 0.1043115438108484, "frac_reward_zero_std": 0.0, "grad_norm": 0.7111307978630066, "kl": 0.021116311359219254, "learning_rate": 9.91976523015293e-07, "loss": 0.0008446965366601944, "reward": 2.743468713760376, "reward_std": 0.5661642551422119, "rewards/IngredientFormatReward/mean": 0.9618340969085694, "rewards/IngredientFormatReward/std": 0.1569497775286436, "rewards/IngredientMatchReward/mean": 0.5765910625457764, "rewards/IngredientMatchReward/std": 0.30966550707817075, "rewards/IngredientQuantityMatchReward/mean": 0.5284810245037079, "rewards/IngredientQuantityMatchReward/std": 0.43139131665229796, "rewards/TotalKcalExactMatchReward/mean": 0.6765625, "rewards/TotalKcalExactMatchReward/std": 0.4641339421272278, "step": 375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 513.0, "completions/mean_length": 394.6640625, "completions/min_length": 245.8, "epoch": 0.10570236439499305, "frac_reward_zero_std": 0.025, "grad_norm": 0.7693688869476318, "kl": 0.02106798751046881, "learning_rate": 9.915609636842911e-07, "loss": 0.0008427182212471962, "reward": 2.7324641227722166, "reward_std": 0.5566068172454834, "rewards/IngredientFormatReward/mean": 0.9533333301544189, "rewards/IngredientFormatReward/std": 0.2035890117287636, "rewards/IngredientMatchReward/mean": 0.5244115948677063, "rewards/IngredientMatchReward/std": 0.28770632147789, "rewards/IngredientQuantityMatchReward/mean": 0.5828442573547363, "rewards/IngredientQuantityMatchReward/std": 0.4409846901893616, "rewards/TotalKcalExactMatchReward/mean": 0.671875, "rewards/TotalKcalExactMatchReward/std": 0.46691768169403075, "step": 380 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 511.4, "completions/mean_length": 389.6796875, "completions/min_length": 251.2, "epoch": 0.1070931849791377, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6907310485839844, "kl": 0.020320191455539316, "learning_rate": 9.911350043144958e-07, "loss": 0.0008129674009978771, "reward": 2.8975246429443358, "reward_std": 0.4829355239868164, "rewards/IngredientFormatReward/mean": 0.964453125, "rewards/IngredientFormatReward/std": 0.14441812708973883, "rewards/IngredientMatchReward/mean": 0.5704482913017273, "rewards/IngredientMatchReward/std": 0.3032479345798492, "rewards/IngredientQuantityMatchReward/mean": 0.6235607147216797, "rewards/IngredientQuantityMatchReward/std": 0.40991583466529846, "rewards/TotalKcalExactMatchReward/mean": 0.7390625, "rewards/TotalKcalExactMatchReward/std": 0.4243670880794525, "step": 385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 513.0, "completions/mean_length": 383.9171875, "completions/min_length": 242.2, "epoch": 0.10848400556328233, "frac_reward_zero_std": 0.0, "grad_norm": 0.7413637042045593, "kl": 0.019201053376309573, "learning_rate": 9.906986539180012e-07, "loss": 0.0007680667564272881, "reward": 2.87926664352417, "reward_std": 0.5338155388832092, "rewards/IngredientFormatReward/mean": 0.9664322853088378, "rewards/IngredientFormatReward/std": 0.16732385158538818, "rewards/IngredientMatchReward/mean": 0.540427815914154, "rewards/IngredientMatchReward/std": 0.30549808144569396, "rewards/IngredientQuantityMatchReward/mean": 0.6224065780639648, "rewards/IngredientQuantityMatchReward/std": 0.43633565306663513, "rewards/TotalKcalExactMatchReward/mean": 0.75, "rewards/TotalKcalExactMatchReward/std": 0.42876131534576417, "step": 390 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 512.6, "completions/mean_length": 391.2921875, "completions/min_length": 259.4, "epoch": 0.10987482614742698, "frac_reward_zero_std": 0.025, "grad_norm": 0.6869736313819885, "kl": 0.019715036067645998, "learning_rate": 9.90251921726747e-07, "loss": 0.0007884832099080086, "reward": 2.91544189453125, "reward_std": 0.4922062695026398, "rewards/IngredientFormatReward/mean": 0.9751822829246521, "rewards/IngredientFormatReward/std": 0.13432122766971588, "rewards/IngredientMatchReward/mean": 0.5604154467582703, "rewards/IngredientMatchReward/std": 0.29352020025253295, "rewards/IngredientQuantityMatchReward/mean": 0.5985941112041473, "rewards/IngredientQuantityMatchReward/std": 0.41760724782943726, "rewards/TotalKcalExactMatchReward/mean": 0.78125, "rewards/TotalKcalExactMatchReward/std": 0.41098737716674805, "step": 395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 498.2, "completions/mean_length": 381.2296875, "completions/min_length": 242.6, "epoch": 0.11126564673157163, "frac_reward_zero_std": 0.0, "grad_norm": 0.7603372931480408, "kl": 0.02081260222475976, "learning_rate": 9.897948171923218e-07, "loss": 0.0008325819857418537, "reward": 2.7142991065979003, "reward_std": 0.5576951742172241, "rewards/IngredientFormatReward/mean": 0.9718750238418579, "rewards/IngredientFormatReward/std": 0.13910246267914772, "rewards/IngredientMatchReward/mean": 0.5314646005630493, "rewards/IngredientMatchReward/std": 0.30190695226192477, "rewards/IngredientQuantityMatchReward/mean": 0.4953344762325287, "rewards/IngredientQuantityMatchReward/std": 0.46032541394233706, "rewards/TotalKcalExactMatchReward/mean": 0.715625, "rewards/TotalKcalExactMatchReward/std": 0.4464111149311066, "step": 400 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 513.0, "completions/mean_length": 383.66875, "completions/min_length": 222.8, "epoch": 0.11265646731571627, "frac_reward_zero_std": 0.0125, "grad_norm": 0.740298867225647, "kl": 0.021776563371531665, "learning_rate": 9.893273499857642e-07, "loss": 0.000871221162378788, "reward": 2.886891746520996, "reward_std": 0.5212242007255554, "rewards/IngredientFormatReward/mean": 0.9692820072174072, "rewards/IngredientFormatReward/std": 0.1418784774839878, "rewards/IngredientMatchReward/mean": 0.5620102286338806, "rewards/IngredientMatchReward/std": 0.30565056800842283, "rewards/IngredientQuantityMatchReward/mean": 0.6118495345115662, "rewards/IngredientQuantityMatchReward/std": 0.42542343139648436, "rewards/TotalKcalExactMatchReward/mean": 0.74375, "rewards/TotalKcalExactMatchReward/std": 0.42996495962142944, "step": 405 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 510.2, "completions/mean_length": 387.7109375, "completions/min_length": 241.8, "epoch": 0.11404728789986092, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7782815098762512, "kl": 0.01931304099271074, "learning_rate": 9.888495299973573e-07, "loss": 0.0007724935188889503, "reward": 2.9141572952270507, "reward_std": 0.49541053771972654, "rewards/IngredientFormatReward/mean": 0.9708593845367431, "rewards/IngredientFormatReward/std": 0.15200091004371644, "rewards/IngredientMatchReward/mean": 0.6145932674407959, "rewards/IngredientMatchReward/std": 0.27650038003921507, "rewards/IngredientQuantityMatchReward/mean": 0.6583922863006592, "rewards/IngredientQuantityMatchReward/std": 0.4239464819431305, "rewards/TotalKcalExactMatchReward/mean": 0.6703125, "rewards/TotalKcalExactMatchReward/std": 0.4658561825752258, "step": 410 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 510.2, "completions/mean_length": 379.5578125, "completions/min_length": 243.4, "epoch": 0.11543810848400557, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7625982165336609, "kl": 0.021549660118762404, "learning_rate": 9.883613673364195e-07, "loss": 0.0008620643056929111, "reward": 2.9068836212158202, "reward_std": 0.5135960221290589, "rewards/IngredientFormatReward/mean": 0.9680729150772095, "rewards/IngredientFormatReward/std": 0.16229501515626907, "rewards/IngredientMatchReward/mean": 0.5588231742382049, "rewards/IngredientMatchReward/std": 0.2922717064619064, "rewards/IngredientQuantityMatchReward/mean": 0.6237375855445861, "rewards/IngredientQuantityMatchReward/std": 0.42023454904556273, "rewards/TotalKcalExactMatchReward/mean": 0.75625, "rewards/TotalKcalExactMatchReward/std": 0.428866970539093, "step": 415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.4, "completions/mean_length": 378.0703125, "completions/min_length": 237.6, "epoch": 0.1168289290681502, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7636048793792725, "kl": 0.021920134348329158, "learning_rate": 9.878628723310913e-07, "loss": 0.0008768060244619847, "reward": 2.9490123748779298, "reward_std": 0.5071363091468811, "rewards/IngredientFormatReward/mean": 0.971484386920929, "rewards/IngredientFormatReward/std": 0.13410393372178078, "rewards/IngredientMatchReward/mean": 0.5426004528999329, "rewards/IngredientMatchReward/std": 0.29528992176055907, "rewards/IngredientQuantityMatchReward/mean": 0.6427400946617127, "rewards/IngredientQuantityMatchReward/std": 0.41394662857055664, "rewards/TotalKcalExactMatchReward/mean": 0.7921875, "rewards/TotalKcalExactMatchReward/std": 0.3955236434936523, "step": 420 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 513.0, "completions/mean_length": 401.3625, "completions/min_length": 252.2, "epoch": 0.11821974965229486, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7036556005477905, "kl": 0.021358940622303636, "learning_rate": 9.87354055528116e-07, "loss": 0.00085435900837183, "reward": 2.8730720043182374, "reward_std": 0.5078805685043335, "rewards/IngredientFormatReward/mean": 0.9601302146911621, "rewards/IngredientFormatReward/std": 0.18644485622644424, "rewards/IngredientMatchReward/mean": 0.5474907159805298, "rewards/IngredientMatchReward/std": 0.2955142557621002, "rewards/IngredientQuantityMatchReward/mean": 0.6092010855674743, "rewards/IngredientQuantityMatchReward/std": 0.4215419411659241, "rewards/TotalKcalExactMatchReward/mean": 0.75625, "rewards/TotalKcalExactMatchReward/std": 0.4189519166946411, "step": 425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 512.4, "completions/mean_length": 394.6328125, "completions/min_length": 259.0, "epoch": 0.11961057023643949, "frac_reward_zero_std": 0.0, "grad_norm": 0.7593802809715271, "kl": 0.020391820115037262, "learning_rate": 9.868349276926173e-07, "loss": 0.000815667025744915, "reward": 3.0347278118133545, "reward_std": 0.4388155698776245, "rewards/IngredientFormatReward/mean": 0.9790736675262451, "rewards/IngredientFormatReward/std": 0.12180586419999599, "rewards/IngredientMatchReward/mean": 0.5882833242416382, "rewards/IngredientMatchReward/std": 0.28604965209960936, "rewards/IngredientQuantityMatchReward/mean": 0.687683379650116, "rewards/IngredientQuantityMatchReward/std": 0.39389697313308714, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.4065669775009155, "step": 430 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0296875, "completions/max_length": 513.0, "completions/mean_length": 394.275, "completions/min_length": 262.0, "epoch": 0.12100139082058414, "frac_reward_zero_std": 0.025, "grad_norm": 0.7075633406639099, "kl": 0.0222692629089579, "learning_rate": 9.86305499807871e-07, "loss": 0.0008907708339393139, "reward": 2.8322206020355223, "reward_std": 0.45790958404541016, "rewards/IngredientFormatReward/mean": 0.9591406226158142, "rewards/IngredientFormatReward/std": 0.16454651802778245, "rewards/IngredientMatchReward/mean": 0.5731777012348175, "rewards/IngredientMatchReward/std": 0.2920497000217438, "rewards/IngredientQuantityMatchReward/mean": 0.5842772364616394, "rewards/IngredientQuantityMatchReward/std": 0.4166082203388214, "rewards/TotalKcalExactMatchReward/mean": 0.715625, "rewards/TotalKcalExactMatchReward/std": 0.43869020938873293, "step": 435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 502.6, "completions/mean_length": 388.5390625, "completions/min_length": 224.8, "epoch": 0.12239221140472879, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7498156428337097, "kl": 0.021729087352287024, "learning_rate": 9.857657830750726e-07, "loss": 0.0008691315539181233, "reward": 2.9173073768615723, "reward_std": 0.469376403093338, "rewards/IngredientFormatReward/mean": 0.9806249976158142, "rewards/IngredientFormatReward/std": 0.10909082591533661, "rewards/IngredientMatchReward/mean": 0.5659883618354797, "rewards/IngredientMatchReward/std": 0.28756594061851504, "rewards/IngredientQuantityMatchReward/mean": 0.6035065531730652, "rewards/IngredientQuantityMatchReward/std": 0.4320640921592712, "rewards/TotalKcalExactMatchReward/mean": 0.7671875, "rewards/TotalKcalExactMatchReward/std": 0.4184816122055054, "step": 440 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 513.0, "completions/mean_length": 401.8625, "completions/min_length": 271.8, "epoch": 0.12378303198887343, "frac_reward_zero_std": 0.0, "grad_norm": 0.7427554130554199, "kl": 0.022438473138026892, "learning_rate": 9.852157889131007e-07, "loss": 0.0008975569158792496, "reward": 2.7711973667144774, "reward_std": 0.5837087631225586, "rewards/IngredientFormatReward/mean": 0.9584747195243836, "rewards/IngredientFormatReward/std": 0.19273171424865723, "rewards/IngredientMatchReward/mean": 0.5494079709053039, "rewards/IngredientMatchReward/std": 0.2919416844844818, "rewards/IngredientQuantityMatchReward/mean": 0.588314664363861, "rewards/IngredientQuantityMatchReward/std": 0.42805092930793764, "rewards/TotalKcalExactMatchReward/mean": 0.675, "rewards/TotalKcalExactMatchReward/std": 0.4670551478862762, "step": 445 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05, "completions/max_length": 513.0, "completions/mean_length": 400.6421875, "completions/min_length": 252.4, "epoch": 0.12517385257301808, "frac_reward_zero_std": 0.0, "grad_norm": 0.7527221441268921, "kl": 0.025158376456238328, "learning_rate": 9.846555289582756e-07, "loss": 0.0010063996538519858, "reward": 2.800438404083252, "reward_std": 0.5684214293956756, "rewards/IngredientFormatReward/mean": 0.9457663774490357, "rewards/IngredientFormatReward/std": 0.20938537269830704, "rewards/IngredientMatchReward/mean": 0.5764552235603333, "rewards/IngredientMatchReward/std": 0.31859682202339173, "rewards/IngredientQuantityMatchReward/mean": 0.6110292553901673, "rewards/IngredientQuantityMatchReward/std": 0.4231915295124054, "rewards/TotalKcalExactMatchReward/mean": 0.6671875, "rewards/TotalKcalExactMatchReward/std": 0.46875752210617067, "step": 450 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0265625, "completions/max_length": 513.0, "completions/mean_length": 389.359375, "completions/min_length": 236.0, "epoch": 0.12656467315716272, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7649770975112915, "kl": 0.023254029813688248, "learning_rate": 9.840850150641117e-07, "loss": 0.0009299272671341896, "reward": 2.8438170909881593, "reward_std": 0.48624941110610964, "rewards/IngredientFormatReward/mean": 0.9634375095367431, "rewards/IngredientFormatReward/std": 0.1753060534596443, "rewards/IngredientMatchReward/mean": 0.6139800667762756, "rewards/IngredientMatchReward/std": 0.30183743834495547, "rewards/IngredientQuantityMatchReward/mean": 0.525774621963501, "rewards/IngredientQuantityMatchReward/std": 0.42514413595199585, "rewards/TotalKcalExactMatchReward/mean": 0.740625, "rewards/TotalKcalExactMatchReward/std": 0.43582022190093994, "step": 455 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0515625, "completions/max_length": 513.0, "completions/mean_length": 403.421875, "completions/min_length": 276.4, "epoch": 0.12795549374130738, "frac_reward_zero_std": 0.025, "grad_norm": 1.5184186697006226, "kl": 0.027110138500574976, "learning_rate": 9.835042593010686e-07, "loss": 0.001084165833890438, "reward": 2.839724397659302, "reward_std": 0.5796327233314514, "rewards/IngredientFormatReward/mean": 0.9483072996139527, "rewards/IngredientFormatReward/std": 0.20597576946020127, "rewards/IngredientMatchReward/mean": 0.5643192052841186, "rewards/IngredientMatchReward/std": 0.3047496736049652, "rewards/IngredientQuantityMatchReward/mean": 0.6192853331565857, "rewards/IngredientQuantityMatchReward/std": 0.421432638168335, "rewards/TotalKcalExactMatchReward/mean": 0.7078125, "rewards/TotalKcalExactMatchReward/std": 0.4554017841815948, "step": 460 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 508.2, "completions/mean_length": 388.903125, "completions/min_length": 250.0, "epoch": 0.12934631432545202, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6998506784439087, "kl": 0.023188903753180057, "learning_rate": 9.829132739562948e-07, "loss": 0.0009276852011680603, "reward": 2.9770085334777834, "reward_std": 0.5292151272296906, "rewards/IngredientFormatReward/mean": 0.9733333349227905, "rewards/IngredientFormatReward/std": 0.13938566744327546, "rewards/IngredientMatchReward/mean": 0.6149696111679077, "rewards/IngredientMatchReward/std": 0.3145597755908966, "rewards/IngredientQuantityMatchReward/mean": 0.6762055993080139, "rewards/IngredientQuantityMatchReward/std": 0.4129923641681671, "rewards/TotalKcalExactMatchReward/mean": 0.7125, "rewards/TotalKcalExactMatchReward/std": 0.453085857629776, "step": 465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.040625, "completions/max_length": 513.0, "completions/mean_length": 391.6515625, "completions/min_length": 231.8, "epoch": 0.13073713490959665, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7366269826889038, "kl": 0.02417063304455951, "learning_rate": 9.823120715333675e-07, "loss": 0.0009667634963989258, "reward": 2.8089338302612306, "reward_std": 0.5877117991447449, "rewards/IngredientFormatReward/mean": 0.9498549342155457, "rewards/IngredientFormatReward/std": 0.20365288555622102, "rewards/IngredientMatchReward/mean": 0.5744320631027222, "rewards/IngredientMatchReward/std": 0.305409836769104, "rewards/IngredientQuantityMatchReward/mean": 0.5783968687057495, "rewards/IngredientQuantityMatchReward/std": 0.4373714983463287, "rewards/TotalKcalExactMatchReward/mean": 0.70625, "rewards/TotalKcalExactMatchReward/std": 0.4441069960594177, "step": 470 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.4, "completions/mean_length": 392.5515625, "completions/min_length": 270.8, "epoch": 0.13212795549374132, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7693325281143188, "kl": 0.022445485449861736, "learning_rate": 9.817006647520284e-07, "loss": 0.0008979140780866146, "reward": 2.906987428665161, "reward_std": 0.4892770886421204, "rewards/IngredientFormatReward/mean": 0.9798437476158142, "rewards/IngredientFormatReward/std": 0.11782345212996007, "rewards/IngredientMatchReward/mean": 0.5608668208122254, "rewards/IngredientMatchReward/std": 0.29060490131378175, "rewards/IngredientQuantityMatchReward/mean": 0.635026752948761, "rewards/IngredientQuantityMatchReward/std": 0.4279853343963623, "rewards/TotalKcalExactMatchReward/mean": 0.73125, "rewards/TotalKcalExactMatchReward/std": 0.4427589774131775, "step": 475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.034375, "completions/max_length": 513.0, "completions/mean_length": 400.5890625, "completions/min_length": 251.4, "epoch": 0.13351877607788595, "frac_reward_zero_std": 0.025, "grad_norm": 0.7207197546958923, "kl": 0.023216928401961923, "learning_rate": 9.810790665479146e-07, "loss": 0.0009287634864449501, "reward": 2.842219924926758, "reward_std": 0.46143017411231996, "rewards/IngredientFormatReward/mean": 0.9690104246139526, "rewards/IngredientFormatReward/std": 0.16153716892004014, "rewards/IngredientMatchReward/mean": 0.5861247420310974, "rewards/IngredientMatchReward/std": 0.2880786955356598, "rewards/IngredientQuantityMatchReward/mean": 0.5964597225189209, "rewards/IngredientQuantityMatchReward/std": 0.42537760734558105, "rewards/TotalKcalExactMatchReward/mean": 0.690625, "rewards/TotalKcalExactMatchReward/std": 0.44918020367622374, "step": 480 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 497.4, "completions/mean_length": 390.6265625, "completions/min_length": 247.2, "epoch": 0.1349095966620306, "frac_reward_zero_std": 0.0, "grad_norm": 0.7332528829574585, "kl": 0.024050121754407884, "learning_rate": 9.80447290072285e-07, "loss": 0.0009621541015803814, "reward": 2.9791000843048097, "reward_std": 0.5197779953479766, "rewards/IngredientFormatReward/mean": 0.9775520920753479, "rewards/IngredientFormatReward/std": 0.11693610101938248, "rewards/IngredientMatchReward/mean": 0.6081776976585388, "rewards/IngredientMatchReward/std": 0.31099793314933777, "rewards/IngredientQuantityMatchReward/mean": 0.6293077826499939, "rewards/IngredientQuantityMatchReward/std": 0.4286587595939636, "rewards/TotalKcalExactMatchReward/mean": 0.7640625, "rewards/TotalKcalExactMatchReward/std": 0.42141618132591246, "step": 485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0328125, "completions/max_length": 512.8, "completions/mean_length": 400.8421875, "completions/min_length": 268.2, "epoch": 0.13630041724617525, "frac_reward_zero_std": 0.0125, "grad_norm": 0.706557035446167, "kl": 0.02498110675951466, "learning_rate": 9.798053486917416e-07, "loss": 0.000999157875776291, "reward": 2.904594373703003, "reward_std": 0.5024032175540925, "rewards/IngredientFormatReward/mean": 0.9638020753860473, "rewards/IngredientFormatReward/std": 0.17550685703754426, "rewards/IngredientMatchReward/mean": 0.5668285012245178, "rewards/IngredientMatchReward/std": 0.29408658146858213, "rewards/IngredientQuantityMatchReward/mean": 0.6130263328552246, "rewards/IngredientQuantityMatchReward/std": 0.42733028531074524, "rewards/TotalKcalExactMatchReward/mean": 0.7609375, "rewards/TotalKcalExactMatchReward/std": 0.42357649803161623, "step": 490 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 513.0, "completions/mean_length": 392.090625, "completions/min_length": 258.2, "epoch": 0.1376912378303199, "frac_reward_zero_std": 0.0125, "grad_norm": 0.701976478099823, "kl": 0.023832752835005522, "learning_rate": 9.791532559879475e-07, "loss": 0.000953151099383831, "reward": 2.8906358242034913, "reward_std": 0.5113727450370789, "rewards/IngredientFormatReward/mean": 0.9754389882087707, "rewards/IngredientFormatReward/std": 0.14507034569978713, "rewards/IngredientMatchReward/mean": 0.5291600346565246, "rewards/IngredientMatchReward/std": 0.2910259008407593, "rewards/IngredientQuantityMatchReward/mean": 0.5954118013381958, "rewards/IngredientQuantityMatchReward/std": 0.42795549631118773, "rewards/TotalKcalExactMatchReward/mean": 0.790625, "rewards/TotalKcalExactMatchReward/std": 0.38760373592376707, "step": 495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 512.6, "completions/mean_length": 397.425, "completions/min_length": 252.0, "epoch": 0.13908205841446453, "frac_reward_zero_std": 0.05, "grad_norm": 0.7409101128578186, "kl": 0.02587097561918199, "learning_rate": 9.784910257573383e-07, "loss": 0.0010347630828619002, "reward": 2.857800817489624, "reward_std": 0.4655668616294861, "rewards/IngredientFormatReward/mean": 0.9700000047683716, "rewards/IngredientFormatReward/std": 0.1601935938000679, "rewards/IngredientMatchReward/mean": 0.601234495639801, "rewards/IngredientMatchReward/std": 0.2889039397239685, "rewards/IngredientQuantityMatchReward/mean": 0.5396912932395935, "rewards/IngredientQuantityMatchReward/std": 0.44021721482276915, "rewards/TotalKcalExactMatchReward/mean": 0.746875, "rewards/TotalKcalExactMatchReward/std": 0.43169044852256777, "step": 500 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 508.8, "completions/mean_length": 387.69375, "completions/min_length": 247.6, "epoch": 0.1404728789986092, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7838031053543091, "kl": 0.02346238095778972, "learning_rate": 9.778186720108318e-07, "loss": 0.0009385457262396813, "reward": 2.9772964477539063, "reward_std": 0.4686002552509308, "rewards/IngredientFormatReward/mean": 0.9853124976158142, "rewards/IngredientFormatReward/std": 0.09816802144050599, "rewards/IngredientMatchReward/mean": 0.5914366364479064, "rewards/IngredientMatchReward/std": 0.3054099023342133, "rewards/IngredientQuantityMatchReward/mean": 0.5958597302436829, "rewards/IngredientQuantityMatchReward/std": 0.4336180329322815, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.38523083329200747, "step": 505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 506.6, "completions/mean_length": 394.96875, "completions/min_length": 266.4, "epoch": 0.14186369958275383, "frac_reward_zero_std": 0.025, "grad_norm": 0.6723412275314331, "kl": 0.02323896362213418, "learning_rate": 9.771362089735307e-07, "loss": 0.0009296289645135403, "reward": 2.9428701400756836, "reward_std": 0.42799751162528993, "rewards/IngredientFormatReward/mean": 0.9716517925262451, "rewards/IngredientFormatReward/std": 0.14892885833978653, "rewards/IngredientMatchReward/mean": 0.5667912840843201, "rewards/IngredientMatchReward/std": 0.2957393884658813, "rewards/IngredientQuantityMatchReward/mean": 0.6091145873069763, "rewards/IngredientQuantityMatchReward/std": 0.44083576202392577, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.392047244310379, "step": 510 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 510.6, "completions/mean_length": 389.0640625, "completions/min_length": 264.2, "epoch": 0.14325452016689846, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6448062658309937, "kl": 0.02515047767665237, "learning_rate": 9.76443651084421e-07, "loss": 0.0010061558336019517, "reward": 2.700420045852661, "reward_std": 0.4924412965774536, "rewards/IngredientFormatReward/mean": 0.9638020992279053, "rewards/IngredientFormatReward/std": 0.163498455286026, "rewards/IngredientMatchReward/mean": 0.5357006430625916, "rewards/IngredientMatchReward/std": 0.3113968729972839, "rewards/IngredientQuantityMatchReward/mean": 0.4884172558784485, "rewards/IngredientQuantityMatchReward/std": 0.44572351574897767, "rewards/TotalKcalExactMatchReward/mean": 0.7125, "rewards/TotalKcalExactMatchReward/std": 0.4334396183490753, "step": 515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 504.4, "completions/mean_length": 389.028125, "completions/min_length": 240.6, "epoch": 0.14464534075104313, "frac_reward_zero_std": 0.025, "grad_norm": 0.6788337230682373, "kl": 0.024838618678040804, "learning_rate": 9.757410129960675e-07, "loss": 0.0009936664253473281, "reward": 2.8270545482635496, "reward_std": 0.48341248631477357, "rewards/IngredientFormatReward/mean": 0.9769010424613953, "rewards/IngredientFormatReward/std": 0.13005183711647988, "rewards/IngredientMatchReward/mean": 0.5910993456840515, "rewards/IngredientMatchReward/std": 0.28249636888504026, "rewards/IngredientQuantityMatchReward/mean": 0.5684291124343872, "rewards/IngredientQuantityMatchReward/std": 0.41839559674263, "rewards/TotalKcalExactMatchReward/mean": 0.690625, "rewards/TotalKcalExactMatchReward/std": 0.46186931133270265, "step": 520 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 507.8, "completions/mean_length": 395.584375, "completions/min_length": 251.8, "epoch": 0.14603616133518776, "frac_reward_zero_std": 0.025, "grad_norm": 0.6823793649673462, "kl": 0.023881654208526015, "learning_rate": 9.750283095743037e-07, "loss": 0.0009551938623189926, "reward": 2.8670840740203856, "reward_std": 0.45495287179946897, "rewards/IngredientFormatReward/mean": 0.9732961416244507, "rewards/IngredientFormatReward/std": 0.12547686100006103, "rewards/IngredientMatchReward/mean": 0.5820461392402649, "rewards/IngredientMatchReward/std": 0.293103963136673, "rewards/IngredientQuantityMatchReward/mean": 0.5586167931556701, "rewards/IngredientQuantityMatchReward/std": 0.43080027103424073, "rewards/TotalKcalExactMatchReward/mean": 0.753125, "rewards/TotalKcalExactMatchReward/std": 0.4183549165725708, "step": 525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 505.2, "completions/mean_length": 390.6859375, "completions/min_length": 241.6, "epoch": 0.1474269819193324, "frac_reward_zero_std": 0.0, "grad_norm": 0.6847018599510193, "kl": 0.026205667469184846, "learning_rate": 9.74305555897917e-07, "loss": 0.0010480815544724464, "reward": 2.9751473903656005, "reward_std": 0.45612571835517884, "rewards/IngredientFormatReward/mean": 0.9853515863418579, "rewards/IngredientFormatReward/std": 0.08599194474518299, "rewards/IngredientMatchReward/mean": 0.6294636487960815, "rewards/IngredientMatchReward/std": 0.2975753009319305, "rewards/IngredientQuantityMatchReward/mean": 0.5993947505950927, "rewards/IngredientQuantityMatchReward/std": 0.4184840977191925, "rewards/TotalKcalExactMatchReward/mean": 0.7609375, "rewards/TotalKcalExactMatchReward/std": 0.42358742356300355, "step": 530 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 513.0, "completions/mean_length": 386.5671875, "completions/min_length": 250.2, "epoch": 0.14881780250347706, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7143814563751221, "kl": 0.025549968774430453, "learning_rate": 9.735727672583297e-07, "loss": 0.0010221581906080246, "reward": 3.0114527225494383, "reward_std": 0.4811578273773193, "rewards/IngredientFormatReward/mean": 0.9715104222297668, "rewards/IngredientFormatReward/std": 0.15780526846647264, "rewards/IngredientMatchReward/mean": 0.5821434855461121, "rewards/IngredientMatchReward/std": 0.3114396154880524, "rewards/IngredientQuantityMatchReward/mean": 0.6687362551689148, "rewards/IngredientQuantityMatchReward/std": 0.4153988003730774, "rewards/TotalKcalExactMatchReward/mean": 0.7890625, "rewards/TotalKcalExactMatchReward/std": 0.4070970892906189, "step": 535 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 513.0, "completions/mean_length": 395.7578125, "completions/min_length": 277.4, "epoch": 0.1502086230876217, "frac_reward_zero_std": 0.05, "grad_norm": 0.7409384250640869, "kl": 0.025168495741672813, "learning_rate": 9.728299591592752e-07, "loss": 0.0010109595954418183, "reward": 2.85003867149353, "reward_std": 0.43052482008934023, "rewards/IngredientFormatReward/mean": 0.983802080154419, "rewards/IngredientFormatReward/std": 0.11858281642198562, "rewards/IngredientMatchReward/mean": 0.5917337536811829, "rewards/IngredientMatchReward/std": 0.2788821399211884, "rewards/IngredientQuantityMatchReward/mean": 0.5541902899742126, "rewards/IngredientQuantityMatchReward/std": 0.4308002769947052, "rewards/TotalKcalExactMatchReward/mean": 0.7203125, "rewards/TotalKcalExactMatchReward/std": 0.4451679825782776, "step": 540 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 508.4, "completions/mean_length": 397.5484375, "completions/min_length": 261.0, "epoch": 0.15159944367176634, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7681187987327576, "kl": 0.02836602572351694, "learning_rate": 9.72077147316471e-07, "loss": 0.0011343616992235185, "reward": 2.882566547393799, "reward_std": 0.49046033024787905, "rewards/IngredientFormatReward/mean": 0.9757552146911621, "rewards/IngredientFormatReward/std": 0.13507271111011504, "rewards/IngredientMatchReward/mean": 0.5827213525772095, "rewards/IngredientMatchReward/std": 0.2956298977136612, "rewards/IngredientQuantityMatchReward/mean": 0.6037775635719299, "rewards/IngredientQuantityMatchReward/std": 0.42004692554473877, "rewards/TotalKcalExactMatchReward/mean": 0.7203125, "rewards/TotalKcalExactMatchReward/std": 0.443309360742569, "step": 545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 504.0, "completions/mean_length": 387.8671875, "completions/min_length": 250.6, "epoch": 0.15299026425591097, "frac_reward_zero_std": 0.025, "grad_norm": 0.7030057907104492, "kl": 0.026008617551997303, "learning_rate": 9.71314347657285e-07, "loss": 0.0010403728112578392, "reward": 2.8892337799072267, "reward_std": 0.48583821058273313, "rewards/IngredientFormatReward/mean": 0.9875, "rewards/IngredientFormatReward/std": 0.09785700291395187, "rewards/IngredientMatchReward/mean": 0.5934883534908295, "rewards/IngredientMatchReward/std": 0.28317511677742, "rewards/IngredientQuantityMatchReward/mean": 0.5379329025745392, "rewards/IngredientQuantityMatchReward/std": 0.4267972707748413, "rewards/TotalKcalExactMatchReward/mean": 0.7703125, "rewards/TotalKcalExactMatchReward/std": 0.4082271695137024, "step": 550 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 513.0, "completions/mean_length": 396.6796875, "completions/min_length": 256.6, "epoch": 0.15438108484005564, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7302446365356445, "kl": 0.02743447886314243, "learning_rate": 9.705415763203992e-07, "loss": 0.0010974638164043427, "reward": 2.793476915359497, "reward_std": 0.5167571306228638, "rewards/IngredientFormatReward/mean": 0.9539973974227905, "rewards/IngredientFormatReward/std": 0.1849471166729927, "rewards/IngredientMatchReward/mean": 0.5718799471855164, "rewards/IngredientMatchReward/std": 0.29791882634162903, "rewards/IngredientQuantityMatchReward/mean": 0.5785369992256164, "rewards/IngredientQuantityMatchReward/std": 0.4268043220043182, "rewards/TotalKcalExactMatchReward/mean": 0.6890625, "rewards/TotalKcalExactMatchReward/std": 0.4540803492069244, "step": 555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 513.0, "completions/mean_length": 392.6734375, "completions/min_length": 274.4, "epoch": 0.15577190542420027, "frac_reward_zero_std": 0.025, "grad_norm": 0.6885735392570496, "kl": 0.025130884489044547, "learning_rate": 9.697588496554677e-07, "loss": 0.0010053345933556557, "reward": 2.993966054916382, "reward_std": 0.5204974055290222, "rewards/IngredientFormatReward/mean": 0.9615625023841858, "rewards/IngredientFormatReward/std": 0.17939773947000504, "rewards/IngredientMatchReward/mean": 0.6134548664093018, "rewards/IngredientMatchReward/std": 0.2914328217506409, "rewards/IngredientQuantityMatchReward/mean": 0.6142611861228943, "rewards/IngredientQuantityMatchReward/std": 0.4277330875396729, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.39028285145759584, "step": 560 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 505.4, "completions/mean_length": 389.209375, "completions/min_length": 267.0, "epoch": 0.1571627260083449, "frac_reward_zero_std": 0.025, "grad_norm": 0.7373332381248474, "kl": 0.02833310468122363, "learning_rate": 9.689661842227719e-07, "loss": 0.0011331411078572273, "reward": 2.9521562576293947, "reward_std": 0.4717487573623657, "rewards/IngredientFormatReward/mean": 0.9821354150772095, "rewards/IngredientFormatReward/std": 0.1252045825123787, "rewards/IngredientMatchReward/mean": 0.5781901121139527, "rewards/IngredientMatchReward/std": 0.287116551399231, "rewards/IngredientQuantityMatchReward/mean": 0.6090183138847352, "rewards/IngredientQuantityMatchReward/std": 0.4261681318283081, "rewards/TotalKcalExactMatchReward/mean": 0.7828125, "rewards/TotalKcalExactMatchReward/std": 0.40791478753089905, "step": 565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 509.4, "completions/mean_length": 389.15625, "completions/min_length": 257.8, "epoch": 0.15855354659248957, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7094981074333191, "kl": 0.026204002369195224, "learning_rate": 9.681635967928686e-07, "loss": 0.001048055663704872, "reward": 2.8415787696838377, "reward_std": 0.46802008152008057, "rewards/IngredientFormatReward/mean": 0.9778348207473755, "rewards/IngredientFormatReward/std": 0.12536680959165097, "rewards/IngredientMatchReward/mean": 0.5831461310386657, "rewards/IngredientMatchReward/std": 0.3007568359375, "rewards/IngredientQuantityMatchReward/mean": 0.5384102821350097, "rewards/IngredientQuantityMatchReward/std": 0.45931328535079957, "rewards/TotalKcalExactMatchReward/mean": 0.7421875, "rewards/TotalKcalExactMatchReward/std": 0.4242383182048798, "step": 570 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 509.2, "completions/mean_length": 389.178125, "completions/min_length": 263.2, "epoch": 0.1599443671766342, "frac_reward_zero_std": 0.05, "grad_norm": 0.686071515083313, "kl": 0.02771181467687711, "learning_rate": 9.673511043462365e-07, "loss": 0.0011084911413490773, "reward": 2.981544256210327, "reward_std": 0.4777001917362213, "rewards/IngredientFormatReward/mean": 0.971484375, "rewards/IngredientFormatReward/std": 0.14188916683197023, "rewards/IngredientMatchReward/mean": 0.6547222256660461, "rewards/IngredientMatchReward/std": 0.29094510078430175, "rewards/IngredientQuantityMatchReward/mean": 0.5881502151489257, "rewards/IngredientQuantityMatchReward/std": 0.4195889294147491, "rewards/TotalKcalExactMatchReward/mean": 0.7671875, "rewards/TotalKcalExactMatchReward/std": 0.42057693004608154, "step": 575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 513.0, "completions/mean_length": 388.5984375, "completions/min_length": 261.8, "epoch": 0.16133518776077885, "frac_reward_zero_std": 0.05, "grad_norm": 0.6606355309486389, "kl": 0.027671499620191754, "learning_rate": 9.665287240729166e-07, "loss": 0.0011070670560002327, "reward": 2.8866348266601562, "reward_std": 0.48432177901268003, "rewards/IngredientFormatReward/mean": 0.969348955154419, "rewards/IngredientFormatReward/std": 0.16037327349185942, "rewards/IngredientMatchReward/mean": 0.5834052681922912, "rewards/IngredientMatchReward/std": 0.2894372344017029, "rewards/IngredientQuantityMatchReward/mean": 0.6104430437088013, "rewards/IngredientQuantityMatchReward/std": 0.42995097041130065, "rewards/TotalKcalExactMatchReward/mean": 0.7234375, "rewards/TotalKcalExactMatchReward/std": 0.43584696054458616, "step": 580 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0359375, "completions/max_length": 509.2, "completions/mean_length": 392.1859375, "completions/min_length": 256.4, "epoch": 0.1627260083449235, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6764155626296997, "kl": 0.029804628575220704, "learning_rate": 9.656964733721475e-07, "loss": 0.0011925380676984787, "reward": 2.829722595214844, "reward_std": 0.5019806742668151, "rewards/IngredientFormatReward/mean": 0.9627343773841858, "rewards/IngredientFormatReward/std": 0.16002234518527986, "rewards/IngredientMatchReward/mean": 0.569929325580597, "rewards/IngredientMatchReward/std": 0.2886444807052612, "rewards/IngredientQuantityMatchReward/mean": 0.5314339637756348, "rewards/IngredientQuantityMatchReward/std": 0.42082526087760924, "rewards/TotalKcalExactMatchReward/mean": 0.765625, "rewards/TotalKcalExactMatchReward/std": 0.4208135664463043, "step": 585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 504.8, "completions/mean_length": 387.7296875, "completions/min_length": 246.6, "epoch": 0.16411682892906815, "frac_reward_zero_std": 0.075, "grad_norm": 0.7083275318145752, "kl": 0.02589824195019901, "learning_rate": 9.648543698519991e-07, "loss": 0.00103579331189394, "reward": 3.0169702053070067, "reward_std": 0.3853384256362915, "rewards/IngredientFormatReward/mean": 0.9901041507720947, "rewards/IngredientFormatReward/std": 0.06783321052789688, "rewards/IngredientMatchReward/mean": 0.613008451461792, "rewards/IngredientMatchReward/std": 0.3062650740146637, "rewards/IngredientQuantityMatchReward/mean": 0.648232614994049, "rewards/IngredientQuantityMatchReward/std": 0.4172336578369141, "rewards/TotalKcalExactMatchReward/mean": 0.765625, "rewards/TotalKcalExactMatchReward/std": 0.4179598748683929, "step": 590 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 500.0, "completions/mean_length": 388.96875, "completions/min_length": 222.4, "epoch": 0.16550764951321278, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6607844233512878, "kl": 0.02781644803471863, "learning_rate": 9.64002431328998e-07, "loss": 0.0011126678436994553, "reward": 2.947980499267578, "reward_std": 0.45839241743087766, "rewards/IngredientFormatReward/mean": 0.9763541698455811, "rewards/IngredientFormatReward/std": 0.1387864574790001, "rewards/IngredientMatchReward/mean": 0.5945783078670501, "rewards/IngredientMatchReward/std": 0.3033637821674347, "rewards/IngredientQuantityMatchReward/mean": 0.6051730632781982, "rewards/IngredientQuantityMatchReward/std": 0.4249738872051239, "rewards/TotalKcalExactMatchReward/mean": 0.771875, "rewards/TotalKcalExactMatchReward/std": 0.4144205808639526, "step": 595 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 505.6, "completions/mean_length": 397.390625, "completions/min_length": 273.6, "epoch": 0.16689847009735745, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7403410077095032, "kl": 0.027860931726172565, "learning_rate": 9.63140675827753e-07, "loss": 0.0011144131422042847, "reward": 2.9059377670288087, "reward_std": 0.42294371128082275, "rewards/IngredientFormatReward/mean": 0.9762500047683715, "rewards/IngredientFormatReward/std": 0.1115086704492569, "rewards/IngredientMatchReward/mean": 0.5822805166244507, "rewards/IngredientMatchReward/std": 0.30057679414749144, "rewards/IngredientQuantityMatchReward/mean": 0.5958447217941284, "rewards/IngredientQuantityMatchReward/std": 0.42643269896507263, "rewards/TotalKcalExactMatchReward/mean": 0.7515625, "rewards/TotalKcalExactMatchReward/std": 0.42623581290245055, "step": 600 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 509.0, "completions/mean_length": 391.7703125, "completions/min_length": 271.6, "epoch": 0.16828929068150209, "frac_reward_zero_std": 0.025, "grad_norm": 0.6867882609367371, "kl": 1.017303869058378, "learning_rate": 9.62269121580571e-07, "loss": 0.040814584493637084, "reward": 2.9245573043823243, "reward_std": 0.40745121240615845, "rewards/IngredientFormatReward/mean": 0.9918750047683715, "rewards/IngredientFormatReward/std": 0.0676910012960434, "rewards/IngredientMatchReward/mean": 0.569877827167511, "rewards/IngredientMatchReward/std": 0.2939007043838501, "rewards/IngredientQuantityMatchReward/mean": 0.5737420082092285, "rewards/IngredientQuantityMatchReward/std": 0.4358260929584503, "rewards/TotalKcalExactMatchReward/mean": 0.7890625, "rewards/TotalKcalExactMatchReward/std": 0.3946849763393402, "step": 605 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 512.0, "completions/mean_length": 393.0453125, "completions/min_length": 263.8, "epoch": 0.16968011126564672, "frac_reward_zero_std": 0.025, "grad_norm": 0.69991534948349, "kl": 0.027150334184989335, "learning_rate": 9.613877870270734e-07, "loss": 0.0010859736241400242, "reward": 2.950673246383667, "reward_std": 0.46758312582969663, "rewards/IngredientFormatReward/mean": 0.9743489623069763, "rewards/IngredientFormatReward/std": 0.1289362832903862, "rewards/IngredientMatchReward/mean": 0.6121106266975402, "rewards/IngredientMatchReward/std": 0.2894917517900467, "rewards/IngredientQuantityMatchReward/mean": 0.5939012289047241, "rewards/IngredientQuantityMatchReward/std": 0.44014000296592715, "rewards/TotalKcalExactMatchReward/mean": 0.7703125, "rewards/TotalKcalExactMatchReward/std": 0.4208824336528778, "step": 610 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 510.2, "completions/mean_length": 390.50625, "completions/min_length": 252.6, "epoch": 0.17107093184979139, "frac_reward_zero_std": 0.05, "grad_norm": 0.7449471950531006, "kl": 0.02868313742801547, "learning_rate": 9.60496690813805e-07, "loss": 0.0011473964899778367, "reward": 2.9799468994140623, "reward_std": 0.4946486234664917, "rewards/IngredientFormatReward/mean": 0.9736718773841858, "rewards/IngredientFormatReward/std": 0.14484046772122383, "rewards/IngredientMatchReward/mean": 0.630078113079071, "rewards/IngredientMatchReward/std": 0.29432931542396545, "rewards/IngredientQuantityMatchReward/mean": 0.5996343314647674, "rewards/IngredientQuantityMatchReward/std": 0.4255487024784088, "rewards/TotalKcalExactMatchReward/mean": 0.7765625, "rewards/TotalKcalExactMatchReward/std": 0.41085425615310667, "step": 615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 509.0, "completions/mean_length": 389.6984375, "completions/min_length": 280.0, "epoch": 0.17246175243393602, "frac_reward_zero_std": 0.025, "grad_norm": 0.6722802519798279, "kl": 0.028958051186054944, "learning_rate": 9.595958517938399e-07, "loss": 0.0011583505198359489, "reward": 2.8262381076812746, "reward_std": 0.4465320944786072, "rewards/IngredientFormatReward/mean": 0.9739062547683716, "rewards/IngredientFormatReward/std": 0.12499363049864769, "rewards/IngredientMatchReward/mean": 0.565862488746643, "rewards/IngredientMatchReward/std": 0.29269099831581114, "rewards/IngredientQuantityMatchReward/mean": 0.5802194237709045, "rewards/IngredientQuantityMatchReward/std": 0.437412816286087, "rewards/TotalKcalExactMatchReward/mean": 0.70625, "rewards/TotalKcalExactMatchReward/std": 0.43868969678878783, "step": 620 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 512.2, "completions/mean_length": 391.6125, "completions/min_length": 269.6, "epoch": 0.17385257301808066, "frac_reward_zero_std": 0.025, "grad_norm": 0.7302356958389282, "kl": 0.028412992018274963, "learning_rate": 9.58685289026382e-07, "loss": 0.0011363377794623374, "reward": 2.9949509143829345, "reward_std": 0.456043142080307, "rewards/IngredientFormatReward/mean": 0.9746986746788024, "rewards/IngredientFormatReward/std": 0.1321501724421978, "rewards/IngredientMatchReward/mean": 0.6073921084403991, "rewards/IngredientMatchReward/std": 0.31130966544151306, "rewards/IngredientQuantityMatchReward/mean": 0.6112977027893066, "rewards/IngredientQuantityMatchReward/std": 0.4173098862171173, "rewards/TotalKcalExactMatchReward/mean": 0.8015625, "rewards/TotalKcalExactMatchReward/std": 0.39602966904640197, "step": 625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 506.8, "completions/mean_length": 391.35, "completions/min_length": 248.6, "epoch": 0.17524339360222532, "frac_reward_zero_std": 0.0125, "grad_norm": 0.727810263633728, "kl": 0.0283606713404879, "learning_rate": 9.577650217763627e-07, "loss": 0.0011344054713845254, "reward": 2.885877180099487, "reward_std": 0.49911447763442995, "rewards/IngredientFormatReward/mean": 0.9827343702316285, "rewards/IngredientFormatReward/std": 0.11650933995842934, "rewards/IngredientMatchReward/mean": 0.5809616804122925, "rewards/IngredientMatchReward/std": 0.2889358103275299, "rewards/IngredientQuantityMatchReward/mean": 0.6143685460090638, "rewards/IngredientQuantityMatchReward/std": 0.4223813652992249, "rewards/TotalKcalExactMatchReward/mean": 0.7078125, "rewards/TotalKcalExactMatchReward/std": 0.4538150727748871, "step": 630 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.6, "completions/mean_length": 393.303125, "completions/min_length": 251.4, "epoch": 0.17663421418636996, "frac_reward_zero_std": 0.025, "grad_norm": 0.6584274172782898, "kl": 0.029251058853697033, "learning_rate": 9.56835069514032e-07, "loss": 0.0011701955460011958, "reward": 2.936937427520752, "reward_std": 0.40667486786842344, "rewards/IngredientFormatReward/mean": 0.9897916793823243, "rewards/IngredientFormatReward/std": 0.07407835721969605, "rewards/IngredientMatchReward/mean": 0.5751605868339539, "rewards/IngredientMatchReward/std": 0.2886354625225067, "rewards/IngredientQuantityMatchReward/mean": 0.6251101851463318, "rewards/IngredientQuantityMatchReward/std": 0.4174710988998413, "rewards/TotalKcalExactMatchReward/mean": 0.746875, "rewards/TotalKcalExactMatchReward/std": 0.4202993929386139, "step": 635 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 504.0, "completions/mean_length": 387.21875, "completions/min_length": 256.6, "epoch": 0.1780250347705146, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6709960103034973, "kl": 0.028330153250135482, "learning_rate": 9.558954519145486e-07, "loss": 0.0011332341469824313, "reward": 2.886409616470337, "reward_std": 0.4505835294723511, "rewards/IngredientFormatReward/mean": 0.9907552003860474, "rewards/IngredientFormatReward/std": 0.06122845821082592, "rewards/IngredientMatchReward/mean": 0.577808165550232, "rewards/IngredientMatchReward/std": 0.29142323732376096, "rewards/IngredientQuantityMatchReward/mean": 0.5990962386131287, "rewards/IngredientQuantityMatchReward/std": 0.4226303339004517, "rewards/TotalKcalExactMatchReward/mean": 0.71875, "rewards/TotalKcalExactMatchReward/std": 0.4469615936279297, "step": 640 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 507.2, "completions/mean_length": 384.828125, "completions/min_length": 226.8, "epoch": 0.17941585535465926, "frac_reward_zero_std": 0.0125, "grad_norm": 0.656096339225769, "kl": 0.028859713370911776, "learning_rate": 9.54946188857561e-07, "loss": 0.001154566276818514, "reward": 2.9817068099975588, "reward_std": 0.4848079979419708, "rewards/IngredientFormatReward/mean": 0.9825000047683716, "rewards/IngredientFormatReward/std": 0.10897158831357956, "rewards/IngredientMatchReward/mean": 0.6160367250442504, "rewards/IngredientMatchReward/std": 0.2831864595413208, "rewards/IngredientQuantityMatchReward/mean": 0.6144201159477234, "rewards/IngredientQuantityMatchReward/std": 0.42764689326286315, "rewards/TotalKcalExactMatchReward/mean": 0.76875, "rewards/TotalKcalExactMatchReward/std": 0.4078401982784271, "step": 645 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 506.8, "completions/mean_length": 389.315625, "completions/min_length": 266.2, "epoch": 0.1808066759388039, "frac_reward_zero_std": 0.025, "grad_norm": 0.734682559967041, "kl": 0.02891454305499792, "learning_rate": 9.539873004267892e-07, "loss": 0.0011331122368574142, "reward": 2.9698495388031008, "reward_std": 0.45514548420906065, "rewards/IngredientFormatReward/mean": 0.9760416746139526, "rewards/IngredientFormatReward/std": 0.11592378914356231, "rewards/IngredientMatchReward/mean": 0.5972185134887695, "rewards/IngredientMatchReward/std": 0.29427924156188967, "rewards/IngredientQuantityMatchReward/mean": 0.6325268268585205, "rewards/IngredientQuantityMatchReward/std": 0.4187918961048126, "rewards/TotalKcalExactMatchReward/mean": 0.7640625, "rewards/TotalKcalExactMatchReward/std": 0.40054896771907805, "step": 650 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 512.0, "completions/mean_length": 400.5484375, "completions/min_length": 270.0, "epoch": 0.18219749652294853, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7253914475440979, "kl": 0.030589212966151535, "learning_rate": 9.530188069095984e-07, "loss": 0.001223677024245262, "reward": 2.8579261302948, "reward_std": 0.5048339784145355, "rewards/IngredientFormatReward/mean": 0.9809375047683716, "rewards/IngredientFormatReward/std": 0.1079106204211712, "rewards/IngredientMatchReward/mean": 0.5796428680419922, "rewards/IngredientMatchReward/std": 0.2922792613506317, "rewards/IngredientQuantityMatchReward/mean": 0.5660957932472229, "rewards/IngredientQuantityMatchReward/std": 0.4285131454467773, "rewards/TotalKcalExactMatchReward/mean": 0.73125, "rewards/TotalKcalExactMatchReward/std": 0.4407511293888092, "step": 655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 512.0, "completions/mean_length": 391.3125, "completions/min_length": 278.4, "epoch": 0.1835883171070932, "frac_reward_zero_std": 0.025, "grad_norm": 0.7367101311683655, "kl": 0.029405112494714558, "learning_rate": 9.520407287965706e-07, "loss": 0.0011763956397771835, "reward": 3.055669593811035, "reward_std": 0.4588906943798065, "rewards/IngredientFormatReward/mean": 0.984375, "rewards/IngredientFormatReward/std": 0.10880739241838455, "rewards/IngredientMatchReward/mean": 0.6255388259887695, "rewards/IngredientMatchReward/std": 0.28291895389556887, "rewards/IngredientQuantityMatchReward/mean": 0.6410683155059814, "rewards/IngredientQuantityMatchReward/std": 0.39713342785835265, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.3906814754009247, "step": 660 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 510.4, "completions/mean_length": 391.725, "completions/min_length": 256.8, "epoch": 0.18497913769123783, "frac_reward_zero_std": 0.025, "grad_norm": 0.6968327164649963, "kl": 0.029885214404202998, "learning_rate": 9.510530867810705e-07, "loss": 0.0011955580674111843, "reward": 3.0517891883850097, "reward_std": 0.42123398184776306, "rewards/IngredientFormatReward/mean": 0.9899999976158143, "rewards/IngredientFormatReward/std": 0.08557360619306564, "rewards/IngredientMatchReward/mean": 0.6036179542541504, "rewards/IngredientMatchReward/std": 0.28850804567337035, "rewards/IngredientQuantityMatchReward/mean": 0.6831712603569031, "rewards/IngredientQuantityMatchReward/std": 0.4143078148365021, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.4119704604148865, "step": 665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 508.2, "completions/mean_length": 396.1671875, "completions/min_length": 277.6, "epoch": 0.18636995827538247, "frac_reward_zero_std": 0.0, "grad_norm": 0.704643189907074, "kl": 0.02852991851978004, "learning_rate": 9.50055901758808e-07, "loss": 0.00114138200879097, "reward": 2.9828824043273925, "reward_std": 0.40985679626464844, "rewards/IngredientFormatReward/mean": 0.9801543951034546, "rewards/IngredientFormatReward/std": 0.11804641857743263, "rewards/IngredientMatchReward/mean": 0.5995734095573425, "rewards/IngredientMatchReward/std": 0.2708533525466919, "rewards/IngredientQuantityMatchReward/mean": 0.6375296354293823, "rewards/IngredientQuantityMatchReward/std": 0.40709404945373534, "rewards/TotalKcalExactMatchReward/mean": 0.765625, "rewards/TotalKcalExactMatchReward/std": 0.4051818013191223, "step": 670 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 509.6, "completions/mean_length": 402.6734375, "completions/min_length": 274.2, "epoch": 0.18776077885952713, "frac_reward_zero_std": 0.0375, "grad_norm": 0.9530069828033447, "kl": 0.030310024251230062, "learning_rate": 9.49049194827396e-07, "loss": 0.0012126181274652482, "reward": 2.9908456325531008, "reward_std": 0.4299203336238861, "rewards/IngredientFormatReward/mean": 0.9764843821525574, "rewards/IngredientFormatReward/std": 0.12506350949406625, "rewards/IngredientMatchReward/mean": 0.619918167591095, "rewards/IngredientMatchReward/std": 0.29712865948677064, "rewards/IngredientQuantityMatchReward/mean": 0.6335056185722351, "rewards/IngredientQuantityMatchReward/std": 0.41634988188743594, "rewards/TotalKcalExactMatchReward/mean": 0.7609375, "rewards/TotalKcalExactMatchReward/std": 0.42713966965675354, "step": 675 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 510.0, "completions/mean_length": 397.6875, "completions/min_length": 253.6, "epoch": 0.18915159944367177, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6784078478813171, "kl": 0.029615208157338202, "learning_rate": 9.480329872859039e-07, "loss": 0.00118463933467865, "reward": 2.971098279953003, "reward_std": 0.45131182074546816, "rewards/IngredientFormatReward/mean": 0.9823176980018615, "rewards/IngredientFormatReward/std": 0.11630072444677353, "rewards/IngredientMatchReward/mean": 0.6179464221000671, "rewards/IngredientMatchReward/std": 0.30161031484603884, "rewards/IngredientQuantityMatchReward/mean": 0.5692716717720032, "rewards/IngredientQuantityMatchReward/std": 0.4135533392429352, "rewards/TotalKcalExactMatchReward/mean": 0.8015625, "rewards/TotalKcalExactMatchReward/std": 0.3857778489589691, "step": 680 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 513.0, "completions/mean_length": 399.109375, "completions/min_length": 272.2, "epoch": 0.1905424200278164, "frac_reward_zero_std": 0.0, "grad_norm": 0.7147411704063416, "kl": 0.0303641744190827, "learning_rate": 9.470073006344073e-07, "loss": 0.0012148549780249597, "reward": 2.9252397537231447, "reward_std": 0.4890766263008118, "rewards/IngredientFormatReward/mean": 0.9698437452316284, "rewards/IngredientFormatReward/std": 0.15541768670082093, "rewards/IngredientMatchReward/mean": 0.5790067076683044, "rewards/IngredientMatchReward/std": 0.310345721244812, "rewards/IngredientQuantityMatchReward/mean": 0.6013892769813538, "rewards/IngredientQuantityMatchReward/std": 0.4297428786754608, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.4112145364284515, "step": 685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 513.0, "completions/mean_length": 394.259375, "completions/min_length": 249.4, "epoch": 0.19193324061196107, "frac_reward_zero_std": 0.025, "grad_norm": 0.7676776647567749, "kl": 0.042987666395492855, "learning_rate": 9.459721565735328e-07, "loss": 0.0017181294038891791, "reward": 2.856154441833496, "reward_std": 0.518483692407608, "rewards/IngredientFormatReward/mean": 0.9700000047683716, "rewards/IngredientFormatReward/std": 0.15938313901424409, "rewards/IngredientMatchReward/mean": 0.5849032878875733, "rewards/IngredientMatchReward/std": 0.3013963460922241, "rewards/IngredientQuantityMatchReward/mean": 0.5918761491775513, "rewards/IngredientQuantityMatchReward/std": 0.43730496764183047, "rewards/TotalKcalExactMatchReward/mean": 0.709375, "rewards/TotalKcalExactMatchReward/std": 0.4479576826095581, "step": 690 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0265625, "completions/max_length": 513.0, "completions/mean_length": 393.9359375, "completions/min_length": 236.4, "epoch": 0.1933240611961057, "frac_reward_zero_std": 0.025, "grad_norm": 0.714293897151947, "kl": 0.03143521044403315, "learning_rate": 9.449275770039993e-07, "loss": 0.001257462427020073, "reward": 2.9983524322509765, "reward_std": 0.49345011711120607, "rewards/IngredientFormatReward/mean": 0.9696093797683716, "rewards/IngredientFormatReward/std": 0.16928862631320954, "rewards/IngredientMatchReward/mean": 0.5991536498069763, "rewards/IngredientMatchReward/std": 0.32413737177848817, "rewards/IngredientQuantityMatchReward/mean": 0.6608394384384155, "rewards/IngredientQuantityMatchReward/std": 0.4169635772705078, "rewards/TotalKcalExactMatchReward/mean": 0.76875, "rewards/TotalKcalExactMatchReward/std": 0.4014790415763855, "step": 695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0296875, "completions/max_length": 512.4, "completions/mean_length": 400.2984375, "completions/min_length": 245.2, "epoch": 0.19471488178025034, "frac_reward_zero_std": 0.025, "grad_norm": 0.7118370532989502, "kl": 0.02870661865454167, "learning_rate": 9.43873584026154e-07, "loss": 0.0011482758447527886, "reward": 2.994645929336548, "reward_std": 0.46502137184143066, "rewards/IngredientFormatReward/mean": 0.9630468726158142, "rewards/IngredientFormatReward/std": 0.15404580980539323, "rewards/IngredientMatchReward/mean": 0.6109213829040527, "rewards/IngredientMatchReward/std": 0.2893530786037445, "rewards/IngredientQuantityMatchReward/mean": 0.5847402095794678, "rewards/IngredientQuantityMatchReward/std": 0.425648158788681, "rewards/TotalKcalExactMatchReward/mean": 0.8359375, "rewards/TotalKcalExactMatchReward/std": 0.3676931142807007, "step": 700 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 510.0, "completions/mean_length": 396.5609375, "completions/min_length": 254.8, "epoch": 0.19610570236439498, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6505905985832214, "kl": 0.02851622838061303, "learning_rate": 9.428101999395055e-07, "loss": 0.0011407439596951008, "reward": 2.909848928451538, "reward_std": 0.44104580879211425, "rewards/IngredientFormatReward/mean": 0.9778682947158813, "rewards/IngredientFormatReward/std": 0.12648722752928734, "rewards/IngredientMatchReward/mean": 0.568949019908905, "rewards/IngredientMatchReward/std": 0.28893651366233825, "rewards/IngredientQuantityMatchReward/mean": 0.6161566257476807, "rewards/IngredientQuantityMatchReward/std": 0.4073750078678131, "rewards/TotalKcalExactMatchReward/mean": 0.746875, "rewards/TotalKcalExactMatchReward/std": 0.427871572971344, "step": 705 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 507.4, "completions/mean_length": 391.8578125, "completions/min_length": 227.0, "epoch": 0.19749652294853964, "frac_reward_zero_std": 0.05, "grad_norm": 0.6460047364234924, "kl": 0.032377722975797954, "learning_rate": 9.417374472422513e-07, "loss": 0.0012954175472259521, "reward": 2.9442656517028807, "reward_std": 0.5029145121574402, "rewards/IngredientFormatReward/mean": 0.9696875095367432, "rewards/IngredientFormatReward/std": 0.16133667081594466, "rewards/IngredientMatchReward/mean": 0.6018706560134888, "rewards/IngredientMatchReward/std": 0.2942827999591827, "rewards/IngredientQuantityMatchReward/mean": 0.6336450338363647, "rewards/IngredientQuantityMatchReward/std": 0.4295430064201355, "rewards/TotalKcalExactMatchReward/mean": 0.7390625, "rewards/TotalKcalExactMatchReward/std": 0.4340041160583496, "step": 710 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 503.8, "completions/mean_length": 390.2859375, "completions/min_length": 284.8, "epoch": 0.19888734353268428, "frac_reward_zero_std": 0.025, "grad_norm": 0.6602519154548645, "kl": 0.028656962723471225, "learning_rate": 9.406553486308027e-07, "loss": 0.0011462993919849397, "reward": 2.86570086479187, "reward_std": 0.4690316140651703, "rewards/IngredientFormatReward/mean": 0.990416657924652, "rewards/IngredientFormatReward/std": 0.06260524652898311, "rewards/IngredientMatchReward/mean": 0.5725049614906311, "rewards/IngredientMatchReward/std": 0.3110696792602539, "rewards/IngredientQuantityMatchReward/mean": 0.588716697692871, "rewards/IngredientQuantityMatchReward/std": 0.4408229410648346, "rewards/TotalKcalExactMatchReward/mean": 0.7140625, "rewards/TotalKcalExactMatchReward/std": 0.44996110796928407, "step": 715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 505.2, "completions/mean_length": 398.0546875, "completions/min_length": 244.6, "epoch": 0.20027816411682892, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6826504468917847, "kl": 0.03144657907541841, "learning_rate": 9.395639269993034e-07, "loss": 0.001257948949933052, "reward": 3.1293591499328612, "reward_std": 0.491245037317276, "rewards/IngredientFormatReward/mean": 0.9729166626930237, "rewards/IngredientFormatReward/std": 0.13929663002490997, "rewards/IngredientMatchReward/mean": 0.644785475730896, "rewards/IngredientMatchReward/std": 0.2948901176452637, "rewards/IngredientQuantityMatchReward/mean": 0.6647819876670837, "rewards/IngredientQuantityMatchReward/std": 0.39904108047485354, "rewards/TotalKcalExactMatchReward/mean": 0.846875, "rewards/TotalKcalExactMatchReward/std": 0.35378908812999726, "step": 720 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 513.0, "completions/mean_length": 401.9, "completions/min_length": 255.8, "epoch": 0.20166898470097358, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6098330616950989, "kl": 0.03146871651988477, "learning_rate": 9.384632054391466e-07, "loss": 0.0012590238824486733, "reward": 2.974373722076416, "reward_std": 0.5352123498916626, "rewards/IngredientFormatReward/mean": 0.96015625, "rewards/IngredientFormatReward/std": 0.1843802809715271, "rewards/IngredientMatchReward/mean": 0.603830623626709, "rewards/IngredientMatchReward/std": 0.29342788457870483, "rewards/IngredientQuantityMatchReward/mean": 0.6353869080543518, "rewards/IngredientQuantityMatchReward/std": 0.39622330069541933, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.41315234899520875, "step": 725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 513.0, "completions/mean_length": 394.6984375, "completions/min_length": 251.6, "epoch": 0.20305980528511822, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7605530023574829, "kl": 0.03163579052779823, "learning_rate": 9.373532072384851e-07, "loss": 0.0012655130587518216, "reward": 2.9362371921539308, "reward_std": 0.5152683258056641, "rewards/IngredientFormatReward/mean": 0.9728515625, "rewards/IngredientFormatReward/std": 0.14956897497177124, "rewards/IngredientMatchReward/mean": 0.6218024492263794, "rewards/IngredientMatchReward/std": 0.27714348435401914, "rewards/IngredientQuantityMatchReward/mean": 0.6118955969810486, "rewards/IngredientQuantityMatchReward/std": 0.41608654856681826, "rewards/TotalKcalExactMatchReward/mean": 0.7296875, "rewards/TotalKcalExactMatchReward/std": 0.4384908080101013, "step": 730 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 513.0, "completions/mean_length": 395.328125, "completions/min_length": 267.2, "epoch": 0.20445062586926285, "frac_reward_zero_std": 0.025, "grad_norm": 0.6673536896705627, "kl": 0.0321547920582816, "learning_rate": 9.362339558817393e-07, "loss": 0.0012860743328928948, "reward": 2.9911890983581544, "reward_std": 0.5093835830688477, "rewards/IngredientFormatReward/mean": 0.9742559552192688, "rewards/IngredientFormatReward/std": 0.15330008417367935, "rewards/IngredientMatchReward/mean": 0.5867342472076416, "rewards/IngredientMatchReward/std": 0.3136568009853363, "rewards/IngredientQuantityMatchReward/mean": 0.6192614316940308, "rewards/IngredientQuantityMatchReward/std": 0.4280283749103546, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.39159988760948183, "step": 735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 508.4, "completions/mean_length": 394.5765625, "completions/min_length": 253.8, "epoch": 0.20584144645340752, "frac_reward_zero_std": 0.0, "grad_norm": 0.7442731261253357, "kl": 0.0327463579364121, "learning_rate": 9.351054750491004e-07, "loss": 0.0013101151213049888, "reward": 2.9522375583648683, "reward_std": 0.4935662269592285, "rewards/IngredientFormatReward/mean": 0.9732142925262451, "rewards/IngredientFormatReward/std": 0.13986825197935104, "rewards/IngredientMatchReward/mean": 0.6265885353088378, "rewards/IngredientMatchReward/std": 0.30202017426490785, "rewards/IngredientQuantityMatchReward/mean": 0.563372266292572, "rewards/IngredientQuantityMatchReward/std": 0.4121869564056396, "rewards/TotalKcalExactMatchReward/mean": 0.7890625, "rewards/TotalKcalExactMatchReward/std": 0.39316596984863283, "step": 740 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 501.4, "completions/mean_length": 391.06875, "completions/min_length": 247.2, "epoch": 0.20723226703755215, "frac_reward_zero_std": 0.025, "grad_norm": 0.6689611673355103, "kl": 0.03534543386194855, "learning_rate": 9.339677886160291e-07, "loss": 0.0014133938588202, "reward": 2.8360384464263917, "reward_std": 0.46426122784614565, "rewards/IngredientFormatReward/mean": 0.9857031226158142, "rewards/IngredientFormatReward/std": 0.09839616119861602, "rewards/IngredientMatchReward/mean": 0.6112010240554809, "rewards/IngredientMatchReward/std": 0.2886400640010834, "rewards/IngredientQuantityMatchReward/mean": 0.5438217937946319, "rewards/IngredientQuantityMatchReward/std": 0.44390941858291627, "rewards/TotalKcalExactMatchReward/mean": 0.6953125, "rewards/TotalKcalExactMatchReward/std": 0.45529637932777406, "step": 745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0265625, "completions/max_length": 513.0, "completions/mean_length": 399.2765625, "completions/min_length": 263.8, "epoch": 0.2086230876216968, "frac_reward_zero_std": 0.025, "grad_norm": 0.6407611966133118, "kl": 0.030646953172981738, "learning_rate": 9.328209206527502e-07, "loss": 0.0012261301279067993, "reward": 2.9185956954956054, "reward_std": 0.4603721797466278, "rewards/IngredientFormatReward/mean": 0.9682291746139526, "rewards/IngredientFormatReward/std": 0.16202533841133118, "rewards/IngredientMatchReward/mean": 0.5754563689231873, "rewards/IngredientMatchReward/std": 0.29731823205947877, "rewards/IngredientQuantityMatchReward/mean": 0.5983476281166077, "rewards/IngredientQuantityMatchReward/std": 0.412556517124176, "rewards/TotalKcalExactMatchReward/mean": 0.7765625, "rewards/TotalKcalExactMatchReward/std": 0.4096756637096405, "step": 750 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 511.6, "completions/mean_length": 396.646875, "completions/min_length": 257.2, "epoch": 0.21001390820584145, "frac_reward_zero_std": 0.025, "grad_norm": 0.6754993200302124, "kl": 0.031907440279610455, "learning_rate": 9.31664895423744e-07, "loss": 0.0012764555402100086, "reward": 2.9990989208221435, "reward_std": 0.39861383438110354, "rewards/IngredientFormatReward/mean": 0.9821874976158143, "rewards/IngredientFormatReward/std": 0.10982606858015061, "rewards/IngredientMatchReward/mean": 0.6273164629936219, "rewards/IngredientMatchReward/std": 0.28818137645721437, "rewards/IngredientQuantityMatchReward/mean": 0.5927198767662049, "rewards/IngredientQuantityMatchReward/std": 0.4226322829723358, "rewards/TotalKcalExactMatchReward/mean": 0.796875, "rewards/TotalKcalExactMatchReward/std": 0.399892657995224, "step": 755 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 509.2, "completions/mean_length": 397.95625, "completions/min_length": 269.4, "epoch": 0.2114047287899861, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6390153169631958, "kl": 0.03162261028774083, "learning_rate": 9.30499737387233e-07, "loss": 0.001264964509755373, "reward": 3.1403099060058595, "reward_std": 0.43912080526351926, "rewards/IngredientFormatReward/mean": 0.9791666746139527, "rewards/IngredientFormatReward/std": 0.12517903745174408, "rewards/IngredientMatchReward/mean": 0.6559573531150817, "rewards/IngredientMatchReward/std": 0.2926880419254303, "rewards/IngredientQuantityMatchReward/mean": 0.695810866355896, "rewards/IngredientQuantityMatchReward/std": 0.3989140212535858, "rewards/TotalKcalExactMatchReward/mean": 0.809375, "rewards/TotalKcalExactMatchReward/std": 0.3878820836544037, "step": 760 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/mean_length": 398.6171875, "completions/min_length": 273.2, "epoch": 0.21279554937413073, "frac_reward_zero_std": 0.05, "grad_norm": 0.6783661246299744, "kl": 0.030737385945394637, "learning_rate": 9.293254711946633e-07, "loss": 0.0012293593958020211, "reward": 3.055101490020752, "reward_std": 0.4194644451141357, "rewards/IngredientFormatReward/mean": 0.9799218773841858, "rewards/IngredientFormatReward/std": 0.1248725950717926, "rewards/IngredientMatchReward/mean": 0.631891131401062, "rewards/IngredientMatchReward/std": 0.2970883846282959, "rewards/IngredientQuantityMatchReward/mean": 0.6479760408401489, "rewards/IngredientQuantityMatchReward/std": 0.4189946413040161, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.3833641231060028, "step": 765 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 506.2, "completions/mean_length": 385.9453125, "completions/min_length": 250.6, "epoch": 0.2141863699582754, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7975168824195862, "kl": 0.034082498284988105, "learning_rate": 9.281421216901842e-07, "loss": 0.0013631395064294337, "reward": 2.953678321838379, "reward_std": 0.4751779854297638, "rewards/IngredientFormatReward/mean": 0.9660937547683716, "rewards/IngredientFormatReward/std": 0.14769160524010658, "rewards/IngredientMatchReward/mean": 0.6321670532226562, "rewards/IngredientMatchReward/std": 0.29352579116821287, "rewards/IngredientQuantityMatchReward/mean": 0.5944800078868866, "rewards/IngredientQuantityMatchReward/std": 0.4254342496395111, "rewards/TotalKcalExactMatchReward/mean": 0.7609375, "rewards/TotalKcalExactMatchReward/std": 0.3947984278202057, "step": 770 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 513.0, "completions/mean_length": 393.13125, "completions/min_length": 252.8, "epoch": 0.21557719054242003, "frac_reward_zero_std": 0.025, "grad_norm": 0.7856742143630981, "kl": 0.03544444795697928, "learning_rate": 9.269497139101223e-07, "loss": 0.0014177929610013963, "reward": 3.043894386291504, "reward_std": 0.4352409660816193, "rewards/IngredientFormatReward/mean": 0.9726301908493042, "rewards/IngredientFormatReward/std": 0.14467429965734482, "rewards/IngredientMatchReward/mean": 0.5969146966934205, "rewards/IngredientMatchReward/std": 0.28962491154670716, "rewards/IngredientQuantityMatchReward/mean": 0.6805995225906372, "rewards/IngredientQuantityMatchReward/std": 0.4172350108623505, "rewards/TotalKcalExactMatchReward/mean": 0.79375, "rewards/TotalKcalExactMatchReward/std": 0.40479028820991514, "step": 775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 506.8, "completions/mean_length": 389.1390625, "completions/min_length": 251.8, "epoch": 0.21696801112656466, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6804711222648621, "kl": 0.036672121356241406, "learning_rate": 9.257482730824515e-07, "loss": 0.0014671500772237777, "reward": 2.9757599353790285, "reward_std": 0.45886964797973634, "rewards/IngredientFormatReward/mean": 0.9778125047683716, "rewards/IngredientFormatReward/std": 0.11689424216747284, "rewards/IngredientMatchReward/mean": 0.6214930772781372, "rewards/IngredientMatchReward/std": 0.3079483389854431, "rewards/IngredientQuantityMatchReward/mean": 0.5920794010162354, "rewards/IngredientQuantityMatchReward/std": 0.39307602643966677, "rewards/TotalKcalExactMatchReward/mean": 0.784375, "rewards/TotalKcalExactMatchReward/std": 0.4060091316699982, "step": 780 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 509.0, "completions/mean_length": 387.490625, "completions/min_length": 225.2, "epoch": 0.21835883171070933, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6856589913368225, "kl": 0.0358359067235142, "learning_rate": 9.245378246262592e-07, "loss": 0.0014331799000501632, "reward": 3.0270840644836428, "reward_std": 0.4414548695087433, "rewards/IngredientFormatReward/mean": 0.9877976179122925, "rewards/IngredientFormatReward/std": 0.08675730489194393, "rewards/IngredientMatchReward/mean": 0.6201382637023926, "rewards/IngredientMatchReward/std": 0.29792742133140565, "rewards/IngredientQuantityMatchReward/mean": 0.583210676908493, "rewards/IngredientQuantityMatchReward/std": 0.4317150294780731, "rewards/TotalKcalExactMatchReward/mean": 0.8359375, "rewards/TotalKcalExactMatchReward/std": 0.3566481202840805, "step": 785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 510.4, "completions/mean_length": 393.415625, "completions/min_length": 265.8, "epoch": 0.21974965229485396, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7207933664321899, "kl": 0.034403257770463824, "learning_rate": 9.233183941512093e-07, "loss": 0.001376278605312109, "reward": 2.964992141723633, "reward_std": 0.46569212675094607, "rewards/IngredientFormatReward/mean": 0.979296875, "rewards/IngredientFormatReward/std": 0.09206215813755989, "rewards/IngredientMatchReward/mean": 0.619907021522522, "rewards/IngredientMatchReward/std": 0.29936749339103697, "rewards/IngredientQuantityMatchReward/mean": 0.5767257273197174, "rewards/IngredientQuantityMatchReward/std": 0.41478478312492373, "rewards/TotalKcalExactMatchReward/mean": 0.7890625, "rewards/TotalKcalExactMatchReward/std": 0.39120323956012726, "step": 790 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 497.6, "completions/mean_length": 385.078125, "completions/min_length": 243.4, "epoch": 0.2211404728789986, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7506348490715027, "kl": 0.03393740386236459, "learning_rate": 9.220900074569993e-07, "loss": 0.0013577956706285477, "reward": 2.997026968002319, "reward_std": 0.46164332032203675, "rewards/IngredientFormatReward/mean": 0.9756770968437195, "rewards/IngredientFormatReward/std": 0.11836831197142601, "rewards/IngredientMatchReward/mean": 0.5768706798553467, "rewards/IngredientMatchReward/std": 0.29588728547096255, "rewards/IngredientQuantityMatchReward/mean": 0.6960417270660401, "rewards/IngredientQuantityMatchReward/std": 0.42147263288497927, "rewards/TotalKcalExactMatchReward/mean": 0.7484375, "rewards/TotalKcalExactMatchReward/std": 0.42918060421943666, "step": 795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0296875, "completions/max_length": 508.6, "completions/mean_length": 390.1328125, "completions/min_length": 251.8, "epoch": 0.22253129346314326, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7214834094047546, "kl": 0.035237916419282554, "learning_rate": 9.20852690532815e-07, "loss": 0.0014095636084675788, "reward": 2.8804359912872313, "reward_std": 0.4783861994743347, "rewards/IngredientFormatReward/mean": 0.969921875, "rewards/IngredientFormatReward/std": 0.15964724868535995, "rewards/IngredientMatchReward/mean": 0.5503304958343506, "rewards/IngredientMatchReward/std": 0.29477221965789796, "rewards/IngredientQuantityMatchReward/mean": 0.6039335608482361, "rewards/IngredientQuantityMatchReward/std": 0.4312254786491394, "rewards/TotalKcalExactMatchReward/mean": 0.75625, "rewards/TotalKcalExactMatchReward/std": 0.4224544405937195, "step": 800 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 510.6, "completions/mean_length": 391.7359375, "completions/min_length": 265.0, "epoch": 0.2239221140472879, "frac_reward_zero_std": 0.025, "grad_norm": 0.662796676158905, "kl": 0.03562974405940622, "learning_rate": 9.196064695567809e-07, "loss": 0.0014253812842071056, "reward": 2.928542375564575, "reward_std": 0.4484716713428497, "rewards/IngredientFormatReward/mean": 0.9722395896911621, "rewards/IngredientFormatReward/std": 0.1353061430156231, "rewards/IngredientMatchReward/mean": 0.5792367458343506, "rewards/IngredientMatchReward/std": 0.2964724600315094, "rewards/IngredientQuantityMatchReward/mean": 0.6426909327507019, "rewards/IngredientQuantityMatchReward/std": 0.42488672733306887, "rewards/TotalKcalExactMatchReward/mean": 0.734375, "rewards/TotalKcalExactMatchReward/std": 0.4239367187023163, "step": 805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 513.0, "completions/mean_length": 396.98125, "completions/min_length": 254.8, "epoch": 0.22531293463143254, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6553943753242493, "kl": 0.03560564590152353, "learning_rate": 9.183513708954057e-07, "loss": 0.0014241338707506658, "reward": 2.88757643699646, "reward_std": 0.46099640130996705, "rewards/IngredientFormatReward/mean": 0.9710416674613953, "rewards/IngredientFormatReward/std": 0.16456654220819472, "rewards/IngredientMatchReward/mean": 0.576538336277008, "rewards/IngredientMatchReward/std": 0.29418975710868833, "rewards/IngredientQuantityMatchReward/mean": 0.6118714094161988, "rewards/IngredientQuantityMatchReward/std": 0.42434260845184324, "rewards/TotalKcalExactMatchReward/mean": 0.728125, "rewards/TotalKcalExactMatchReward/std": 0.4403379440307617, "step": 810 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 513.0, "completions/mean_length": 389.453125, "completions/min_length": 255.2, "epoch": 0.2267037552155772, "frac_reward_zero_std": 0.05, "grad_norm": 0.7021270394325256, "kl": 0.040182554116472605, "learning_rate": 9.17087421103025e-07, "loss": 0.001607796922326088, "reward": 2.849439525604248, "reward_std": 0.45260084271430967, "rewards/IngredientFormatReward/mean": 0.9757291674613953, "rewards/IngredientFormatReward/std": 0.14797138273715973, "rewards/IngredientMatchReward/mean": 0.5811318755149841, "rewards/IngredientMatchReward/std": 0.2990729808807373, "rewards/IngredientQuantityMatchReward/mean": 0.6097660839557648, "rewards/IngredientQuantityMatchReward/std": 0.41675270199775694, "rewards/TotalKcalExactMatchReward/mean": 0.6828125, "rewards/TotalKcalExactMatchReward/std": 0.4664593517780304, "step": 815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 507.2, "completions/mean_length": 391.60625, "completions/min_length": 274.0, "epoch": 0.22809457579972184, "frac_reward_zero_std": 0.025, "grad_norm": 0.713983952999115, "kl": 0.03592538379598409, "learning_rate": 9.158146469212393e-07, "loss": 0.0014371214434504508, "reward": 2.941563129425049, "reward_std": 0.4839325726032257, "rewards/IngredientFormatReward/mean": 0.9802864551544189, "rewards/IngredientFormatReward/std": 0.12603867128491403, "rewards/IngredientMatchReward/mean": 0.605546236038208, "rewards/IngredientMatchReward/std": 0.28818662762641906, "rewards/IngredientQuantityMatchReward/mean": 0.5791679739952087, "rewards/IngredientQuantityMatchReward/std": 0.4372753441333771, "rewards/TotalKcalExactMatchReward/mean": 0.7765625, "rewards/TotalKcalExactMatchReward/std": 0.4141713917255402, "step": 820 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 513.0, "completions/mean_length": 402.3546875, "completions/min_length": 259.2, "epoch": 0.22948539638386647, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7204667329788208, "kl": 0.06811709401663393, "learning_rate": 9.145330752783481e-07, "loss": 0.0027270957827568052, "reward": 2.9213661193847655, "reward_std": 0.4193553149700165, "rewards/IngredientFormatReward/mean": 0.9707142949104309, "rewards/IngredientFormatReward/std": 0.15083834379911423, "rewards/IngredientMatchReward/mean": 0.569453752040863, "rewards/IngredientMatchReward/std": 0.2871735870838165, "rewards/IngredientQuantityMatchReward/mean": 0.6218230545520782, "rewards/IngredientQuantityMatchReward/std": 0.40918234586715696, "rewards/TotalKcalExactMatchReward/mean": 0.759375, "rewards/TotalKcalExactMatchReward/std": 0.422866028547287, "step": 825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.6, "completions/mean_length": 396.4234375, "completions/min_length": 233.4, "epoch": 0.23087621696801114, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6867784857749939, "kl": 0.038253002404235306, "learning_rate": 9.132427332887802e-07, "loss": 0.0015299856662750245, "reward": 2.9125558853149416, "reward_std": 0.44560866951942446, "rewards/IngredientFormatReward/mean": 0.9753385543823242, "rewards/IngredientFormatReward/std": 0.13101519346237184, "rewards/IngredientMatchReward/mean": 0.6180487394332885, "rewards/IngredientMatchReward/std": 0.2829564154148102, "rewards/IngredientQuantityMatchReward/mean": 0.6363561630249024, "rewards/IngredientQuantityMatchReward/std": 0.41775635480880735, "rewards/TotalKcalExactMatchReward/mean": 0.6828125, "rewards/TotalKcalExactMatchReward/std": 0.4539078712463379, "step": 830 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 507.6, "completions/mean_length": 393.5390625, "completions/min_length": 267.8, "epoch": 0.23226703755215578, "frac_reward_zero_std": 0.05, "grad_norm": 0.6686710715293884, "kl": 0.03419446239713579, "learning_rate": 9.119436482525204e-07, "loss": 0.0013676073402166366, "reward": 2.9858155727386473, "reward_std": 0.4048700749874115, "rewards/IngredientFormatReward/mean": 0.9871354103088379, "rewards/IngredientFormatReward/std": 0.08223003149032593, "rewards/IngredientMatchReward/mean": 0.6355809688568115, "rewards/IngredientMatchReward/std": 0.27816588878631593, "rewards/IngredientQuantityMatchReward/mean": 0.6396617531776428, "rewards/IngredientQuantityMatchReward/std": 0.41819279789924624, "rewards/TotalKcalExactMatchReward/mean": 0.7234375, "rewards/TotalKcalExactMatchReward/std": 0.4450302064418793, "step": 835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 509.0, "completions/mean_length": 399.4953125, "completions/min_length": 257.4, "epoch": 0.2336578581363004, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6076139211654663, "kl": 0.03447474674321711, "learning_rate": 9.106358476545312e-07, "loss": 0.0013787638396024703, "reward": 3.042753314971924, "reward_std": 0.4141663730144501, "rewards/IngredientFormatReward/mean": 0.9867931604385376, "rewards/IngredientFormatReward/std": 0.08491213321685791, "rewards/IngredientMatchReward/mean": 0.6473660826683044, "rewards/IngredientMatchReward/std": 0.3052875936031342, "rewards/IngredientQuantityMatchReward/mean": 0.6289066672325134, "rewards/IngredientQuantityMatchReward/std": 0.4121295094490051, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.411685985326767, "step": 840 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 499.6, "completions/mean_length": 400.1671875, "completions/min_length": 276.0, "epoch": 0.23504867872044508, "frac_reward_zero_std": 0.0, "grad_norm": 0.7678961753845215, "kl": 0.035930186160840094, "learning_rate": 9.093193591641721e-07, "loss": 0.0014372747391462326, "reward": 2.9903510570526124, "reward_std": 0.4582275986671448, "rewards/IngredientFormatReward/mean": 0.9776562452316284, "rewards/IngredientFormatReward/std": 0.10337670594453811, "rewards/IngredientMatchReward/mean": 0.6180661082267761, "rewards/IngredientMatchReward/std": 0.28336626291275024, "rewards/IngredientQuantityMatchReward/mean": 0.6196286797523498, "rewards/IngredientQuantityMatchReward/std": 0.4030408561229706, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.4152782797813416, "step": 845 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.028125, "completions/max_length": 513.0, "completions/mean_length": 401.3109375, "completions/min_length": 261.4, "epoch": 0.2364394993045897, "frac_reward_zero_std": 0.0, "grad_norm": 0.7015578150749207, "kl": 0.034239031421020624, "learning_rate": 9.079942106346137e-07, "loss": 0.001369218062609434, "reward": 2.9080883026123048, "reward_std": 0.5065367937088012, "rewards/IngredientFormatReward/mean": 0.9693750023841858, "rewards/IngredientFormatReward/std": 0.1656607583165169, "rewards/IngredientMatchReward/mean": 0.6202536940574646, "rewards/IngredientMatchReward/std": 0.31288120746612547, "rewards/IngredientQuantityMatchReward/mean": 0.5918970346450806, "rewards/IngredientQuantityMatchReward/std": 0.4122667610645294, "rewards/TotalKcalExactMatchReward/mean": 0.7265625, "rewards/TotalKcalExactMatchReward/std": 0.4379298210144043, "step": 850 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 511.8, "completions/mean_length": 395.1578125, "completions/min_length": 222.0, "epoch": 0.23783031988873435, "frac_reward_zero_std": 0.075, "grad_norm": 0.7002923488616943, "kl": 0.038634613458998504, "learning_rate": 9.066604301022485e-07, "loss": 0.0015293240547180175, "reward": 3.0154534339904786, "reward_std": 0.4475682139396667, "rewards/IngredientFormatReward/mean": 0.9821093797683715, "rewards/IngredientFormatReward/std": 0.0994268886744976, "rewards/IngredientMatchReward/mean": 0.6168365597724914, "rewards/IngredientMatchReward/std": 0.28127440214157107, "rewards/IngredientQuantityMatchReward/mean": 0.6883825063705444, "rewards/IngredientQuantityMatchReward/std": 0.394818377494812, "rewards/TotalKcalExactMatchReward/mean": 0.728125, "rewards/TotalKcalExactMatchReward/std": 0.4414012312889099, "step": 855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 509.2, "completions/mean_length": 390.896875, "completions/min_length": 248.2, "epoch": 0.23922114047287898, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6400624513626099, "kl": 0.03780011597555131, "learning_rate": 9.053180457860977e-07, "loss": 0.0015122408978641034, "reward": 2.980843448638916, "reward_std": 0.4466714680194855, "rewards/IngredientFormatReward/mean": 0.9730208158493042, "rewards/IngredientFormatReward/std": 0.1319844603538513, "rewards/IngredientMatchReward/mean": 0.6098846673965455, "rewards/IngredientMatchReward/std": 0.2906585156917572, "rewards/IngredientQuantityMatchReward/mean": 0.6370004773139953, "rewards/IngredientQuantityMatchReward/std": 0.422407865524292, "rewards/TotalKcalExactMatchReward/mean": 0.7609375, "rewards/TotalKcalExactMatchReward/std": 0.4248836100101471, "step": 860 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 513.0, "completions/mean_length": 395.340625, "completions/min_length": 265.8, "epoch": 0.24061196105702365, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6853905320167542, "kl": 0.03545564888045192, "learning_rate": 9.039670860872144e-07, "loss": 0.0014183764345943929, "reward": 2.9594481945037843, "reward_std": 0.45008849501609804, "rewards/IngredientFormatReward/mean": 0.9819270730018616, "rewards/IngredientFormatReward/std": 0.12875129282474518, "rewards/IngredientMatchReward/mean": 0.5874163150787354, "rewards/IngredientMatchReward/std": 0.2873119831085205, "rewards/IngredientQuantityMatchReward/mean": 0.5994797468185424, "rewards/IngredientQuantityMatchReward/std": 0.44308061003684995, "rewards/TotalKcalExactMatchReward/mean": 0.790625, "rewards/TotalKcalExactMatchReward/std": 0.38857262432575224, "step": 865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 494.8, "completions/mean_length": 397.6859375, "completions/min_length": 257.8, "epoch": 0.24200278164116829, "frac_reward_zero_std": 0.05, "grad_norm": 0.714350700378418, "kl": 0.03471857472322881, "learning_rate": 9.026075795880821e-07, "loss": 0.001388669479638338, "reward": 3.029931926727295, "reward_std": 0.39870219230651854, "rewards/IngredientFormatReward/mean": 0.9884895801544189, "rewards/IngredientFormatReward/std": 0.06684594321995974, "rewards/IngredientMatchReward/mean": 0.619605016708374, "rewards/IngredientMatchReward/std": 0.30052986145019533, "rewards/IngredientQuantityMatchReward/mean": 0.6343373656272888, "rewards/IngredientQuantityMatchReward/std": 0.43777188658714294, "rewards/TotalKcalExactMatchReward/mean": 0.7875, "rewards/TotalKcalExactMatchReward/std": 0.40009031295776365, "step": 870 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 513.0, "completions/mean_length": 395.43125, "completions/min_length": 236.8, "epoch": 0.24339360222531292, "frac_reward_zero_std": 0.075, "grad_norm": 0.6846738457679749, "kl": 0.03683779882267117, "learning_rate": 9.012395550520109e-07, "loss": 0.0014737180434167385, "reward": 3.0627113342285157, "reward_std": 0.4080630362033844, "rewards/IngredientFormatReward/mean": 0.9711979389190674, "rewards/IngredientFormatReward/std": 0.15553708821535112, "rewards/IngredientMatchReward/mean": 0.6409418463706971, "rewards/IngredientMatchReward/std": 0.2837235748767853, "rewards/IngredientQuantityMatchReward/mean": 0.6411966204643249, "rewards/IngredientQuantityMatchReward/std": 0.42754223942756653, "rewards/TotalKcalExactMatchReward/mean": 0.809375, "rewards/TotalKcalExactMatchReward/std": 0.36027163565158843, "step": 875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 513.0, "completions/mean_length": 401.959375, "completions/min_length": 261.6, "epoch": 0.24478442280945759, "frac_reward_zero_std": 0.0, "grad_norm": 0.6933212280273438, "kl": 0.038240592228248715, "learning_rate": 8.998630414225283e-07, "loss": 0.001529487781226635, "reward": 2.945117950439453, "reward_std": 0.4785862982273102, "rewards/IngredientFormatReward/mean": 0.9691666722297668, "rewards/IngredientFormatReward/std": 0.16549014002084733, "rewards/IngredientMatchReward/mean": 0.5762072086334229, "rewards/IngredientMatchReward/std": 0.28135599493980407, "rewards/IngredientQuantityMatchReward/mean": 0.6075566649436951, "rewards/IngredientQuantityMatchReward/std": 0.41557438373565675, "rewards/TotalKcalExactMatchReward/mean": 0.7921875, "rewards/TotalKcalExactMatchReward/std": 0.39553323984146116, "step": 880 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 509.0, "completions/mean_length": 398.0453125, "completions/min_length": 242.4, "epoch": 0.24617524339360222, "frac_reward_zero_std": 0.025, "grad_norm": 0.687666118144989, "kl": 0.03755876780487597, "learning_rate": 8.984780678227668e-07, "loss": 0.001502517983317375, "reward": 2.8259692668914793, "reward_std": 0.48776999711990354, "rewards/IngredientFormatReward/mean": 0.9772265672683715, "rewards/IngredientFormatReward/std": 0.12334810197353363, "rewards/IngredientMatchReward/mean": 0.5587667465209961, "rewards/IngredientMatchReward/std": 0.31351504027843474, "rewards/IngredientQuantityMatchReward/mean": 0.5915385007858276, "rewards/IngredientQuantityMatchReward/std": 0.4429964065551758, "rewards/TotalKcalExactMatchReward/mean": 0.6984375, "rewards/TotalKcalExactMatchReward/std": 0.45358685255050657, "step": 885 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 513.0, "completions/mean_length": 401.665625, "completions/min_length": 274.0, "epoch": 0.24756606397774686, "frac_reward_zero_std": 0.025, "grad_norm": 0.6766310334205627, "kl": 0.037341193715110424, "learning_rate": 8.970846635548482e-07, "loss": 0.0014935942366719247, "reward": 2.9456982135772707, "reward_std": 0.4546304702758789, "rewards/IngredientFormatReward/mean": 0.9774999976158142, "rewards/IngredientFormatReward/std": 0.14295051991939545, "rewards/IngredientMatchReward/mean": 0.5852505087852478, "rewards/IngredientMatchReward/std": 0.27505454421043396, "rewards/IngredientQuantityMatchReward/mean": 0.593885064125061, "rewards/IngredientQuantityMatchReward/std": 0.4204001009464264, "rewards/TotalKcalExactMatchReward/mean": 0.7890625, "rewards/TotalKcalExactMatchReward/std": 0.39809203147888184, "step": 890 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 513.0, "completions/mean_length": 395.15625, "completions/min_length": 263.8, "epoch": 0.24895688456189152, "frac_reward_zero_std": 0.0625, "grad_norm": 0.637706458568573, "kl": 0.034991304646246134, "learning_rate": 8.956828580992633e-07, "loss": 0.0013996277004480362, "reward": 2.9801607608795164, "reward_std": 0.44261011481285095, "rewards/IngredientFormatReward/mean": 0.9754166722297668, "rewards/IngredientFormatReward/std": 0.14382019639015198, "rewards/IngredientMatchReward/mean": 0.5979985237121582, "rewards/IngredientMatchReward/std": 0.29598821997642516, "rewards/IngredientQuantityMatchReward/mean": 0.6020581245422363, "rewards/IngredientQuantityMatchReward/std": 0.44247824549674986, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.39540945887565615, "step": 895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 510.4, "completions/mean_length": 399.5375, "completions/min_length": 264.8, "epoch": 0.25034770514603616, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7266840934753418, "kl": 0.03753926283679902, "learning_rate": 8.942726811142478e-07, "loss": 0.0015015792101621627, "reward": 2.8965949535369875, "reward_std": 0.422959041595459, "rewards/IngredientFormatReward/mean": 0.975, "rewards/IngredientFormatReward/std": 0.1332007184624672, "rewards/IngredientMatchReward/mean": 0.6016995429992675, "rewards/IngredientMatchReward/std": 0.2900406777858734, "rewards/IngredientQuantityMatchReward/mean": 0.5808328747749328, "rewards/IngredientQuantityMatchReward/std": 0.45071707367897035, "rewards/TotalKcalExactMatchReward/mean": 0.7390625, "rewards/TotalKcalExactMatchReward/std": 0.4362715780735016, "step": 900 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 508.2, "completions/mean_length": 394.2484375, "completions/min_length": 261.0, "epoch": 0.2517385257301808, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7164000272750854, "kl": 0.04179991343989968, "learning_rate": 8.928541624351562e-07, "loss": 0.001672225072979927, "reward": 3.075339603424072, "reward_std": 0.454181969165802, "rewards/IngredientFormatReward/mean": 0.9723437547683715, "rewards/IngredientFormatReward/std": 0.13345520794391633, "rewards/IngredientMatchReward/mean": 0.6349938035011291, "rewards/IngredientMatchReward/std": 0.3050593674182892, "rewards/IngredientQuantityMatchReward/mean": 0.6820645570755005, "rewards/IngredientQuantityMatchReward/std": 0.40236865282058715, "rewards/TotalKcalExactMatchReward/mean": 0.7859375, "rewards/TotalKcalExactMatchReward/std": 0.3957382678985596, "step": 905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 501.0, "completions/mean_length": 384.8, "completions/min_length": 225.2, "epoch": 0.25312934631432543, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7135931849479675, "kl": 0.03884520079009235, "learning_rate": 8.914273320738288e-07, "loss": 0.0015718767419457436, "reward": 2.922934722900391, "reward_std": 0.4588244318962097, "rewards/IngredientFormatReward/mean": 0.9789843797683716, "rewards/IngredientFormatReward/std": 0.11647437512874603, "rewards/IngredientMatchReward/mean": 0.6133823275566102, "rewards/IngredientMatchReward/std": 0.30278020203113554, "rewards/IngredientQuantityMatchReward/mean": 0.6290055751800537, "rewards/IngredientQuantityMatchReward/std": 0.42403936982154844, "rewards/TotalKcalExactMatchReward/mean": 0.7015625, "rewards/TotalKcalExactMatchReward/std": 0.452223014831543, "step": 910 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 513.0, "completions/mean_length": 396.36875, "completions/min_length": 258.8, "epoch": 0.2545201668984701, "frac_reward_zero_std": 0.025, "grad_norm": 0.6942304372787476, "kl": 0.03717799754813313, "learning_rate": 8.89992220217958e-07, "loss": 0.0014871499501168729, "reward": 3.0345030784606934, "reward_std": 0.5153680682182312, "rewards/IngredientFormatReward/mean": 0.9729166507720948, "rewards/IngredientFormatReward/std": 0.157105515897274, "rewards/IngredientMatchReward/mean": 0.6512928009033203, "rewards/IngredientMatchReward/std": 0.287622344493866, "rewards/IngredientQuantityMatchReward/mean": 0.6587311625480652, "rewards/IngredientQuantityMatchReward/std": 0.3968525528907776, "rewards/TotalKcalExactMatchReward/mean": 0.7515625, "rewards/TotalKcalExactMatchReward/std": 0.42349056601524354, "step": 915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 513.0, "completions/mean_length": 394.059375, "completions/min_length": 245.6, "epoch": 0.25591098748261476, "frac_reward_zero_std": 0.05, "grad_norm": 0.7068426012992859, "kl": 0.036225578258745374, "learning_rate": 8.885488572304488e-07, "loss": 0.0014492036774754525, "reward": 3.032380723953247, "reward_std": 0.4576831817626953, "rewards/IngredientFormatReward/mean": 0.9752864480018616, "rewards/IngredientFormatReward/std": 0.1494772642850876, "rewards/IngredientMatchReward/mean": 0.5961935997009278, "rewards/IngredientMatchReward/std": 0.2839872181415558, "rewards/IngredientQuantityMatchReward/mean": 0.6327756881713867, "rewards/IngredientQuantityMatchReward/std": 0.402322381734848, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.3686569333076477, "step": 920 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 513.0, "completions/mean_length": 396.165625, "completions/min_length": 255.8, "epoch": 0.2573018080667594, "frac_reward_zero_std": 0.0, "grad_norm": 0.6422480344772339, "kl": 0.03873362694866955, "learning_rate": 8.870972736487774e-07, "loss": 0.0015496846288442611, "reward": 3.0351743698120117, "reward_std": 0.45523384809494016, "rewards/IngredientFormatReward/mean": 0.9772916674613953, "rewards/IngredientFormatReward/std": 0.13808016031980513, "rewards/IngredientMatchReward/mean": 0.6586216449737549, "rewards/IngredientMatchReward/std": 0.2872202157974243, "rewards/IngredientQuantityMatchReward/mean": 0.602386087179184, "rewards/IngredientQuantityMatchReward/std": 0.4141529083251953, "rewards/TotalKcalExactMatchReward/mean": 0.796875, "rewards/TotalKcalExactMatchReward/std": 0.3945878803730011, "step": 925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 513.0, "completions/mean_length": 391.778125, "completions/min_length": 267.2, "epoch": 0.25869262865090403, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7084170579910278, "kl": 0.03909153123386204, "learning_rate": 8.856375001843441e-07, "loss": 0.0015634719282388687, "reward": 2.907779502868652, "reward_std": 0.4718652844429016, "rewards/IngredientFormatReward/mean": 0.9831250071525574, "rewards/IngredientFormatReward/std": 0.1244356945157051, "rewards/IngredientMatchReward/mean": 0.5946900010108948, "rewards/IngredientMatchReward/std": 0.2881698727607727, "rewards/IngredientQuantityMatchReward/mean": 0.5846520006656647, "rewards/IngredientQuantityMatchReward/std": 0.42135303020477294, "rewards/TotalKcalExactMatchReward/mean": 0.7453125, "rewards/TotalKcalExactMatchReward/std": 0.4285692572593689, "step": 930 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 510.8, "completions/mean_length": 398.696875, "completions/min_length": 265.8, "epoch": 0.26008344923504867, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7023848295211792, "kl": 0.0367943222168833, "learning_rate": 8.84169567721824e-07, "loss": 0.0014716638252139091, "reward": 2.995437002182007, "reward_std": 0.4420520782470703, "rewards/IngredientFormatReward/mean": 0.9889955401420594, "rewards/IngredientFormatReward/std": 0.08887074701488018, "rewards/IngredientMatchReward/mean": 0.621924614906311, "rewards/IngredientMatchReward/std": 0.2982144713401794, "rewards/IngredientQuantityMatchReward/mean": 0.6204544305801392, "rewards/IngredientQuantityMatchReward/std": 0.41932734847068787, "rewards/TotalKcalExactMatchReward/mean": 0.7640625, "rewards/TotalKcalExactMatchReward/std": 0.42419267892837526, "step": 935 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 511.6, "completions/mean_length": 392.1734375, "completions/min_length": 251.2, "epoch": 0.2614742698191933, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6219667196273804, "kl": 0.038923771027475594, "learning_rate": 8.826935073185133e-07, "loss": 0.0015569287352263927, "reward": 2.885612678527832, "reward_std": 0.47264568209648133, "rewards/IngredientFormatReward/mean": 0.9668750047683716, "rewards/IngredientFormatReward/std": 0.15271946042776108, "rewards/IngredientMatchReward/mean": 0.5864410042762757, "rewards/IngredientMatchReward/std": 0.31077885031700136, "rewards/IngredientQuantityMatchReward/mean": 0.563546645641327, "rewards/IngredientQuantityMatchReward/std": 0.4403660535812378, "rewards/TotalKcalExactMatchReward/mean": 0.76875, "rewards/TotalKcalExactMatchReward/std": 0.40360012352466584, "step": 940 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0265625, "completions/max_length": 513.0, "completions/mean_length": 395.6859375, "completions/min_length": 261.0, "epoch": 0.26286509040333794, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6733676195144653, "kl": 0.04225626492407173, "learning_rate": 8.812093502036731e-07, "loss": 0.0016901016235351563, "reward": 3.065718173980713, "reward_std": 0.5067483425140381, "rewards/IngredientFormatReward/mean": 0.9720312595367432, "rewards/IngredientFormatReward/std": 0.1448599338531494, "rewards/IngredientMatchReward/mean": 0.6443012237548829, "rewards/IngredientMatchReward/std": 0.30254427194595335, "rewards/IngredientQuantityMatchReward/mean": 0.6712606906890869, "rewards/IngredientQuantityMatchReward/std": 0.3889262616634369, "rewards/TotalKcalExactMatchReward/mean": 0.778125, "rewards/TotalKcalExactMatchReward/std": 0.4119017422199249, "step": 945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.040625, "completions/max_length": 513.0, "completions/mean_length": 396.7359375, "completions/min_length": 269.4, "epoch": 0.26425591098748263, "frac_reward_zero_std": 0.025, "grad_norm": 0.6687275767326355, "kl": 0.04134470855351537, "learning_rate": 8.79717127777867e-07, "loss": 0.0016542974859476089, "reward": 2.9952784061431883, "reward_std": 0.4762373924255371, "rewards/IngredientFormatReward/mean": 0.9588392972946167, "rewards/IngredientFormatReward/std": 0.18952201455831527, "rewards/IngredientMatchReward/mean": 0.5773170948028564, "rewards/IngredientMatchReward/std": 0.30151690244674684, "rewards/IngredientQuantityMatchReward/mean": 0.6888094902038574, "rewards/IngredientQuantityMatchReward/std": 0.4004581332206726, "rewards/TotalKcalExactMatchReward/mean": 0.7703125, "rewards/TotalKcalExactMatchReward/std": 0.40736610889434816, "step": 950 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0265625, "completions/max_length": 513.0, "completions/mean_length": 397.571875, "completions/min_length": 241.2, "epoch": 0.26564673157162727, "frac_reward_zero_std": 0.05, "grad_norm": 0.6472366452217102, "kl": 0.06367715783417224, "learning_rate": 8.782168716122987e-07, "loss": 0.0025473890826106073, "reward": 3.036988115310669, "reward_std": 0.45366032123565675, "rewards/IngredientFormatReward/mean": 0.9667447924613952, "rewards/IngredientFormatReward/std": 0.16739132702350618, "rewards/IngredientMatchReward/mean": 0.6112258076667786, "rewards/IngredientMatchReward/std": 0.29761743545532227, "rewards/IngredientQuantityMatchReward/mean": 0.6605799674987793, "rewards/IngredientQuantityMatchReward/std": 0.41313338875770567, "rewards/TotalKcalExactMatchReward/mean": 0.7984375, "rewards/TotalKcalExactMatchReward/std": 0.3953717231750488, "step": 955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 509.8, "completions/mean_length": 398.71875, "completions/min_length": 282.6, "epoch": 0.2670375521557719, "frac_reward_zero_std": 0.05, "grad_norm": 0.5986440777778625, "kl": 0.036805032892152666, "learning_rate": 8.767086134481425e-07, "loss": 0.0014722667634487151, "reward": 2.9474908351898192, "reward_std": 0.43001130819320676, "rewards/IngredientFormatReward/mean": 0.98203125, "rewards/IngredientFormatReward/std": 0.12097890824079513, "rewards/IngredientMatchReward/mean": 0.6146249771118164, "rewards/IngredientMatchReward/std": 0.28172118961811066, "rewards/IngredientQuantityMatchReward/mean": 0.6195845484733582, "rewards/IngredientQuantityMatchReward/std": 0.42852511405944826, "rewards/TotalKcalExactMatchReward/mean": 0.73125, "rewards/TotalKcalExactMatchReward/std": 0.4326680898666382, "step": 960 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 513.0, "completions/mean_length": 396.1828125, "completions/min_length": 271.2, "epoch": 0.26842837273991654, "frac_reward_zero_std": 0.075, "grad_norm": 0.6705757975578308, "kl": 0.03636681609787047, "learning_rate": 8.751923851958727e-07, "loss": 0.0014547625556588172, "reward": 2.9844186305999756, "reward_std": 0.43761271238327026, "rewards/IngredientFormatReward/mean": 0.9759374976158142, "rewards/IngredientFormatReward/std": 0.14842937886714935, "rewards/IngredientMatchReward/mean": 0.6150954723358154, "rewards/IngredientMatchReward/std": 0.30773370862007143, "rewards/IngredientQuantityMatchReward/mean": 0.629323172569275, "rewards/IngredientQuantityMatchReward/std": 0.41692014336586, "rewards/TotalKcalExactMatchReward/mean": 0.7640625, "rewards/TotalKcalExactMatchReward/std": 0.42084046006202697, "step": 965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 513.0, "completions/mean_length": 394.0953125, "completions/min_length": 245.0, "epoch": 0.2698191933240612, "frac_reward_zero_std": 0.05, "grad_norm": 0.6692220568656921, "kl": 0.03661972593981773, "learning_rate": 8.736682189345879e-07, "loss": 0.0014646416530013084, "reward": 3.068998193740845, "reward_std": 0.3991569995880127, "rewards/IngredientFormatReward/mean": 0.9866927027702331, "rewards/IngredientFormatReward/std": 0.10508427172899246, "rewards/IngredientMatchReward/mean": 0.6692838549613953, "rewards/IngredientMatchReward/std": 0.29518154859542844, "rewards/IngredientQuantityMatchReward/mean": 0.6364591598510743, "rewards/IngredientQuantityMatchReward/std": 0.41511212587356566, "rewards/TotalKcalExactMatchReward/mean": 0.7765625, "rewards/TotalKcalExactMatchReward/std": 0.39989053606987, "step": 970 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0265625, "completions/max_length": 511.8, "completions/mean_length": 403.2828125, "completions/min_length": 268.6, "epoch": 0.2712100139082058, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6330889463424683, "kl": 0.034678391995839775, "learning_rate": 8.721361469113322e-07, "loss": 0.0013872714713215827, "reward": 3.068177604675293, "reward_std": 0.45619659423828124, "rewards/IngredientFormatReward/mean": 0.9731250047683716, "rewards/IngredientFormatReward/std": 0.13897247463464737, "rewards/IngredientMatchReward/mean": 0.6402989029884338, "rewards/IngredientMatchReward/std": 0.27887474894523623, "rewards/IngredientQuantityMatchReward/mean": 0.6235036015510559, "rewards/IngredientQuantityMatchReward/std": 0.40562615394592283, "rewards/TotalKcalExactMatchReward/mean": 0.83125, "rewards/TotalKcalExactMatchReward/std": 0.3756143867969513, "step": 975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 510.4, "completions/mean_length": 391.396875, "completions/min_length": 250.6, "epoch": 0.2726008344923505, "frac_reward_zero_std": 0.05, "grad_norm": 0.7315894365310669, "kl": 0.03695494781713933, "learning_rate": 8.705962015404142e-07, "loss": 0.0014784792438149452, "reward": 3.024481010437012, "reward_std": 0.43837874531745913, "rewards/IngredientFormatReward/mean": 0.9778645753860473, "rewards/IngredientFormatReward/std": 0.1255320355296135, "rewards/IngredientMatchReward/mean": 0.6787772893905639, "rewards/IngredientMatchReward/std": 0.2739121735095978, "rewards/IngredientQuantityMatchReward/mean": 0.6569016218185425, "rewards/IngredientQuantityMatchReward/std": 0.4188307225704193, "rewards/TotalKcalExactMatchReward/mean": 0.7109375, "rewards/TotalKcalExactMatchReward/std": 0.43596926927566526, "step": 980 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 509.8, "completions/mean_length": 394.14375, "completions/min_length": 246.2, "epoch": 0.27399165507649514, "frac_reward_zero_std": 0.0125, "grad_norm": 0.679327666759491, "kl": 0.03838320455979556, "learning_rate": 8.690484154027191e-07, "loss": 0.0015355561859905719, "reward": 3.014692544937134, "reward_std": 0.4545248806476593, "rewards/IngredientFormatReward/mean": 0.9801562428474426, "rewards/IngredientFormatReward/std": 0.11655383929610252, "rewards/IngredientMatchReward/mean": 0.5849051475524902, "rewards/IngredientMatchReward/std": 0.27766265869140627, "rewards/IngredientQuantityMatchReward/mean": 0.6105686783790588, "rewards/IngredientQuantityMatchReward/std": 0.439173287153244, "rewards/TotalKcalExactMatchReward/mean": 0.8390625, "rewards/TotalKcalExactMatchReward/std": 0.3558571696281433, "step": 985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 503.2, "completions/mean_length": 394.9859375, "completions/min_length": 292.0, "epoch": 0.2753824756606398, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6654589176177979, "kl": 0.03780271227005869, "learning_rate": 8.674928212450214e-07, "loss": 0.0015188731253147126, "reward": 2.919002819061279, "reward_std": 0.4351273596286774, "rewards/IngredientFormatReward/mean": 0.9859375, "rewards/IngredientFormatReward/std": 0.08695523887872696, "rewards/IngredientMatchReward/mean": 0.6129117131233215, "rewards/IngredientMatchReward/std": 0.30980420112609863, "rewards/IngredientQuantityMatchReward/mean": 0.5967161536216736, "rewards/IngredientQuantityMatchReward/std": 0.4323438584804535, "rewards/TotalKcalExactMatchReward/mean": 0.7234375, "rewards/TotalKcalExactMatchReward/std": 0.44485429525375364, "step": 990 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 505.4, "completions/mean_length": 398.0296875, "completions/min_length": 250.4, "epoch": 0.2767732962447844, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7313476800918579, "kl": 0.039417910808697346, "learning_rate": 8.659294519792908e-07, "loss": 0.001576758362352848, "reward": 3.0119237899780273, "reward_std": 0.45336784720420836, "rewards/IngredientFormatReward/mean": 0.9878125071525574, "rewards/IngredientFormatReward/std": 0.07935706898570061, "rewards/IngredientMatchReward/mean": 0.6188765048980713, "rewards/IngredientMatchReward/std": 0.29016692042350767, "rewards/IngredientQuantityMatchReward/mean": 0.6177348375320435, "rewards/IngredientQuantityMatchReward/std": 0.43465182185173035, "rewards/TotalKcalExactMatchReward/mean": 0.7875, "rewards/TotalKcalExactMatchReward/std": 0.3942445933818817, "step": 995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 507.0, "completions/mean_length": 394.6484375, "completions/min_length": 251.0, "epoch": 0.27816411682892905, "frac_reward_zero_std": 0.025, "grad_norm": 0.6560540199279785, "kl": 0.037446612119674684, "learning_rate": 8.643583406819964e-07, "loss": 0.0014979524537920951, "reward": 3.034720516204834, "reward_std": 0.44605728387832644, "rewards/IngredientFormatReward/mean": 0.9841778397560119, "rewards/IngredientFormatReward/std": 0.09864606074988842, "rewards/IngredientMatchReward/mean": 0.6110633611679077, "rewards/IngredientMatchReward/std": 0.2937765657901764, "rewards/IngredientQuantityMatchReward/mean": 0.6644793748855591, "rewards/IngredientQuantityMatchReward/std": 0.4109896540641785, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.41746620535850526, "step": 1000 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 512.4, "completions/mean_length": 392.9859375, "completions/min_length": 262.4, "epoch": 0.2795549374130737, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7340788841247559, "kl": 0.039979275036603215, "learning_rate": 8.627795205934068e-07, "loss": 0.0015989819541573525, "reward": 2.9803890228271483, "reward_std": 0.4077727794647217, "rewards/IngredientFormatReward/mean": 0.9702473878860474, "rewards/IngredientFormatReward/std": 0.13495282642543316, "rewards/IngredientMatchReward/mean": 0.6199850082397461, "rewards/IngredientMatchReward/std": 0.28927130699157716, "rewards/IngredientQuantityMatchReward/mean": 0.6557816982269287, "rewards/IngredientQuantityMatchReward/std": 0.426509028673172, "rewards/TotalKcalExactMatchReward/mean": 0.734375, "rewards/TotalKcalExactMatchReward/std": 0.43847213983535765, "step": 1005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 508.6, "completions/mean_length": 393.7125, "completions/min_length": 267.4, "epoch": 0.2809457579972184, "frac_reward_zero_std": 0.075, "grad_norm": 0.6542072296142578, "kl": 0.03828480986412615, "learning_rate": 8.611930251168866e-07, "loss": 0.0015314262360334395, "reward": 2.9594085693359373, "reward_std": 0.3793847501277924, "rewards/IngredientFormatReward/mean": 0.9762500047683715, "rewards/IngredientFormatReward/std": 0.1294731840491295, "rewards/IngredientMatchReward/mean": 0.5929135680198669, "rewards/IngredientMatchReward/std": 0.2810113072395325, "rewards/IngredientQuantityMatchReward/mean": 0.6214950919151306, "rewards/IngredientQuantityMatchReward/std": 0.42402665615081786, "rewards/TotalKcalExactMatchReward/mean": 0.76875, "rewards/TotalKcalExactMatchReward/std": 0.4016814470291138, "step": 1010 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 505.4, "completions/mean_length": 387.6640625, "completions/min_length": 250.4, "epoch": 0.282336578581363, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6601881980895996, "kl": 0.0400760383810848, "learning_rate": 8.595988878181902e-07, "loss": 0.0016030745580792427, "reward": 3.040871238708496, "reward_std": 0.4805952966213226, "rewards/IngredientFormatReward/mean": 0.9822395801544189, "rewards/IngredientFormatReward/std": 0.10235036537051201, "rewards/IngredientMatchReward/mean": 0.6395058393478393, "rewards/IngredientMatchReward/std": 0.2921601474285126, "rewards/IngredientQuantityMatchReward/mean": 0.6347507834434509, "rewards/IngredientQuantityMatchReward/std": 0.4206187427043915, "rewards/TotalKcalExactMatchReward/mean": 0.784375, "rewards/TotalKcalExactMatchReward/std": 0.41050392389297485, "step": 1015 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 510.4, "completions/mean_length": 398.084375, "completions/min_length": 273.4, "epoch": 0.28372739916550765, "frac_reward_zero_std": 0.05, "grad_norm": 0.6317110061645508, "kl": 0.036205627908930185, "learning_rate": 8.579971424247512e-07, "loss": 0.0014481697231531142, "reward": 2.9704572200775146, "reward_std": 0.43727081418037417, "rewards/IngredientFormatReward/mean": 0.9856250047683716, "rewards/IngredientFormatReward/std": 0.10253802984952927, "rewards/IngredientMatchReward/mean": 0.6059505343437195, "rewards/IngredientMatchReward/std": 0.2872993677854538, "rewards/IngredientQuantityMatchReward/mean": 0.5570067405700684, "rewards/IngredientQuantityMatchReward/std": 0.44095311164855955, "rewards/TotalKcalExactMatchReward/mean": 0.821875, "rewards/TotalKcalExactMatchReward/std": 0.3698591083288193, "step": 1020 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.028125, "completions/max_length": 507.8, "completions/mean_length": 396.3828125, "completions/min_length": 228.4, "epoch": 0.2851182197496523, "frac_reward_zero_std": 0.05, "grad_norm": 0.6201499700546265, "kl": 0.03740023162681609, "learning_rate": 8.563878228249685e-07, "loss": 0.001496100425720215, "reward": 2.8762183666229246, "reward_std": 0.49682642221450807, "rewards/IngredientFormatReward/mean": 0.9622656106948853, "rewards/IngredientFormatReward/std": 0.14582059532403946, "rewards/IngredientMatchReward/mean": 0.5401600360870361, "rewards/IngredientMatchReward/std": 0.3013901233673096, "rewards/IngredientQuantityMatchReward/mean": 0.6019176304340362, "rewards/IngredientQuantityMatchReward/std": 0.43271024227142335, "rewards/TotalKcalExactMatchReward/mean": 0.771875, "rewards/TotalKcalExactMatchReward/std": 0.4094280481338501, "step": 1025 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 513.0, "completions/mean_length": 399.1984375, "completions/min_length": 283.4, "epoch": 0.2865090403337969, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6777713894844055, "kl": 0.037024815729819235, "learning_rate": 8.547709630674908e-07, "loss": 0.0014808710664510726, "reward": 2.9829100131988526, "reward_std": 0.47250267267227175, "rewards/IngredientFormatReward/mean": 0.9870833396911621, "rewards/IngredientFormatReward/std": 0.10598021894693374, "rewards/IngredientMatchReward/mean": 0.619783627986908, "rewards/IngredientMatchReward/std": 0.283379340171814, "rewards/IngredientQuantityMatchReward/mean": 0.5760430872440339, "rewards/IngredientQuantityMatchReward/std": 0.4405200660228729, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.3969113349914551, "step": 1030 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 511.0, "completions/mean_length": 397.3640625, "completions/min_length": 270.2, "epoch": 0.28789986091794156, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6470330357551575, "kl": 0.039516926556825635, "learning_rate": 8.531465973604944e-07, "loss": 0.0015805866569280624, "reward": 2.9705513954162597, "reward_std": 0.47205381989479067, "rewards/IngredientFormatReward/mean": 0.9770312547683716, "rewards/IngredientFormatReward/std": 0.12652091942727567, "rewards/IngredientMatchReward/mean": 0.5878050565719605, "rewards/IngredientMatchReward/std": 0.2737370431423187, "rewards/IngredientQuantityMatchReward/mean": 0.6104025483131409, "rewards/IngredientQuantityMatchReward/std": 0.4259393811225891, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.3765485107898712, "step": 1035 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 509.6, "completions/mean_length": 396.15, "completions/min_length": 267.6, "epoch": 0.28929068150208626, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7152348160743713, "kl": 0.03850674827117473, "learning_rate": 8.515147600709603e-07, "loss": 0.0015403028577566148, "reward": 2.8930485248565674, "reward_std": 0.48774529099464414, "rewards/IngredientFormatReward/mean": 0.9825520753860474, "rewards/IngredientFormatReward/std": 0.11014169603586196, "rewards/IngredientMatchReward/mean": 0.5573319673538208, "rewards/IngredientMatchReward/std": 0.3069951593875885, "rewards/IngredientQuantityMatchReward/mean": 0.5453519821166992, "rewards/IngredientQuantityMatchReward/std": 0.4497275948524475, "rewards/TotalKcalExactMatchReward/mean": 0.8078125, "rewards/TotalKcalExactMatchReward/std": 0.38788450360298155, "step": 1040 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 513.0, "completions/mean_length": 401.728125, "completions/min_length": 243.6, "epoch": 0.2906815020862309, "frac_reward_zero_std": 0.025, "grad_norm": 0.714313805103302, "kl": 0.07018633377738297, "learning_rate": 8.498754857239471e-07, "loss": 0.0028061190620064735, "reward": 2.914971446990967, "reward_std": 0.48313209414482117, "rewards/IngredientFormatReward/mean": 0.9662500023841858, "rewards/IngredientFormatReward/std": 0.17367809414863586, "rewards/IngredientMatchReward/mean": 0.6251264691352845, "rewards/IngredientMatchReward/std": 0.3060845613479614, "rewards/IngredientQuantityMatchReward/mean": 0.6126574218273163, "rewards/IngredientQuantityMatchReward/std": 0.4163425207138062, "rewards/TotalKcalExactMatchReward/mean": 0.7109375, "rewards/TotalKcalExactMatchReward/std": 0.4510436117649078, "step": 1045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 509.8, "completions/mean_length": 387.565625, "completions/min_length": 240.4, "epoch": 0.29207232267037553, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6646053791046143, "kl": 0.040469748759642245, "learning_rate": 8.482288090018608e-07, "loss": 0.0016189534217119217, "reward": 3.0161834716796876, "reward_std": 0.4042413294315338, "rewards/IngredientFormatReward/mean": 0.98828125, "rewards/IngredientFormatReward/std": 0.07882513403892517, "rewards/IngredientMatchReward/mean": 0.6237146496772766, "rewards/IngredientMatchReward/std": 0.2907561004161835, "rewards/IngredientQuantityMatchReward/mean": 0.5995000958442688, "rewards/IngredientQuantityMatchReward/std": 0.41379945278167723, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.39606988430023193, "step": 1050 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 509.6, "completions/mean_length": 398.89375, "completions/min_length": 272.2, "epoch": 0.29346314325452016, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7717159986495972, "kl": 0.03940560757182539, "learning_rate": 8.465747647437204e-07, "loss": 0.0015761561691761017, "reward": 3.0119256019592284, "reward_std": 0.4466732025146484, "rewards/IngredientFormatReward/mean": 0.982239592075348, "rewards/IngredientFormatReward/std": 0.0915367141366005, "rewards/IngredientMatchReward/mean": 0.6272364735603333, "rewards/IngredientMatchReward/std": 0.3005757510662079, "rewards/IngredientQuantityMatchReward/mean": 0.5852620124816894, "rewards/IngredientQuantityMatchReward/std": 0.4200045645236969, "rewards/TotalKcalExactMatchReward/mean": 0.8171875, "rewards/TotalKcalExactMatchReward/std": 0.37832899689674376, "step": 1055 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 509.8, "completions/mean_length": 396.9375, "completions/min_length": 273.2, "epoch": 0.2948539638386648, "frac_reward_zero_std": 0.0375, "grad_norm": 0.73525470495224, "kl": 0.04403948448598385, "learning_rate": 8.44913387944421e-07, "loss": 0.0017619559541344643, "reward": 2.986718940734863, "reward_std": 0.44211495518684385, "rewards/IngredientFormatReward/mean": 0.9837351202964782, "rewards/IngredientFormatReward/std": 0.10179438814520836, "rewards/IngredientMatchReward/mean": 0.6092330098152161, "rewards/IngredientMatchReward/std": 0.3025040805339813, "rewards/IngredientQuantityMatchReward/mean": 0.6500007748603821, "rewards/IngredientQuantityMatchReward/std": 0.4024610221385956, "rewards/TotalKcalExactMatchReward/mean": 0.74375, "rewards/TotalKcalExactMatchReward/std": 0.4365465521812439, "step": 1060 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 511.2, "completions/mean_length": 400.3953125, "completions/min_length": 285.6, "epoch": 0.29624478442280944, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7194033265113831, "kl": 0.04031976000405848, "learning_rate": 8.43244713753994e-07, "loss": 0.0016128033399581908, "reward": 2.9168705463409426, "reward_std": 0.48592537045478823, "rewards/IngredientFormatReward/mean": 0.9639881014823913, "rewards/IngredientFormatReward/std": 0.151922045648098, "rewards/IngredientMatchReward/mean": 0.6120442748069763, "rewards/IngredientMatchReward/std": 0.30550934076309205, "rewards/IngredientQuantityMatchReward/mean": 0.5580257177352905, "rewards/IngredientQuantityMatchReward/std": 0.4364677667617798, "rewards/TotalKcalExactMatchReward/mean": 0.7828125, "rewards/TotalKcalExactMatchReward/std": 0.3887697756290436, "step": 1065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 506.8, "completions/mean_length": 390.7953125, "completions/min_length": 247.0, "epoch": 0.29763560500695413, "frac_reward_zero_std": 0.05, "grad_norm": 0.7260541319847107, "kl": 0.03924874525982887, "learning_rate": 8.415687774768626e-07, "loss": 0.001570044457912445, "reward": 3.0964519023895263, "reward_std": 0.43312625885009765, "rewards/IngredientFormatReward/mean": 0.9788541674613953, "rewards/IngredientFormatReward/std": 0.12116530686616897, "rewards/IngredientMatchReward/mean": 0.6652752995491028, "rewards/IngredientMatchReward/std": 0.2774786114692688, "rewards/IngredientQuantityMatchReward/mean": 0.6491973519325256, "rewards/IngredientQuantityMatchReward/std": 0.42371960282325744, "rewards/TotalKcalExactMatchReward/mean": 0.803125, "rewards/TotalKcalExactMatchReward/std": 0.3943279206752777, "step": 1070 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 506.2, "completions/mean_length": 397.975, "completions/min_length": 260.6, "epoch": 0.29902642559109877, "frac_reward_zero_std": 0.05, "grad_norm": 0.6908602118492126, "kl": 0.038830549083650114, "learning_rate": 8.39885614571095e-07, "loss": 0.0015533534809947015, "reward": 2.9613656997680664, "reward_std": 0.4897449791431427, "rewards/IngredientFormatReward/mean": 0.966796875, "rewards/IngredientFormatReward/std": 0.14599068462848663, "rewards/IngredientMatchReward/mean": 0.6265476107597351, "rewards/IngredientMatchReward/std": 0.28194483518600466, "rewards/IngredientQuantityMatchReward/mean": 0.6023962616920471, "rewards/IngredientQuantityMatchReward/std": 0.43221556544303896, "rewards/TotalKcalExactMatchReward/mean": 0.765625, "rewards/TotalKcalExactMatchReward/std": 0.409938383102417, "step": 1075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 505.8, "completions/mean_length": 395.203125, "completions/min_length": 283.6, "epoch": 0.3004172461752434, "frac_reward_zero_std": 0.05, "grad_norm": 0.7245336771011353, "kl": 0.038045503990724684, "learning_rate": 8.38195260647655e-07, "loss": 0.0015217063948512078, "reward": 3.103727436065674, "reward_std": 0.4260666728019714, "rewards/IngredientFormatReward/mean": 0.9903125047683716, "rewards/IngredientFormatReward/std": 0.06337217316031456, "rewards/IngredientMatchReward/mean": 0.6669407367706299, "rewards/IngredientMatchReward/std": 0.2760904639959335, "rewards/IngredientQuantityMatchReward/mean": 0.6402241945266723, "rewards/IngredientQuantityMatchReward/std": 0.41369547247886657, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.38739696741104124, "step": 1080 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 513.0, "completions/mean_length": 392.4046875, "completions/min_length": 261.0, "epoch": 0.30180806675938804, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6876221299171448, "kl": 0.041109291440807286, "learning_rate": 8.364977514696474e-07, "loss": 0.0016444934532046317, "reward": 2.9648212432861327, "reward_std": 0.4610382914543152, "rewards/IngredientFormatReward/mean": 0.9824032783508301, "rewards/IngredientFormatReward/std": 0.11955612003803254, "rewards/IngredientMatchReward/mean": 0.5959220051765441, "rewards/IngredientMatchReward/std": 0.2947571128606796, "rewards/IngredientQuantityMatchReward/mean": 0.5974334716796875, "rewards/IngredientQuantityMatchReward/std": 0.4393026113510132, "rewards/TotalKcalExactMatchReward/mean": 0.7890625, "rewards/TotalKcalExactMatchReward/std": 0.39993913769721984, "step": 1085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 503.6, "completions/mean_length": 394.2390625, "completions/min_length": 263.4, "epoch": 0.3031988873435327, "frac_reward_zero_std": 0.1, "grad_norm": 0.6638349294662476, "kl": 0.04468865282833576, "learning_rate": 8.347931229515624e-07, "loss": 0.0017881166189908982, "reward": 2.975427198410034, "reward_std": 0.3665485680103302, "rewards/IngredientFormatReward/mean": 0.9866145730018616, "rewards/IngredientFormatReward/std": 0.10305822119116784, "rewards/IngredientMatchReward/mean": 0.6167863607406616, "rewards/IngredientMatchReward/std": 0.2862016916275024, "rewards/IngredientQuantityMatchReward/mean": 0.6376512527465821, "rewards/IngredientQuantityMatchReward/std": 0.4232665181159973, "rewards/TotalKcalExactMatchReward/mean": 0.734375, "rewards/TotalKcalExactMatchReward/std": 0.43666509389877317, "step": 1090 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 505.4, "completions/mean_length": 389.9375, "completions/min_length": 274.2, "epoch": 0.3045897079276773, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6407614946365356, "kl": 0.044913537241518496, "learning_rate": 8.330814111585147e-07, "loss": 0.0017966246232390404, "reward": 2.969407796859741, "reward_std": 0.4416906535625458, "rewards/IngredientFormatReward/mean": 0.9874553680419922, "rewards/IngredientFormatReward/std": 0.09603629484772683, "rewards/IngredientMatchReward/mean": 0.6119326591491699, "rewards/IngredientMatchReward/std": 0.3116526186466217, "rewards/IngredientQuantityMatchReward/mean": 0.6106447577476501, "rewards/IngredientQuantityMatchReward/std": 0.42712777853012085, "rewards/TotalKcalExactMatchReward/mean": 0.759375, "rewards/TotalKcalExactMatchReward/std": 0.42226200103759765, "step": 1095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 502.2, "completions/mean_length": 389.5390625, "completions/min_length": 252.4, "epoch": 0.30598052851182195, "frac_reward_zero_std": 0.05, "grad_norm": 0.65201735496521, "kl": 0.039539157832041386, "learning_rate": 8.313626523054817e-07, "loss": 0.001581670343875885, "reward": 3.082692050933838, "reward_std": 0.4257974922657013, "rewards/IngredientFormatReward/mean": 0.9868750095367431, "rewards/IngredientFormatReward/std": 0.07427057325839996, "rewards/IngredientMatchReward/mean": 0.6660503268241882, "rewards/IngredientMatchReward/std": 0.2906301259994507, "rewards/IngredientQuantityMatchReward/mean": 0.6297666907310486, "rewards/IngredientQuantityMatchReward/std": 0.4138112425804138, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.4003869414329529, "step": 1100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 508.6, "completions/mean_length": 387.5453125, "completions/min_length": 263.6, "epoch": 0.30737134909596664, "frac_reward_zero_std": 0.0375, "grad_norm": 0.5957731604576111, "kl": 0.04335511876270175, "learning_rate": 8.296368827565364e-07, "loss": 0.0017340987920761108, "reward": 2.981190013885498, "reward_std": 0.45507683157920836, "rewards/IngredientFormatReward/mean": 0.9846875190734863, "rewards/IngredientFormatReward/std": 0.10660690441727638, "rewards/IngredientMatchReward/mean": 0.5576079070568085, "rewards/IngredientMatchReward/std": 0.27225913405418395, "rewards/IngredientQuantityMatchReward/mean": 0.6248321235179901, "rewards/IngredientQuantityMatchReward/std": 0.41838674545288085, "rewards/TotalKcalExactMatchReward/mean": 0.8140625, "rewards/TotalKcalExactMatchReward/std": 0.3690988004207611, "step": 1105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 507.0, "completions/mean_length": 393.3875, "completions/min_length": 273.8, "epoch": 0.3087621696801113, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6201445460319519, "kl": 0.042535467795096336, "learning_rate": 8.279041390240781e-07, "loss": 0.0017013853415846826, "reward": 3.088879632949829, "reward_std": 0.35922372341156006, "rewards/IngredientFormatReward/mean": 0.9940624952316284, "rewards/IngredientFormatReward/std": 0.04400581270456314, "rewards/IngredientMatchReward/mean": 0.6234399914741516, "rewards/IngredientMatchReward/std": 0.27694119811058043, "rewards/IngredientQuantityMatchReward/mean": 0.6651272177696228, "rewards/IngredientQuantityMatchReward/std": 0.42454165816307066, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.3904335916042328, "step": 1110 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 508.8, "completions/mean_length": 387.084375, "completions/min_length": 260.0, "epoch": 0.3101529902642559, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7132757306098938, "kl": 0.04249203451909125, "learning_rate": 8.261644577680602e-07, "loss": 0.0017001090571284294, "reward": 3.0332549095153807, "reward_std": 0.4516888916492462, "rewards/IngredientFormatReward/mean": 0.9883593797683716, "rewards/IngredientFormatReward/std": 0.08581716865301132, "rewards/IngredientMatchReward/mean": 0.619999372959137, "rewards/IngredientMatchReward/std": 0.2996371924877167, "rewards/IngredientQuantityMatchReward/mean": 0.6498961329460144, "rewards/IngredientQuantityMatchReward/std": 0.4136515438556671, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.4019003748893738, "step": 1115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.2, "completions/mean_length": 393.415625, "completions/min_length": 265.8, "epoch": 0.31154381084840055, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6780772805213928, "kl": 0.04389502888079733, "learning_rate": 8.244178757952147e-07, "loss": 0.0017559628933668137, "reward": 3.0198182582855226, "reward_std": 0.41773988008499147, "rewards/IngredientFormatReward/mean": 0.9877343773841858, "rewards/IngredientFormatReward/std": 0.09690560251474381, "rewards/IngredientMatchReward/mean": 0.6150545716285706, "rewards/IngredientMatchReward/std": 0.27234477996826173, "rewards/IngredientQuantityMatchReward/mean": 0.663904333114624, "rewards/IngredientQuantityMatchReward/std": 0.39827964901924134, "rewards/TotalKcalExactMatchReward/mean": 0.753125, "rewards/TotalKcalExactMatchReward/std": 0.4100442737340927, "step": 1120 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 511.2, "completions/mean_length": 386.2671875, "completions/min_length": 242.4, "epoch": 0.3129346314325452, "frac_reward_zero_std": 0.0625, "grad_norm": 0.689258337020874, "kl": 0.04299152567982674, "learning_rate": 8.226644300582729e-07, "loss": 0.0017199600115418434, "reward": 3.042409086227417, "reward_std": 0.4061857044696808, "rewards/IngredientFormatReward/mean": 0.9890625, "rewards/IngredientFormatReward/std": 0.09063329249620437, "rewards/IngredientMatchReward/mean": 0.6030859589576721, "rewards/IngredientMatchReward/std": 0.3205138087272644, "rewards/IngredientQuantityMatchReward/mean": 0.6033856332302093, "rewards/IngredientQuantityMatchReward/std": 0.4283260703086853, "rewards/TotalKcalExactMatchReward/mean": 0.846875, "rewards/TotalKcalExactMatchReward/std": 0.3567110180854797, "step": 1125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 510.2, "completions/mean_length": 389.278125, "completions/min_length": 264.0, "epoch": 0.3143254520166898, "frac_reward_zero_std": 0.025, "grad_norm": 0.7026454210281372, "kl": 0.043634300702251494, "learning_rate": 8.209041576551841e-07, "loss": 0.0017453931272029878, "reward": 3.039749765396118, "reward_std": 0.39314189553260803, "rewards/IngredientFormatReward/mean": 0.9890625, "rewards/IngredientFormatReward/std": 0.09063329249620437, "rewards/IngredientMatchReward/mean": 0.6170356154441834, "rewards/IngredientMatchReward/std": 0.28204314708709716, "rewards/IngredientQuantityMatchReward/mean": 0.6649016618728638, "rewards/IngredientQuantityMatchReward/std": 0.41771509647369387, "rewards/TotalKcalExactMatchReward/mean": 0.76875, "rewards/TotalKcalExactMatchReward/std": 0.412118011713028, "step": 1130 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 507.6, "completions/mean_length": 390.76875, "completions/min_length": 257.2, "epoch": 0.3157162726008345, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6993815302848816, "kl": 0.04059697471093386, "learning_rate": 8.191370958283304e-07, "loss": 0.0016238151118159294, "reward": 3.169062376022339, "reward_std": 0.3827301383018494, "rewards/IngredientFormatReward/mean": 0.99453125, "rewards/IngredientFormatReward/std": 0.059600303322076796, "rewards/IngredientMatchReward/mean": 0.6575991868972778, "rewards/IngredientMatchReward/std": 0.27744790315628054, "rewards/IngredientQuantityMatchReward/mean": 0.6934943556785583, "rewards/IngredientQuantityMatchReward/std": 0.39465083479881286, "rewards/TotalKcalExactMatchReward/mean": 0.8234375, "rewards/TotalKcalExactMatchReward/std": 0.37338051199913025, "step": 1135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 503.4, "completions/mean_length": 395.5109375, "completions/min_length": 277.8, "epoch": 0.31710709318497915, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6774301528930664, "kl": 0.03972219394054264, "learning_rate": 8.173632819637388e-07, "loss": 0.001589289866387844, "reward": 3.1047873497009277, "reward_std": 0.393167108297348, "rewards/IngredientFormatReward/mean": 0.9953125, "rewards/IngredientFormatReward/std": 0.04257904887199402, "rewards/IngredientMatchReward/mean": 0.6563039422035217, "rewards/IngredientMatchReward/std": 0.29307783842086793, "rewards/IngredientQuantityMatchReward/mean": 0.618795907497406, "rewards/IngredientQuantityMatchReward/std": 0.43244861960411074, "rewards/TotalKcalExactMatchReward/mean": 0.834375, "rewards/TotalKcalExactMatchReward/std": 0.3656123340129852, "step": 1140 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 508.8, "completions/mean_length": 387.9, "completions/min_length": 241.2, "epoch": 0.3184979137691238, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6518785357475281, "kl": 0.044394610333256423, "learning_rate": 8.155827535902911e-07, "loss": 0.0017760045826435088, "reward": 3.0225061416625976, "reward_std": 0.3792412966489792, "rewards/IngredientFormatReward/mean": 0.9885416746139526, "rewards/IngredientFormatReward/std": 0.09652584828436375, "rewards/IngredientMatchReward/mean": 0.6106491684913635, "rewards/IngredientMatchReward/std": 0.28019180297851565, "rewards/IngredientQuantityMatchReward/mean": 0.6498778104782105, "rewards/IngredientQuantityMatchReward/std": 0.4284306287765503, "rewards/TotalKcalExactMatchReward/mean": 0.7734375, "rewards/TotalKcalExactMatchReward/std": 0.3934566378593445, "step": 1145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 512.2, "completions/mean_length": 395.36875, "completions/min_length": 268.2, "epoch": 0.3198887343532684, "frac_reward_zero_std": 0.05, "grad_norm": 0.75348961353302, "kl": 0.04309874954633415, "learning_rate": 8.137955483789278e-07, "loss": 0.0017238954082131387, "reward": 3.0311806201934814, "reward_std": 0.4421843349933624, "rewards/IngredientFormatReward/mean": 0.9781510472297669, "rewards/IngredientFormatReward/std": 0.11678568720817566, "rewards/IngredientMatchReward/mean": 0.6269333004951477, "rewards/IngredientMatchReward/std": 0.28258291482925413, "rewards/IngredientQuantityMatchReward/mean": 0.6464088559150696, "rewards/IngredientQuantityMatchReward/std": 0.40041940212249755, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.4114561975002289, "step": 1150 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 509.8, "completions/mean_length": 394.959375, "completions/min_length": 259.4, "epoch": 0.32127955493741306, "frac_reward_zero_std": 0.075, "grad_norm": 0.6728385090827942, "kl": 0.041832944145426154, "learning_rate": 8.120017041418537e-07, "loss": 0.0016738016158342362, "reward": 3.038546895980835, "reward_std": 0.36745981574058534, "rewards/IngredientFormatReward/mean": 0.990625, "rewards/IngredientFormatReward/std": 0.08340958207845688, "rewards/IngredientMatchReward/mean": 0.6225868225097656, "rewards/IngredientMatchReward/std": 0.2717043608427048, "rewards/IngredientQuantityMatchReward/mean": 0.617522656917572, "rewards/IngredientQuantityMatchReward/std": 0.4125868022441864, "rewards/TotalKcalExactMatchReward/mean": 0.8078125, "rewards/TotalKcalExactMatchReward/std": 0.3869707822799683, "step": 1155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 506.2, "completions/mean_length": 388.2078125, "completions/min_length": 245.4, "epoch": 0.3226703755215577, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6846386194229126, "kl": 0.053002631710842255, "learning_rate": 8.102012588317355e-07, "loss": 0.002119516022503376, "reward": 3.0419249057769777, "reward_std": 0.4614200294017792, "rewards/IngredientFormatReward/mean": 0.9736830472946167, "rewards/IngredientFormatReward/std": 0.13447005301713943, "rewards/IngredientMatchReward/mean": 0.6277988672256469, "rewards/IngredientMatchReward/std": 0.2982490539550781, "rewards/IngredientQuantityMatchReward/mean": 0.6357555866241456, "rewards/IngredientQuantityMatchReward/std": 0.4329476416110992, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.38714192509651185, "step": 1160 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 513.0, "completions/mean_length": 396.6859375, "completions/min_length": 257.2, "epoch": 0.3240611961057024, "frac_reward_zero_std": 0.025, "grad_norm": 0.6272140145301819, "kl": 0.047240815870463845, "learning_rate": 8.083942505409008e-07, "loss": 0.0018897855654358864, "reward": 2.928142023086548, "reward_std": 0.4312022864818573, "rewards/IngredientFormatReward/mean": 0.985740327835083, "rewards/IngredientFormatReward/std": 0.10639770179986954, "rewards/IngredientMatchReward/mean": 0.5962034702301026, "rewards/IngredientMatchReward/std": 0.29347813725471494, "rewards/IngredientQuantityMatchReward/mean": 0.557135808467865, "rewards/IngredientQuantityMatchReward/std": 0.4391862154006958, "rewards/TotalKcalExactMatchReward/mean": 0.7890625, "rewards/TotalKcalExactMatchReward/std": 0.4047076404094696, "step": 1165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 512.6, "completions/mean_length": 405.865625, "completions/min_length": 270.8, "epoch": 0.325452016689847, "frac_reward_zero_std": 0.05, "grad_norm": 0.7186459302902222, "kl": 0.04450485638808459, "learning_rate": 8.065807175005306e-07, "loss": 0.0017805084586143493, "reward": 3.0247393608093263, "reward_std": 0.43825086355209353, "rewards/IngredientFormatReward/mean": 0.9800000071525574, "rewards/IngredientFormatReward/std": 0.11633365899324417, "rewards/IngredientMatchReward/mean": 0.6262376070022583, "rewards/IngredientMatchReward/std": 0.2948880285024643, "rewards/IngredientQuantityMatchReward/mean": 0.6560018181800842, "rewards/IngredientQuantityMatchReward/std": 0.3940140724182129, "rewards/TotalKcalExactMatchReward/mean": 0.7625, "rewards/TotalKcalExactMatchReward/std": 0.4198298633098602, "step": 1170 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.028125, "completions/max_length": 513.0, "completions/mean_length": 403.5625, "completions/min_length": 279.8, "epoch": 0.32684283727399166, "frac_reward_zero_std": 0.0375, "grad_norm": 0.652772843837738, "kl": 0.045967866806313394, "learning_rate": 8.047606980798515e-07, "loss": 0.001838766410946846, "reward": 2.9285586357116697, "reward_std": 0.4667127549648285, "rewards/IngredientFormatReward/mean": 0.9732142925262451, "rewards/IngredientFormatReward/std": 0.1576545402407646, "rewards/IngredientMatchReward/mean": 0.6107007622718811, "rewards/IngredientMatchReward/std": 0.2908420205116272, "rewards/IngredientQuantityMatchReward/mean": 0.6087061285972595, "rewards/IngredientQuantityMatchReward/std": 0.40912932753562925, "rewards/TotalKcalExactMatchReward/mean": 0.7359375, "rewards/TotalKcalExactMatchReward/std": 0.4145884871482849, "step": 1175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0265625, "completions/max_length": 511.4, "completions/mean_length": 394.40625, "completions/min_length": 263.8, "epoch": 0.3282336578581363, "frac_reward_zero_std": 0.05, "grad_norm": 0.625441312789917, "kl": 0.06037823753431439, "learning_rate": 8.029342307853238e-07, "loss": 0.0024156935513019564, "reward": 3.06034779548645, "reward_std": 0.47609464526176454, "rewards/IngredientFormatReward/mean": 0.9684374928474426, "rewards/IngredientFormatReward/std": 0.14431809410452842, "rewards/IngredientMatchReward/mean": 0.6360243082046508, "rewards/IngredientMatchReward/std": 0.2919568657875061, "rewards/IngredientQuantityMatchReward/mean": 0.6730734944343567, "rewards/IngredientQuantityMatchReward/std": 0.4052798330783844, "rewards/TotalKcalExactMatchReward/mean": 0.7828125, "rewards/TotalKcalExactMatchReward/std": 0.4118169665336609, "step": 1180 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 499.0, "completions/mean_length": 388.5265625, "completions/min_length": 250.0, "epoch": 0.32962447844228093, "frac_reward_zero_std": 0.025, "grad_norm": 0.6767213940620422, "kl": 0.04451586329378188, "learning_rate": 8.011013542598258e-07, "loss": 0.0017808981239795685, "reward": 3.0574079513549806, "reward_std": 0.4540375053882599, "rewards/IngredientFormatReward/mean": 0.9915625095367432, "rewards/IngredientFormatReward/std": 0.06801376938819885, "rewards/IngredientMatchReward/mean": 0.6417565941810608, "rewards/IngredientMatchReward/std": 0.28428298234939575, "rewards/IngredientQuantityMatchReward/mean": 0.6647138357162475, "rewards/IngredientQuantityMatchReward/std": 0.4178314387798309, "rewards/TotalKcalExactMatchReward/mean": 0.759375, "rewards/TotalKcalExactMatchReward/std": 0.4165681004524231, "step": 1185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.034375, "completions/max_length": 510.0, "completions/mean_length": 400.0953125, "completions/min_length": 257.8, "epoch": 0.33101529902642557, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7180148363113403, "kl": 0.04539623921737075, "learning_rate": 7.992621072818375e-07, "loss": 0.0018160387873649598, "reward": 3.049036645889282, "reward_std": 0.5000029981136322, "rewards/IngredientFormatReward/mean": 0.9583556532859803, "rewards/IngredientFormatReward/std": 0.1660301133990288, "rewards/IngredientMatchReward/mean": 0.6038318634033203, "rewards/IngredientMatchReward/std": 0.29093838334083555, "rewards/IngredientQuantityMatchReward/mean": 0.6680991888046265, "rewards/IngredientQuantityMatchReward/std": 0.4054907917976379, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.37279661893844607, "step": 1190 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 510.0, "completions/mean_length": 392.565625, "completions/min_length": 250.8, "epoch": 0.33240611961057026, "frac_reward_zero_std": 0.05, "grad_norm": 0.7087981700897217, "kl": 0.04504902786575258, "learning_rate": 7.974165287646199e-07, "loss": 0.0018019940704107284, "reward": 3.0593740463256838, "reward_std": 0.4065113961696625, "rewards/IngredientFormatReward/mean": 0.9820684552192688, "rewards/IngredientFormatReward/std": 0.1045232132077217, "rewards/IngredientMatchReward/mean": 0.6230959236621857, "rewards/IngredientMatchReward/std": 0.30069578289985655, "rewards/IngredientQuantityMatchReward/mean": 0.6588971614837646, "rewards/IngredientQuantityMatchReward/std": 0.40588967204093934, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.38163130879402163, "step": 1195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 502.2, "completions/mean_length": 392.3703125, "completions/min_length": 242.8, "epoch": 0.3337969401947149, "frac_reward_zero_std": 0.0875, "grad_norm": 0.7028740048408508, "kl": 0.044297435646876694, "learning_rate": 7.955646577553909e-07, "loss": 0.001771971769630909, "reward": 2.929919147491455, "reward_std": 0.3947701930999756, "rewards/IngredientFormatReward/mean": 0.9732142925262451, "rewards/IngredientFormatReward/std": 0.1214448481798172, "rewards/IngredientMatchReward/mean": 0.6054377436637879, "rewards/IngredientMatchReward/std": 0.2721258968114853, "rewards/IngredientQuantityMatchReward/mean": 0.612204658985138, "rewards/IngredientQuantityMatchReward/std": 0.41552995443344115, "rewards/TotalKcalExactMatchReward/mean": 0.7390625, "rewards/TotalKcalExactMatchReward/std": 0.43984751105308534, "step": 1200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 511.4, "completions/mean_length": 393.684375, "completions/min_length": 260.8, "epoch": 0.33518776077885953, "frac_reward_zero_std": 0.075, "grad_norm": 0.717398464679718, "kl": 0.04384518137667328, "learning_rate": 7.937065334345002e-07, "loss": 0.001753794401884079, "reward": 3.049497127532959, "reward_std": 0.421299946308136, "rewards/IngredientFormatReward/mean": 0.985546886920929, "rewards/IngredientFormatReward/std": 0.10045162588357925, "rewards/IngredientMatchReward/mean": 0.6495393276214599, "rewards/IngredientMatchReward/std": 0.29075068831443784, "rewards/IngredientQuantityMatchReward/mean": 0.670660924911499, "rewards/IngredientQuantityMatchReward/std": 0.41321573257446287, "rewards/TotalKcalExactMatchReward/mean": 0.74375, "rewards/TotalKcalExactMatchReward/std": 0.43112467527389525, "step": 1205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 513.0, "completions/mean_length": 396.015625, "completions/min_length": 255.6, "epoch": 0.33657858136300417, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7146852016448975, "kl": 0.11223286460153759, "learning_rate": 7.918421951145992e-07, "loss": 0.004489917308092117, "reward": 3.091847562789917, "reward_std": 0.3895574867725372, "rewards/IngredientFormatReward/mean": 0.9815104007720947, "rewards/IngredientFormatReward/std": 0.11061891429126262, "rewards/IngredientMatchReward/mean": 0.6332726240158081, "rewards/IngredientMatchReward/std": 0.284361732006073, "rewards/IngredientQuantityMatchReward/mean": 0.6911271452903748, "rewards/IngredientQuantityMatchReward/std": 0.3966558754444122, "rewards/TotalKcalExactMatchReward/mean": 0.7859375, "rewards/TotalKcalExactMatchReward/std": 0.3858437240123749, "step": 1210 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 506.0, "completions/mean_length": 391.6609375, "completions/min_length": 269.2, "epoch": 0.3379694019471488, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6622430086135864, "kl": 0.04423058428801596, "learning_rate": 7.899716822398107e-07, "loss": 0.001769232377409935, "reward": 3.026241827011108, "reward_std": 0.39915978312492373, "rewards/IngredientFormatReward/mean": 0.9831510305404663, "rewards/IngredientFormatReward/std": 0.09449614509940148, "rewards/IngredientMatchReward/mean": 0.6212599039077759, "rewards/IngredientMatchReward/std": 0.2815777659416199, "rewards/IngredientQuantityMatchReward/mean": 0.6343308687210083, "rewards/IngredientQuantityMatchReward/std": 0.4104671120643616, "rewards/TotalKcalExactMatchReward/mean": 0.7875, "rewards/TotalKcalExactMatchReward/std": 0.4089196860790253, "step": 1215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0265625, "completions/max_length": 513.0, "completions/mean_length": 399.0421875, "completions/min_length": 268.8, "epoch": 0.33936022253129344, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6995375752449036, "kl": 0.046378902695141735, "learning_rate": 7.880950343848933e-07, "loss": 0.001855262741446495, "reward": 2.894041156768799, "reward_std": 0.4899939954280853, "rewards/IngredientFormatReward/mean": 0.9703645706176758, "rewards/IngredientFormatReward/std": 0.15981422364711761, "rewards/IngredientMatchReward/mean": 0.6005760192871094, "rewards/IngredientMatchReward/std": 0.29874255061149596, "rewards/IngredientQuantityMatchReward/mean": 0.5606004893779755, "rewards/IngredientQuantityMatchReward/std": 0.4271669626235962, "rewards/TotalKcalExactMatchReward/mean": 0.7625, "rewards/TotalKcalExactMatchReward/std": 0.4213130235671997, "step": 1220 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 508.6, "completions/mean_length": 402.053125, "completions/min_length": 265.8, "epoch": 0.34075104311543813, "frac_reward_zero_std": 0.025, "grad_norm": 0.6287099719047546, "kl": 0.3435618258547038, "learning_rate": 7.862122912544041e-07, "loss": 0.013743305206298828, "reward": 3.0836498737335205, "reward_std": 0.4236464619636536, "rewards/IngredientFormatReward/mean": 0.9829278349876404, "rewards/IngredientFormatReward/std": 0.11173093281686305, "rewards/IngredientMatchReward/mean": 0.6336050748825073, "rewards/IngredientMatchReward/std": 0.2927217662334442, "rewards/IngredientQuantityMatchReward/mean": 0.6374294281005859, "rewards/IngredientQuantityMatchReward/std": 0.40337653160095216, "rewards/TotalKcalExactMatchReward/mean": 0.8296875, "rewards/TotalKcalExactMatchReward/std": 0.3659061789512634, "step": 1225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 504.2, "completions/mean_length": 395.1078125, "completions/min_length": 247.0, "epoch": 0.34214186369958277, "frac_reward_zero_std": 0.025, "grad_norm": 0.6511803269386292, "kl": 0.043997203628532586, "learning_rate": 7.843234926818594e-07, "loss": 0.001759953796863556, "reward": 2.9981056690216064, "reward_std": 0.4053235173225403, "rewards/IngredientFormatReward/mean": 0.981276023387909, "rewards/IngredientFormatReward/std": 0.1175081841647625, "rewards/IngredientMatchReward/mean": 0.5940014004707337, "rewards/IngredientMatchReward/std": 0.2828234672546387, "rewards/IngredientQuantityMatchReward/mean": 0.6415782332420349, "rewards/IngredientQuantityMatchReward/std": 0.4185176372528076, "rewards/TotalKcalExactMatchReward/mean": 0.78125, "rewards/TotalKcalExactMatchReward/std": 0.4035619616508484, "step": 1230 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 509.0, "completions/mean_length": 402.0828125, "completions/min_length": 274.8, "epoch": 0.3435326842837274, "frac_reward_zero_std": 0.05, "grad_norm": 0.8296776413917542, "kl": 0.04468656564131379, "learning_rate": 7.824286786288919e-07, "loss": 0.0017875529825687408, "reward": 3.0703739643096926, "reward_std": 0.4063426792621613, "rewards/IngredientFormatReward/mean": 0.9804166793823242, "rewards/IngredientFormatReward/std": 0.11292581036686897, "rewards/IngredientMatchReward/mean": 0.6367730379104615, "rewards/IngredientMatchReward/std": 0.28677576780319214, "rewards/IngredientQuantityMatchReward/mean": 0.6484968185424804, "rewards/IngredientQuantityMatchReward/std": 0.4008986532688141, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.39468390345573423, "step": 1235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 508.8, "completions/mean_length": 389.0421875, "completions/min_length": 265.6, "epoch": 0.34492350486787204, "frac_reward_zero_std": 0.05, "grad_norm": 0.7355208396911621, "kl": 0.07059146789833903, "learning_rate": 7.805278891844033e-07, "loss": 0.0028071509674191474, "reward": 3.00608606338501, "reward_std": 0.4155315697193146, "rewards/IngredientFormatReward/mean": 0.9853124976158142, "rewards/IngredientFormatReward/std": 0.11259681433439254, "rewards/IngredientMatchReward/mean": 0.6296162247657776, "rewards/IngredientMatchReward/std": 0.3042337715625763, "rewards/IngredientQuantityMatchReward/mean": 0.6302198767662048, "rewards/IngredientQuantityMatchReward/std": 0.43931986689567565, "rewards/TotalKcalExactMatchReward/mean": 0.7609375, "rewards/TotalKcalExactMatchReward/std": 0.42408340573310854, "step": 1240 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 513.0, "completions/mean_length": 401.45625, "completions/min_length": 270.4, "epoch": 0.3463143254520167, "frac_reward_zero_std": 0.075, "grad_norm": 0.7564507126808167, "kl": 0.04235607595182955, "learning_rate": 7.786211645637195e-07, "loss": 0.0016942832618951798, "reward": 3.1188386917114257, "reward_std": 0.4350468873977661, "rewards/IngredientFormatReward/mean": 0.9809375047683716, "rewards/IngredientFormatReward/std": 0.12780899554491043, "rewards/IngredientMatchReward/mean": 0.646865701675415, "rewards/IngredientMatchReward/std": 0.2780866503715515, "rewards/IngredientQuantityMatchReward/mean": 0.6629104495048523, "rewards/IngredientQuantityMatchReward/std": 0.4089032709598541, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.36764142513275144, "step": 1245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 513.0, "completions/mean_length": 404.4109375, "completions/min_length": 265.6, "epoch": 0.3477051460361613, "frac_reward_zero_std": 0.025, "grad_norm": 0.6491013765335083, "kl": 0.04551801586057991, "learning_rate": 7.76708545107737e-07, "loss": 0.0018206179141998292, "reward": 3.0424689769744875, "reward_std": 0.44069910049438477, "rewards/IngredientFormatReward/mean": 0.9783556699752808, "rewards/IngredientFormatReward/std": 0.13486974984407424, "rewards/IngredientMatchReward/mean": 0.6330344676971436, "rewards/IngredientMatchReward/std": 0.29515123963356016, "rewards/IngredientQuantityMatchReward/mean": 0.5982663631439209, "rewards/IngredientQuantityMatchReward/std": 0.4426428616046906, "rewards/TotalKcalExactMatchReward/mean": 0.8328125, "rewards/TotalKcalExactMatchReward/std": 0.37421483397483823, "step": 1250 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0296875, "completions/max_length": 507.4, "completions/mean_length": 401.31875, "completions/min_length": 281.8, "epoch": 0.34909596662030595, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6339758038520813, "kl": 0.045095857931301, "learning_rate": 7.747900712820703e-07, "loss": 0.001803934946656227, "reward": 3.072742462158203, "reward_std": 0.4893173336982727, "rewards/IngredientFormatReward/mean": 0.965625, "rewards/IngredientFormatReward/std": 0.15386179387569426, "rewards/IngredientMatchReward/mean": 0.6180078268051148, "rewards/IngredientMatchReward/std": 0.2904675453901291, "rewards/IngredientQuantityMatchReward/mean": 0.6781721353530884, "rewards/IngredientQuantityMatchReward/std": 0.41462584733963015, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.3824631690979004, "step": 1255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0296875, "completions/max_length": 513.0, "completions/mean_length": 401.753125, "completions/min_length": 249.4, "epoch": 0.35048678720445064, "frac_reward_zero_std": 0.05, "grad_norm": 0.6409927606582642, "kl": 0.04350223378278315, "learning_rate": 7.728657836761959e-07, "loss": 0.0017400886863470078, "reward": 3.152744436264038, "reward_std": 0.45829938650131224, "rewards/IngredientFormatReward/mean": 0.9662500023841858, "rewards/IngredientFormatReward/std": 0.17663300186395645, "rewards/IngredientMatchReward/mean": 0.701597261428833, "rewards/IngredientMatchReward/std": 0.28799739480018616, "rewards/IngredientQuantityMatchReward/mean": 0.6583346724510193, "rewards/IngredientQuantityMatchReward/std": 0.40607632994651793, "rewards/TotalKcalExactMatchReward/mean": 0.8265625, "rewards/TotalKcalExactMatchReward/std": 0.3737199783325195, "step": 1260 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.04375, "completions/max_length": 513.0, "completions/mean_length": 403.8640625, "completions/min_length": 273.0, "epoch": 0.3518776077885953, "frac_reward_zero_std": 0.05, "grad_norm": 0.6250739097595215, "kl": 0.045182203454896805, "learning_rate": 7.709357230025937e-07, "loss": 0.0018076343461871148, "reward": 2.949186897277832, "reward_std": 0.45738943815231325, "rewards/IngredientFormatReward/mean": 0.9545665860176087, "rewards/IngredientFormatReward/std": 0.19560373723506927, "rewards/IngredientMatchReward/mean": 0.6393668532371521, "rewards/IngredientMatchReward/std": 0.3034960389137268, "rewards/IngredientQuantityMatchReward/mean": 0.5896284222602844, "rewards/IngredientQuantityMatchReward/std": 0.4419006407260895, "rewards/TotalKcalExactMatchReward/mean": 0.765625, "rewards/TotalKcalExactMatchReward/std": 0.4109238743782043, "step": 1265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 510.2, "completions/mean_length": 393.5328125, "completions/min_length": 264.8, "epoch": 0.3532684283727399, "frac_reward_zero_std": 0.075, "grad_norm": 0.6803897023200989, "kl": 0.04348803351167589, "learning_rate": 7.689999300958851e-07, "loss": 0.001739520952105522, "reward": 3.1484294891357423, "reward_std": 0.3893423616886139, "rewards/IngredientFormatReward/mean": 0.9821093797683715, "rewards/IngredientFormatReward/std": 0.09933126196265221, "rewards/IngredientMatchReward/mean": 0.6269580841064453, "rewards/IngredientMatchReward/std": 0.3178780019283295, "rewards/IngredientQuantityMatchReward/mean": 0.6643620371818543, "rewards/IngredientQuantityMatchReward/std": 0.40709378123283385, "rewards/TotalKcalExactMatchReward/mean": 0.875, "rewards/TotalKcalExactMatchReward/std": 0.320685601234436, "step": 1270 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 512.6, "completions/mean_length": 396.5296875, "completions/min_length": 276.2, "epoch": 0.35465924895688455, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6363331079483032, "kl": 0.04308997318148613, "learning_rate": 7.670584459119694e-07, "loss": 0.0017237666994333266, "reward": 2.969732904434204, "reward_std": 0.4224898397922516, "rewards/IngredientFormatReward/mean": 0.9850520849227905, "rewards/IngredientFormatReward/std": 0.10458039119839668, "rewards/IngredientMatchReward/mean": 0.568692970275879, "rewards/IngredientMatchReward/std": 0.29052608013153075, "rewards/IngredientQuantityMatchReward/mean": 0.6659879565238953, "rewards/IngredientQuantityMatchReward/std": 0.43609716296195983, "rewards/TotalKcalExactMatchReward/mean": 0.75, "rewards/TotalKcalExactMatchReward/std": 0.4251108765602112, "step": 1275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 508.4, "completions/mean_length": 394.26875, "completions/min_length": 244.4, "epoch": 0.3560500695410292, "frac_reward_zero_std": 0.1125, "grad_norm": 0.7387193441390991, "kl": 0.04524663793854415, "learning_rate": 7.651113115271572e-07, "loss": 0.0018100041896104812, "reward": 2.9772053241729735, "reward_std": 0.3859552562236786, "rewards/IngredientFormatReward/mean": 0.9921875, "rewards/IngredientFormatReward/std": 0.05527795404195786, "rewards/IngredientMatchReward/mean": 0.6179451823234559, "rewards/IngredientMatchReward/std": 0.2892803192138672, "rewards/IngredientQuantityMatchReward/mean": 0.5951976597309112, "rewards/IngredientQuantityMatchReward/std": 0.4320973575115204, "rewards/TotalKcalExactMatchReward/mean": 0.771875, "rewards/TotalKcalExactMatchReward/std": 0.41942211985588074, "step": 1280 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 512.4, "completions/mean_length": 396.628125, "completions/min_length": 286.8, "epoch": 0.3574408901251738, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6847825050354004, "kl": 0.04578770115040243, "learning_rate": 7.631585681373013e-07, "loss": 0.00183718241751194, "reward": 3.059334373474121, "reward_std": 0.3995037257671356, "rewards/IngredientFormatReward/mean": 0.9859635353088378, "rewards/IngredientFormatReward/std": 0.10678645819425583, "rewards/IngredientMatchReward/mean": 0.655921995639801, "rewards/IngredientMatchReward/std": 0.26984080076217654, "rewards/IngredientQuantityMatchReward/mean": 0.5893237948417663, "rewards/IngredientQuantityMatchReward/std": 0.4238016366958618, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.37518492341041565, "step": 1285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 513.0, "completions/mean_length": 404.5703125, "completions/min_length": 256.6, "epoch": 0.3588317107093185, "frac_reward_zero_std": 0.075, "grad_norm": 0.6697726845741272, "kl": 0.05169771097134799, "learning_rate": 7.612002570569254e-07, "loss": 0.0020679634064435957, "reward": 3.087264966964722, "reward_std": 0.41781227588653563, "rewards/IngredientFormatReward/mean": 0.9698958277702332, "rewards/IngredientFormatReward/std": 0.15693533271551133, "rewards/IngredientMatchReward/mean": 0.6202176332473754, "rewards/IngredientMatchReward/std": 0.29056022465229037, "rewards/IngredientQuantityMatchReward/mean": 0.6815265893936158, "rewards/IngredientQuantityMatchReward/std": 0.40200709700584414, "rewards/TotalKcalExactMatchReward/mean": 0.815625, "rewards/TotalKcalExactMatchReward/std": 0.38562028408050536, "step": 1290 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 497.2, "completions/mean_length": 388.734375, "completions/min_length": 261.8, "epoch": 0.36022253129346316, "frac_reward_zero_std": 0.05, "grad_norm": 0.6716930866241455, "kl": 0.04933963837102055, "learning_rate": 7.592364197183494e-07, "loss": 0.001973751001060009, "reward": 3.0686947822570803, "reward_std": 0.42163644433021547, "rewards/IngredientFormatReward/mean": 0.9865829586982727, "rewards/IngredientFormatReward/std": 0.07807523384690285, "rewards/IngredientMatchReward/mean": 0.6540155410766602, "rewards/IngredientMatchReward/std": 0.28045170605182645, "rewards/IngredientQuantityMatchReward/mean": 0.6734087467193604, "rewards/IngredientQuantityMatchReward/std": 0.4206121265888214, "rewards/TotalKcalExactMatchReward/mean": 0.7546875, "rewards/TotalKcalExactMatchReward/std": 0.42576863765716555, "step": 1295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 507.0, "completions/mean_length": 388.784375, "completions/min_length": 261.2, "epoch": 0.3616133518776078, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6620562076568604, "kl": 0.045209857448935506, "learning_rate": 7.572670976708136e-07, "loss": 0.0018085699528455734, "reward": 3.1241397857666016, "reward_std": 0.36960124373435976, "rewards/IngredientFormatReward/mean": 0.9946874976158142, "rewards/IngredientFormatReward/std": 0.04299455732107162, "rewards/IngredientMatchReward/mean": 0.6504470586776734, "rewards/IngredientMatchReward/std": 0.2744769275188446, "rewards/IngredientQuantityMatchReward/mean": 0.6727553248405457, "rewards/IngredientQuantityMatchReward/std": 0.38103885650634767, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.3866784691810608, "step": 1300 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 504.6, "completions/mean_length": 393.6125, "completions/min_length": 282.0, "epoch": 0.36300417246175243, "frac_reward_zero_std": 0.075, "grad_norm": 0.706493079662323, "kl": 0.04794170141685754, "learning_rate": 7.55292332579599e-07, "loss": 0.0019177064299583436, "reward": 3.060842227935791, "reward_std": 0.37944849729537966, "rewards/IngredientFormatReward/mean": 0.9922395706176758, "rewards/IngredientFormatReward/std": 0.05438530929386616, "rewards/IngredientMatchReward/mean": 0.6182112097740173, "rewards/IngredientMatchReward/std": 0.27092549204826355, "rewards/IngredientQuantityMatchReward/mean": 0.6347665190696716, "rewards/IngredientQuantityMatchReward/std": 0.41198580265045165, "rewards/TotalKcalExactMatchReward/mean": 0.815625, "rewards/TotalKcalExactMatchReward/std": 0.37279285192489625, "step": 1305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 507.8, "completions/mean_length": 392.8203125, "completions/min_length": 267.8, "epoch": 0.36439499304589706, "frac_reward_zero_std": 0.025, "grad_norm": 0.6756166815757751, "kl": 0.0481220202986151, "learning_rate": 7.533121662251458e-07, "loss": 0.0019247991964221, "reward": 3.113231086730957, "reward_std": 0.36561869978904726, "rewards/IngredientFormatReward/mean": 0.9901692628860473, "rewards/IngredientFormatReward/std": 0.05964524168521166, "rewards/IngredientMatchReward/mean": 0.6073189616203308, "rewards/IngredientMatchReward/std": 0.30605280995368955, "rewards/IngredientQuantityMatchReward/mean": 0.7329302668571472, "rewards/IngredientQuantityMatchReward/std": 0.3724480032920837, "rewards/TotalKcalExactMatchReward/mean": 0.7828125, "rewards/TotalKcalExactMatchReward/std": 0.4047005772590637, "step": 1310 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 503.2, "completions/mean_length": 390.128125, "completions/min_length": 258.2, "epoch": 0.3657858136300417, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6606735587120056, "kl": 0.05145847264211625, "learning_rate": 7.513266405021703e-07, "loss": 0.0020576369017362593, "reward": 2.954634428024292, "reward_std": 0.4394294261932373, "rewards/IngredientFormatReward/mean": 0.9875, "rewards/IngredientFormatReward/std": 0.09063890352845191, "rewards/IngredientMatchReward/mean": 0.6141071438789367, "rewards/IngredientMatchReward/std": 0.2969283401966095, "rewards/IngredientQuantityMatchReward/mean": 0.6108398795127868, "rewards/IngredientQuantityMatchReward/std": 0.4317851722240448, "rewards/TotalKcalExactMatchReward/mean": 0.7421875, "rewards/TotalKcalExactMatchReward/std": 0.41644211411476134, "step": 1315 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 509.0, "completions/mean_length": 393.3578125, "completions/min_length": 256.4, "epoch": 0.3671766342141864, "frac_reward_zero_std": 0.05, "grad_norm": 0.6441048383712769, "kl": 0.04711353066377342, "learning_rate": 7.49335797418777e-07, "loss": 0.0018845852464437485, "reward": 3.0771498680114746, "reward_std": 0.4205670475959778, "rewards/IngredientFormatReward/mean": 0.9825632333755493, "rewards/IngredientFormatReward/std": 0.09538376778364181, "rewards/IngredientMatchReward/mean": 0.6370135188102722, "rewards/IngredientMatchReward/std": 0.2930262327194214, "rewards/IngredientQuantityMatchReward/mean": 0.68101065158844, "rewards/IngredientQuantityMatchReward/std": 0.39768595099449155, "rewards/TotalKcalExactMatchReward/mean": 0.7765625, "rewards/TotalKcalExactMatchReward/std": 0.4095042526721954, "step": 1320 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 513.0, "completions/mean_length": 399.4453125, "completions/min_length": 271.6, "epoch": 0.36856745479833103, "frac_reward_zero_std": 0.075, "grad_norm": 0.6368516683578491, "kl": 0.05054594988469034, "learning_rate": 7.473396790955714e-07, "loss": 0.0020216893404722213, "reward": 3.068678379058838, "reward_std": 0.4033094227313995, "rewards/IngredientFormatReward/mean": 0.9831770777702331, "rewards/IngredientFormatReward/std": 0.12154540866613388, "rewards/IngredientMatchReward/mean": 0.6302232027053833, "rewards/IngredientMatchReward/std": 0.2985549569129944, "rewards/IngredientQuantityMatchReward/mean": 0.6443406462669372, "rewards/IngredientQuantityMatchReward/std": 0.4117864727973938, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.3913091003894806, "step": 1325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 509.6, "completions/mean_length": 392.634375, "completions/min_length": 273.8, "epoch": 0.36995827538247567, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7256594300270081, "kl": 0.04610037456732243, "learning_rate": 7.453383277647678e-07, "loss": 0.001843712106347084, "reward": 3.0510849952697754, "reward_std": 0.42244759798049925, "rewards/IngredientFormatReward/mean": 0.9765885472297668, "rewards/IngredientFormatReward/std": 0.1349714256823063, "rewards/IngredientMatchReward/mean": 0.6786749958992004, "rewards/IngredientMatchReward/std": 0.29368536472320556, "rewards/IngredientQuantityMatchReward/mean": 0.6880090117454529, "rewards/IngredientQuantityMatchReward/std": 0.3833075225353241, "rewards/TotalKcalExactMatchReward/mean": 0.7078125, "rewards/TotalKcalExactMatchReward/std": 0.45246235728263856, "step": 1330 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 513.0, "completions/mean_length": 390.9390625, "completions/min_length": 239.6, "epoch": 0.3713490959666203, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7128857970237732, "kl": 0.047564028692431745, "learning_rate": 7.433317857692962e-07, "loss": 0.0019028615206480026, "reward": 3.018064594268799, "reward_std": 0.41078593134880065, "rewards/IngredientFormatReward/mean": 0.9841666698455811, "rewards/IngredientFormatReward/std": 0.1172505185008049, "rewards/IngredientMatchReward/mean": 0.595754599571228, "rewards/IngredientMatchReward/std": 0.28117249011993406, "rewards/IngredientQuantityMatchReward/mean": 0.6256433725357056, "rewards/IngredientQuantityMatchReward/std": 0.41258437037467954, "rewards/TotalKcalExactMatchReward/mean": 0.8125, "rewards/TotalKcalExactMatchReward/std": 0.3811466634273529, "step": 1335 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 504.0, "completions/mean_length": 393.603125, "completions/min_length": 281.0, "epoch": 0.37273991655076494, "frac_reward_zero_std": 0.075, "grad_norm": 0.633821427822113, "kl": 0.04494093197863549, "learning_rate": 7.413200955619065e-07, "loss": 0.0017976880073547364, "reward": 3.161479616165161, "reward_std": 0.3582748889923096, "rewards/IngredientFormatReward/mean": 0.9943489551544189, "rewards/IngredientFormatReward/std": 0.05378681570291519, "rewards/IngredientMatchReward/mean": 0.6633042454719543, "rewards/IngredientMatchReward/std": 0.2839880406856537, "rewards/IngredientQuantityMatchReward/mean": 0.6960138440132141, "rewards/IngredientQuantityMatchReward/std": 0.401453024148941, "rewards/TotalKcalExactMatchReward/mean": 0.8078125, "rewards/TotalKcalExactMatchReward/std": 0.3868353426456451, "step": 1340 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 513.0, "completions/mean_length": 395.909375, "completions/min_length": 269.6, "epoch": 0.3741307371349096, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6353825330734253, "kl": 0.047297230316326024, "learning_rate": 7.393032997042697e-07, "loss": 0.0018922967836260795, "reward": 2.9042635917663575, "reward_std": 0.45287638902664185, "rewards/IngredientFormatReward/mean": 0.9635937571525574, "rewards/IngredientFormatReward/std": 0.1764824703335762, "rewards/IngredientMatchReward/mean": 0.6264143109321594, "rewards/IngredientMatchReward/std": 0.2897026062011719, "rewards/IngredientQuantityMatchReward/mean": 0.5861305117607116, "rewards/IngredientQuantityMatchReward/std": 0.4275715291500092, "rewards/TotalKcalExactMatchReward/mean": 0.728125, "rewards/TotalKcalExactMatchReward/std": 0.44556268453598025, "step": 1345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 503.4, "completions/mean_length": 391.7609375, "completions/min_length": 268.2, "epoch": 0.37552155771905427, "frac_reward_zero_std": 0.05, "grad_norm": 0.6813985109329224, "kl": 0.04787277162540704, "learning_rate": 7.372814408660788e-07, "loss": 0.0019150124862790108, "reward": 3.0159966945648193, "reward_std": 0.4248832643032074, "rewards/IngredientFormatReward/mean": 0.9875520706176758, "rewards/IngredientFormatReward/std": 0.0813837306573987, "rewards/IngredientMatchReward/mean": 0.629173481464386, "rewards/IngredientMatchReward/std": 0.3017181038856506, "rewards/IngredientQuantityMatchReward/mean": 0.6195836544036866, "rewards/IngredientQuantityMatchReward/std": 0.403140127658844, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.41130141019821165, "step": 1350 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 513.0, "completions/mean_length": 392.6796875, "completions/min_length": 268.6, "epoch": 0.3769123783031989, "frac_reward_zero_std": 0.0, "grad_norm": 0.7041727304458618, "kl": 0.04741974025964737, "learning_rate": 7.352545618241442e-07, "loss": 0.0018971532583236695, "reward": 2.937150573730469, "reward_std": 0.47615134716033936, "rewards/IngredientFormatReward/mean": 0.9778645873069763, "rewards/IngredientFormatReward/std": 0.1367991030216217, "rewards/IngredientMatchReward/mean": 0.6257552027702331, "rewards/IngredientMatchReward/std": 0.2937148153781891, "rewards/IngredientQuantityMatchReward/mean": 0.5600932717323304, "rewards/IngredientQuantityMatchReward/std": 0.4284715175628662, "rewards/TotalKcalExactMatchReward/mean": 0.7734375, "rewards/TotalKcalExactMatchReward/std": 0.3832641303539276, "step": 1355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 511.2, "completions/mean_length": 399.834375, "completions/min_length": 275.0, "epoch": 0.37830319888734354, "frac_reward_zero_std": 0.05, "grad_norm": 0.6130369901657104, "kl": 0.04342401102185249, "learning_rate": 7.332227054614904e-07, "loss": 0.0017366718500852584, "reward": 3.0526763916015627, "reward_std": 0.42766509056091306, "rewards/IngredientFormatReward/mean": 0.9768749952316285, "rewards/IngredientFormatReward/std": 0.12218592166900635, "rewards/IngredientMatchReward/mean": 0.6191027879714965, "rewards/IngredientMatchReward/std": 0.3084140956401825, "rewards/IngredientQuantityMatchReward/mean": 0.6410735964775085, "rewards/IngredientQuantityMatchReward/std": 0.41126957535743713, "rewards/TotalKcalExactMatchReward/mean": 0.815625, "rewards/TotalKcalExactMatchReward/std": 0.3846336603164673, "step": 1360 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 506.0, "completions/mean_length": 391.740625, "completions/min_length": 240.6, "epoch": 0.3796940194714882, "frac_reward_zero_std": 0.05, "grad_norm": 0.7070378065109253, "kl": 0.04717907048761845, "learning_rate": 7.311859147664472e-07, "loss": 0.00188693106174469, "reward": 3.010082483291626, "reward_std": 0.44024962186813354, "rewards/IngredientFormatReward/mean": 0.9850000023841858, "rewards/IngredientFormatReward/std": 0.09434654638171196, "rewards/IngredientMatchReward/mean": 0.6260190844535828, "rewards/IngredientMatchReward/std": 0.2963875949382782, "rewards/IngredientQuantityMatchReward/mean": 0.6240634083747864, "rewards/IngredientQuantityMatchReward/std": 0.41469027996063235, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.41165693998336794, "step": 1365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 513.0, "completions/mean_length": 394.61875, "completions/min_length": 267.8, "epoch": 0.3810848400556328, "frac_reward_zero_std": 0.025, "grad_norm": 0.6732054948806763, "kl": 0.5468679509591311, "learning_rate": 7.291442328317413e-07, "loss": 0.021924957633018494, "reward": 2.8937462329864503, "reward_std": 0.4447342872619629, "rewards/IngredientFormatReward/mean": 0.9809635400772094, "rewards/IngredientFormatReward/std": 0.12264752835035324, "rewards/IngredientMatchReward/mean": 0.6312493920326233, "rewards/IngredientMatchReward/std": 0.2771876215934753, "rewards/IngredientQuantityMatchReward/mean": 0.5799708127975464, "rewards/IngredientQuantityMatchReward/std": 0.43602352142333983, "rewards/TotalKcalExactMatchReward/mean": 0.7015625, "rewards/TotalKcalExactMatchReward/std": 0.4481926143169403, "step": 1370 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 513.0, "completions/mean_length": 401.4375, "completions/min_length": 286.0, "epoch": 0.38247566063977745, "frac_reward_zero_std": 0.05, "grad_norm": 0.6947128772735596, "kl": 0.051231970288790764, "learning_rate": 7.270977028535846e-07, "loss": 0.0020487431436777117, "reward": 3.081242084503174, "reward_std": 0.42529494166374204, "rewards/IngredientFormatReward/mean": 0.9795535802841187, "rewards/IngredientFormatReward/std": 0.13361096233129502, "rewards/IngredientMatchReward/mean": 0.6574689984321594, "rewards/IngredientMatchReward/std": 0.27973159551620486, "rewards/IngredientQuantityMatchReward/mean": 0.6489070296287537, "rewards/IngredientQuantityMatchReward/std": 0.41112475991249087, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.3904457151889801, "step": 1375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 512.2, "completions/mean_length": 388.3703125, "completions/min_length": 249.2, "epoch": 0.38386648122392214, "frac_reward_zero_std": 0.075, "grad_norm": 0.6509252190589905, "kl": 0.046685748361051084, "learning_rate": 7.250463681307588e-07, "loss": 0.001867828331887722, "reward": 3.0991830825805664, "reward_std": 0.4528281927108765, "rewards/IngredientFormatReward/mean": 0.980078125, "rewards/IngredientFormatReward/std": 0.1193616308271885, "rewards/IngredientMatchReward/mean": 0.6413913607597351, "rewards/IngredientMatchReward/std": 0.29768499433994294, "rewards/IngredientQuantityMatchReward/mean": 0.6730260252952576, "rewards/IngredientQuantityMatchReward/std": 0.4107559084892273, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.3941961944103241, "step": 1380 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 501.4, "completions/mean_length": 390.2609375, "completions/min_length": 258.2, "epoch": 0.3852573018080668, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6028541922569275, "kl": 0.0516342549584806, "learning_rate": 7.229902720637013e-07, "loss": 0.0020651400089263918, "reward": 3.0714120864868164, "reward_std": 0.4244008719921112, "rewards/IngredientFormatReward/mean": 0.9860342264175415, "rewards/IngredientFormatReward/std": 0.08796465378254652, "rewards/IngredientMatchReward/mean": 0.6268959999084472, "rewards/IngredientMatchReward/std": 0.30808745622634887, "rewards/IngredientQuantityMatchReward/mean": 0.6631693601608276, "rewards/IngredientQuantityMatchReward/std": 0.4120981454849243, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.3891636371612549, "step": 1385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.028125, "completions/max_length": 512.8, "completions/mean_length": 394.9046875, "completions/min_length": 259.6, "epoch": 0.3866481223922114, "frac_reward_zero_std": 0.0125, "grad_norm": 0.693778395652771, "kl": 0.04984358986839652, "learning_rate": 7.20929458153586e-07, "loss": 0.0019936652854084967, "reward": 2.9926676750183105, "reward_std": 0.4538565337657928, "rewards/IngredientFormatReward/mean": 0.9710416793823242, "rewards/IngredientFormatReward/std": 0.1352289214730263, "rewards/IngredientMatchReward/mean": 0.5942381978034973, "rewards/IngredientMatchReward/std": 0.296366685628891, "rewards/IngredientQuantityMatchReward/mean": 0.5883253335952758, "rewards/IngredientQuantityMatchReward/std": 0.4092931866645813, "rewards/TotalKcalExactMatchReward/mean": 0.8390625, "rewards/TotalKcalExactMatchReward/std": 0.36484946608543395, "step": 1390 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 501.6, "completions/mean_length": 397.6265625, "completions/min_length": 278.0, "epoch": 0.38803894297635605, "frac_reward_zero_std": 0.025, "grad_norm": 0.6608783006668091, "kl": 0.04760516933165491, "learning_rate": 7.188639700014025e-07, "loss": 0.0019130157306790352, "reward": 3.1505064487457277, "reward_std": 0.3983344674110413, "rewards/IngredientFormatReward/mean": 0.9819270849227906, "rewards/IngredientFormatReward/std": 0.1014548234641552, "rewards/IngredientMatchReward/mean": 0.6463622331619263, "rewards/IngredientMatchReward/std": 0.29087430238723755, "rewards/IngredientQuantityMatchReward/mean": 0.6690921902656555, "rewards/IngredientQuantityMatchReward/std": 0.4150608241558075, "rewards/TotalKcalExactMatchReward/mean": 0.853125, "rewards/TotalKcalExactMatchReward/std": 0.3509778559207916, "step": 1395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 509.4, "completions/mean_length": 395.771875, "completions/min_length": 267.4, "epoch": 0.3894297635605007, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7344890236854553, "kl": 0.048899111011996864, "learning_rate": 7.167938513070344e-07, "loss": 0.001956099271774292, "reward": 2.9775744438171388, "reward_std": 0.34394469261169436, "rewards/IngredientFormatReward/mean": 0.989661455154419, "rewards/IngredientFormatReward/std": 0.06796687468886375, "rewards/IngredientMatchReward/mean": 0.6366368889808655, "rewards/IngredientMatchReward/std": 0.28161700963974, "rewards/IngredientQuantityMatchReward/mean": 0.571588671207428, "rewards/IngredientQuantityMatchReward/std": 0.4268614649772644, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.4140839040279388, "step": 1400 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 503.6, "completions/mean_length": 391.0953125, "completions/min_length": 270.0, "epoch": 0.3908205841446453, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6491956114768982, "kl": 0.047348243673332034, "learning_rate": 7.147191458683348e-07, "loss": 0.0018939247354865074, "reward": 3.1373552322387694, "reward_std": 0.39182621240615845, "rewards/IngredientFormatReward/mean": 0.9948660731315613, "rewards/IngredientFormatReward/std": 0.046136388555169106, "rewards/IngredientMatchReward/mean": 0.6508209228515625, "rewards/IngredientMatchReward/std": 0.2824489861726761, "rewards/IngredientQuantityMatchReward/mean": 0.672918152809143, "rewards/IngredientQuantityMatchReward/std": 0.42214601039886473, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.38358521461486816, "step": 1405 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 511.8, "completions/mean_length": 393.0140625, "completions/min_length": 285.4, "epoch": 0.39221140472878996, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6753212809562683, "kl": 0.047298068576492366, "learning_rate": 7.126398975801988e-07, "loss": 0.001892334222793579, "reward": 2.993410110473633, "reward_std": 0.42139936685562135, "rewards/IngredientFormatReward/mean": 0.991614580154419, "rewards/IngredientFormatReward/std": 0.07206480577588081, "rewards/IngredientMatchReward/mean": 0.6076562523841857, "rewards/IngredientMatchReward/std": 0.28833829760551455, "rewards/IngredientQuantityMatchReward/mean": 0.6097642540931701, "rewards/IngredientQuantityMatchReward/std": 0.4363261044025421, "rewards/TotalKcalExactMatchReward/mean": 0.784375, "rewards/TotalKcalExactMatchReward/std": 0.3828993201255798, "step": 1410 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 510.2, "completions/mean_length": 391.2296875, "completions/min_length": 264.0, "epoch": 0.39360222531293465, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6234867572784424, "kl": 0.04924510028213262, "learning_rate": 7.105561504336355e-07, "loss": 0.001970212534070015, "reward": 3.0320447444915772, "reward_std": 0.4181003987789154, "rewards/IngredientFormatReward/mean": 0.9856250047683716, "rewards/IngredientFormatReward/std": 0.08855490759015083, "rewards/IngredientMatchReward/mean": 0.6343092918395996, "rewards/IngredientMatchReward/std": 0.30618330240249636, "rewards/IngredientQuantityMatchReward/mean": 0.641797935962677, "rewards/IngredientQuantityMatchReward/std": 0.4155651867389679, "rewards/TotalKcalExactMatchReward/mean": 0.7703125, "rewards/TotalKcalExactMatchReward/std": 0.4193932831287384, "step": 1415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 508.4, "completions/mean_length": 395.6515625, "completions/min_length": 264.2, "epoch": 0.3949930458970793, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6549289226531982, "kl": 0.048028711089864375, "learning_rate": 7.084679485148375e-07, "loss": 0.0019211186096072196, "reward": 3.0376140594482424, "reward_std": 0.3858941853046417, "rewards/IngredientFormatReward/mean": 0.9880208253860474, "rewards/IngredientFormatReward/std": 0.08144994545727968, "rewards/IngredientMatchReward/mean": 0.5925304055213928, "rewards/IngredientMatchReward/std": 0.3050950050354004, "rewards/IngredientQuantityMatchReward/mean": 0.6211253523826599, "rewards/IngredientQuantityMatchReward/std": 0.4290615260601044, "rewards/TotalKcalExactMatchReward/mean": 0.8359375, "rewards/TotalKcalExactMatchReward/std": 0.35664623975753784, "step": 1420 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 511.4, "completions/mean_length": 392.915625, "completions/min_length": 250.6, "epoch": 0.3963838664812239, "frac_reward_zero_std": 0.0625, "grad_norm": 0.5656110048294067, "kl": 0.061237924033775926, "learning_rate": 7.06375336004247e-07, "loss": 0.0024501249194145204, "reward": 3.006493091583252, "reward_std": 0.47243286967277526, "rewards/IngredientFormatReward/mean": 0.9774218797683716, "rewards/IngredientFormatReward/std": 0.12839448899030687, "rewards/IngredientMatchReward/mean": 0.6348812222480774, "rewards/IngredientMatchReward/std": 0.301179563999176, "rewards/IngredientQuantityMatchReward/mean": 0.594189977645874, "rewards/IngredientQuantityMatchReward/std": 0.42939573526382446, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.3994594097137451, "step": 1425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 507.8, "completions/mean_length": 391.134375, "completions/min_length": 266.0, "epoch": 0.39777468706536856, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6568824648857117, "kl": 0.05042944210581481, "learning_rate": 7.042783571756228e-07, "loss": 0.0020168770104646684, "reward": 3.0168625831604006, "reward_std": 0.46565952301025393, "rewards/IngredientFormatReward/mean": 0.9702976226806641, "rewards/IngredientFormatReward/std": 0.1451386496424675, "rewards/IngredientMatchReward/mean": 0.6251142144203186, "rewards/IngredientMatchReward/std": 0.2843863904476166, "rewards/IngredientQuantityMatchReward/mean": 0.6386382818222046, "rewards/IngredientQuantityMatchReward/std": 0.4032295286655426, "rewards/TotalKcalExactMatchReward/mean": 0.7828125, "rewards/TotalKcalExactMatchReward/std": 0.40105786323547366, "step": 1430 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 510.2, "completions/mean_length": 391.8359375, "completions/min_length": 276.4, "epoch": 0.3991655076495132, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6645727753639221, "kl": 0.05089432932436466, "learning_rate": 7.021770563951017e-07, "loss": 0.002035984769463539, "reward": 3.065924549102783, "reward_std": 0.42762036323547364, "rewards/IngredientFormatReward/mean": 0.990416657924652, "rewards/IngredientFormatReward/std": 0.07949181646108627, "rewards/IngredientMatchReward/mean": 0.6396770238876343, "rewards/IngredientMatchReward/std": 0.2999457597732544, "rewards/IngredientQuantityMatchReward/mean": 0.6295809030532837, "rewards/IngredientQuantityMatchReward/std": 0.4113676190376282, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.3949193716049194, "step": 1435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 510.4, "completions/mean_length": 393.7078125, "completions/min_length": 288.2, "epoch": 0.40055632823365783, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6648370623588562, "kl": 0.051764291478320956, "learning_rate": 7.000714781202613e-07, "loss": 0.0020709920674562452, "reward": 3.0830320358276366, "reward_std": 0.49956942200660703, "rewards/IngredientFormatReward/mean": 0.979479169845581, "rewards/IngredientFormatReward/std": 0.11220841258764266, "rewards/IngredientMatchReward/mean": 0.6905896306037903, "rewards/IngredientMatchReward/std": 0.27779283225536344, "rewards/IngredientQuantityMatchReward/mean": 0.6395256638526916, "rewards/IngredientQuantityMatchReward/std": 0.4275822460651398, "rewards/TotalKcalExactMatchReward/mean": 0.7734375, "rewards/TotalKcalExactMatchReward/std": 0.4142127275466919, "step": 1440 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 501.4, "completions/mean_length": 382.08125, "completions/min_length": 248.0, "epoch": 0.4019471488178025, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7125925421714783, "kl": 0.05212058457545936, "learning_rate": 6.97961666899179e-07, "loss": 0.002085070312023163, "reward": 2.974539279937744, "reward_std": 0.4162575304508209, "rewards/IngredientFormatReward/mean": 0.9879166603088378, "rewards/IngredientFormatReward/std": 0.08713518846780062, "rewards/IngredientMatchReward/mean": 0.6231901168823242, "rewards/IngredientMatchReward/std": 0.2897790253162384, "rewards/IngredientQuantityMatchReward/mean": 0.5634325981140137, "rewards/IngredientQuantityMatchReward/std": 0.4310961186885834, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.3805993765592575, "step": 1445 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 509.8, "completions/mean_length": 387.584375, "completions/min_length": 254.2, "epoch": 0.40333796940194716, "frac_reward_zero_std": 0.1125, "grad_norm": 0.7150693535804749, "kl": 0.0496115677524358, "learning_rate": 6.958476673694888e-07, "loss": 0.0019844066351652145, "reward": 3.0862731456756594, "reward_std": 0.4122806191444397, "rewards/IngredientFormatReward/mean": 0.9861458301544189, "rewards/IngredientFormatReward/std": 0.10063027888536454, "rewards/IngredientMatchReward/mean": 0.6254792809486389, "rewards/IngredientMatchReward/std": 0.27667003870010376, "rewards/IngredientQuantityMatchReward/mean": 0.6402730107307434, "rewards/IngredientQuantityMatchReward/std": 0.4204639375209808, "rewards/TotalKcalExactMatchReward/mean": 0.834375, "rewards/TotalKcalExactMatchReward/std": 0.363431441783905, "step": 1450 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 512.8, "completions/mean_length": 393.0984375, "completions/min_length": 285.8, "epoch": 0.4047287899860918, "frac_reward_zero_std": 0.025, "grad_norm": 0.6466384530067444, "kl": 0.050380307948216796, "learning_rate": 6.93729524257438e-07, "loss": 0.002015003561973572, "reward": 3.1226918697357178, "reward_std": 0.43267813324928284, "rewards/IngredientFormatReward/mean": 0.9815885305404664, "rewards/IngredientFormatReward/std": 0.11584979891777039, "rewards/IngredientMatchReward/mean": 0.6187383890151977, "rewards/IngredientMatchReward/std": 0.2922286242246628, "rewards/IngredientQuantityMatchReward/mean": 0.6879899024963378, "rewards/IngredientQuantityMatchReward/std": 0.37570016384124755, "rewards/TotalKcalExactMatchReward/mean": 0.834375, "rewards/TotalKcalExactMatchReward/std": 0.35577985644340515, "step": 1455 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 513.0, "completions/mean_length": 395.3609375, "completions/min_length": 266.8, "epoch": 0.40611961057023643, "frac_reward_zero_std": 0.025, "grad_norm": 0.6766321659088135, "kl": 0.08121923571452498, "learning_rate": 6.916072823769398e-07, "loss": 0.003253092244267464, "reward": 2.9472359657287597, "reward_std": 0.44583975076675414, "rewards/IngredientFormatReward/mean": 0.9813020825386047, "rewards/IngredientFormatReward/std": 0.11078893411904574, "rewards/IngredientMatchReward/mean": 0.6114800453186036, "rewards/IngredientMatchReward/std": 0.2834586203098297, "rewards/IngredientQuantityMatchReward/mean": 0.5935163855552673, "rewards/IngredientQuantityMatchReward/std": 0.42633000016212463, "rewards/TotalKcalExactMatchReward/mean": 0.7609375, "rewards/TotalKcalExactMatchReward/std": 0.4214301824569702, "step": 1460 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 513.0, "completions/mean_length": 393.3234375, "completions/min_length": 279.2, "epoch": 0.40751043115438107, "frac_reward_zero_std": 0.025, "grad_norm": 0.6546911597251892, "kl": 0.04946425724774599, "learning_rate": 6.894809866286269e-07, "loss": 0.0019789854064583778, "reward": 3.0060822010040282, "reward_std": 0.42944018840789794, "rewards/IngredientFormatReward/mean": 0.9875, "rewards/IngredientFormatReward/std": 0.1100594773888588, "rewards/IngredientMatchReward/mean": 0.6255816102027894, "rewards/IngredientMatchReward/std": 0.2829381048679352, "rewards/IngredientQuantityMatchReward/mean": 0.646125602722168, "rewards/IngredientQuantityMatchReward/std": 0.40343931317329407, "rewards/TotalKcalExactMatchReward/mean": 0.746875, "rewards/TotalKcalExactMatchReward/std": 0.4242517352104187, "step": 1465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 508.0, "completions/mean_length": 391.7578125, "completions/min_length": 267.0, "epoch": 0.4089012517385257, "frac_reward_zero_std": 0.05, "grad_norm": 0.644096314907074, "kl": 0.04821961410343647, "learning_rate": 6.873506819988985e-07, "loss": 0.0019289355725049973, "reward": 3.0299394607543944, "reward_std": 0.38674944043159487, "rewards/IngredientFormatReward/mean": 0.9899999976158143, "rewards/IngredientFormatReward/std": 0.0847195789217949, "rewards/IngredientMatchReward/mean": 0.6211880207061767, "rewards/IngredientMatchReward/std": 0.27895829677581785, "rewards/IngredientQuantityMatchReward/mean": 0.6437513709068299, "rewards/IngredientQuantityMatchReward/std": 0.42142691612243655, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.41285876035690305, "step": 1470 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 508.0, "completions/mean_length": 387.296875, "completions/min_length": 282.8, "epoch": 0.4102920723226704, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6414361000061035, "kl": 0.05062154838815332, "learning_rate": 6.852164135589724e-07, "loss": 0.0020248919725418093, "reward": 3.091634464263916, "reward_std": 0.39072946906089784, "rewards/IngredientFormatReward/mean": 0.9922135353088379, "rewards/IngredientFormatReward/std": 0.062381112948060036, "rewards/IngredientMatchReward/mean": 0.6525762677192688, "rewards/IngredientMatchReward/std": 0.2883527398109436, "rewards/IngredientQuantityMatchReward/mean": 0.6530946910381317, "rewards/IngredientQuantityMatchReward/std": 0.39076480865478513, "rewards/TotalKcalExactMatchReward/mean": 0.79375, "rewards/TotalKcalExactMatchReward/std": 0.3955763101577759, "step": 1475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 513.0, "completions/mean_length": 392.8875, "completions/min_length": 266.4, "epoch": 0.41168289290681503, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6684796810150146, "kl": 0.052700078953057526, "learning_rate": 6.83078226463928e-07, "loss": 0.0021081626415252685, "reward": 3.0118268489837647, "reward_std": 0.41152234077453614, "rewards/IngredientFormatReward/mean": 0.9853869080543518, "rewards/IngredientFormatReward/std": 0.09761874377727509, "rewards/IngredientMatchReward/mean": 0.5979613065719604, "rewards/IngredientMatchReward/std": 0.2978090465068817, "rewards/IngredientQuantityMatchReward/mean": 0.6112910985946656, "rewards/IngredientQuantityMatchReward/std": 0.4339863657951355, "rewards/TotalKcalExactMatchReward/mean": 0.8171875, "rewards/TotalKcalExactMatchReward/std": 0.37414880394935607, "step": 1480 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 511.8, "completions/mean_length": 386.5359375, "completions/min_length": 260.2, "epoch": 0.41307371349095967, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6525306105613708, "kl": 0.0471834511961788, "learning_rate": 6.809361659517527e-07, "loss": 0.0018876727670431137, "reward": 3.0948177337646485, "reward_std": 0.369168359041214, "rewards/IngredientFormatReward/mean": 0.990234375, "rewards/IngredientFormatReward/std": 0.08957751505076886, "rewards/IngredientMatchReward/mean": 0.6415643811225891, "rewards/IngredientMatchReward/std": 0.2889476329088211, "rewards/IngredientQuantityMatchReward/mean": 0.6364565491676331, "rewards/IngredientQuantityMatchReward/std": 0.4115687429904938, "rewards/TotalKcalExactMatchReward/mean": 0.8265625, "rewards/TotalKcalExactMatchReward/std": 0.37568016052246095, "step": 1485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 508.2, "completions/mean_length": 393.3828125, "completions/min_length": 255.2, "epoch": 0.4144645340751043, "frac_reward_zero_std": 0.0625, "grad_norm": 0.637024998664856, "kl": 0.04654967598617077, "learning_rate": 6.78790277342385e-07, "loss": 0.0018621478229761124, "reward": 3.1724345684051514, "reward_std": 0.40526561737060546, "rewards/IngredientFormatReward/mean": 0.9826022982597351, "rewards/IngredientFormatReward/std": 0.10656376946717501, "rewards/IngredientMatchReward/mean": 0.6608142852783203, "rewards/IngredientMatchReward/std": 0.28252187967300413, "rewards/IngredientQuantityMatchReward/mean": 0.7321430206298828, "rewards/IngredientQuantityMatchReward/std": 0.3806688904762268, "rewards/TotalKcalExactMatchReward/mean": 0.796875, "rewards/TotalKcalExactMatchReward/std": 0.3964962542057037, "step": 1490 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 510.6, "completions/mean_length": 393.2796875, "completions/min_length": 271.4, "epoch": 0.41585535465924894, "frac_reward_zero_std": 0.05, "grad_norm": 0.635031521320343, "kl": 0.048082446400076154, "learning_rate": 6.766406060367543e-07, "loss": 0.001923457160592079, "reward": 3.1764147758483885, "reward_std": 0.4066992163658142, "rewards/IngredientFormatReward/mean": 0.9915624976158142, "rewards/IngredientFormatReward/std": 0.06838876903057098, "rewards/IngredientMatchReward/mean": 0.6770944952964782, "rewards/IngredientMatchReward/std": 0.29843786358833313, "rewards/IngredientQuantityMatchReward/mean": 0.6796328067779541, "rewards/IngredientQuantityMatchReward/std": 0.3973805606365204, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.36534249782562256, "step": 1495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 502.8, "completions/mean_length": 387.5390625, "completions/min_length": 267.0, "epoch": 0.4172461752433936, "frac_reward_zero_std": 0.0375, "grad_norm": 0.706847071647644, "kl": 0.04723797417245805, "learning_rate": 6.744871975158215e-07, "loss": 0.001889602467417717, "reward": 3.0415586471557616, "reward_std": 0.4428049921989441, "rewards/IngredientFormatReward/mean": 0.9885416746139526, "rewards/IngredientFormatReward/std": 0.0781378224492073, "rewards/IngredientMatchReward/mean": 0.6192667484283447, "rewards/IngredientMatchReward/std": 0.27974243760108947, "rewards/IngredientQuantityMatchReward/mean": 0.6540627002716064, "rewards/IngredientQuantityMatchReward/std": 0.42795388102531434, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.412675142288208, "step": 1500 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 502.4, "completions/mean_length": 386.4796875, "completions/min_length": 258.0, "epoch": 0.41863699582753827, "frac_reward_zero_std": 0.0, "grad_norm": 0.6618068814277649, "kl": 0.05870349365286529, "learning_rate": 6.723300973396166e-07, "loss": 0.0023463867604732514, "reward": 3.161744213104248, "reward_std": 0.42260637879371643, "rewards/IngredientFormatReward/mean": 0.9826562404632568, "rewards/IngredientFormatReward/std": 0.09131253361701966, "rewards/IngredientMatchReward/mean": 0.6721211671829224, "rewards/IngredientMatchReward/std": 0.3050087809562683, "rewards/IngredientQuantityMatchReward/mean": 0.6444667696952819, "rewards/IngredientQuantityMatchReward/std": 0.39916192293167113, "rewards/TotalKcalExactMatchReward/mean": 0.8625, "rewards/TotalKcalExactMatchReward/std": 0.34389914870262145, "step": 1505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 501.0, "completions/mean_length": 391.0140625, "completions/min_length": 275.6, "epoch": 0.4200278164116829, "frac_reward_zero_std": 0.025, "grad_norm": 0.6646866202354431, "kl": 0.04706322541460395, "learning_rate": 6.701693511462743e-07, "loss": 0.0018825791776180267, "reward": 3.094019556045532, "reward_std": 0.3848333120346069, "rewards/IngredientFormatReward/mean": 0.9956249952316284, "rewards/IngredientFormatReward/std": 0.035187306255102156, "rewards/IngredientMatchReward/mean": 0.6332756876945496, "rewards/IngredientMatchReward/std": 0.28339311182498933, "rewards/IngredientQuantityMatchReward/mean": 0.6979313969612122, "rewards/IngredientQuantityMatchReward/std": 0.39676953554153443, "rewards/TotalKcalExactMatchReward/mean": 0.7671875, "rewards/TotalKcalExactMatchReward/std": 0.4147461235523224, "step": 1510 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 509.0, "completions/mean_length": 389.246875, "completions/min_length": 254.6, "epoch": 0.42141863699582754, "frac_reward_zero_std": 0.025, "grad_norm": 0.6115160584449768, "kl": 0.04909869590774178, "learning_rate": 6.680050046510687e-07, "loss": 0.0019641295075416565, "reward": 3.0600674152374268, "reward_std": 0.41310828924179077, "rewards/IngredientFormatReward/mean": 0.9803515672683716, "rewards/IngredientFormatReward/std": 0.10586707219481468, "rewards/IngredientMatchReward/mean": 0.6523542881011963, "rewards/IngredientMatchReward/std": 0.2938463926315308, "rewards/IngredientQuantityMatchReward/mean": 0.6664240956306458, "rewards/IngredientQuantityMatchReward/std": 0.40689346194267273, "rewards/TotalKcalExactMatchReward/mean": 0.7609375, "rewards/TotalKcalExactMatchReward/std": 0.3891270637512207, "step": 1515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 506.0, "completions/mean_length": 389.6171875, "completions/min_length": 262.6, "epoch": 0.4228094575799722, "frac_reward_zero_std": 0.025, "grad_norm": 0.6689432859420776, "kl": 0.051034286990761756, "learning_rate": 6.658371036454464e-07, "loss": 0.002041558176279068, "reward": 3.0063760757446287, "reward_std": 0.4658251166343689, "rewards/IngredientFormatReward/mean": 0.9912500023841858, "rewards/IngredientFormatReward/std": 0.06588455513119698, "rewards/IngredientMatchReward/mean": 0.5811309456825257, "rewards/IngredientMatchReward/std": 0.2976264595985413, "rewards/IngredientQuantityMatchReward/mean": 0.6136826276779175, "rewards/IngredientQuantityMatchReward/std": 0.4315095841884613, "rewards/TotalKcalExactMatchReward/mean": 0.8203125, "rewards/TotalKcalExactMatchReward/std": 0.3791858911514282, "step": 1520 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 498.4, "completions/mean_length": 391.5, "completions/min_length": 246.0, "epoch": 0.4242002781641168, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6664140224456787, "kl": 0.045389461284503343, "learning_rate": 6.636656939960569e-07, "loss": 0.0018157461658120156, "reward": 3.1146235942840574, "reward_std": 0.41344255208969116, "rewards/IngredientFormatReward/mean": 0.9910788655281066, "rewards/IngredientFormatReward/std": 0.05945278406143188, "rewards/IngredientMatchReward/mean": 0.6309400081634522, "rewards/IngredientMatchReward/std": 0.30454255938529967, "rewards/IngredientQuantityMatchReward/mean": 0.6738548040390014, "rewards/IngredientQuantityMatchReward/std": 0.39767482280731203, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.37096219062805175, "step": 1525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 508.4, "completions/mean_length": 387.33125, "completions/min_length": 264.2, "epoch": 0.42559109874826145, "frac_reward_zero_std": 0.075, "grad_norm": 0.6901034712791443, "kl": 0.049771115835756066, "learning_rate": 6.614908216437831e-07, "loss": 0.0019908515736460687, "reward": 3.086984157562256, "reward_std": 0.46495330333709717, "rewards/IngredientFormatReward/mean": 0.9900111675262451, "rewards/IngredientFormatReward/std": 0.08576798439025879, "rewards/IngredientMatchReward/mean": 0.6765054702758789, "rewards/IngredientMatchReward/std": 0.27772613167762755, "rewards/IngredientQuantityMatchReward/mean": 0.6423426389694213, "rewards/IngredientQuantityMatchReward/std": 0.4157872200012207, "rewards/TotalKcalExactMatchReward/mean": 0.778125, "rewards/TotalKcalExactMatchReward/std": 0.4095184445381165, "step": 1530 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 511.0, "completions/mean_length": 400.621875, "completions/min_length": 261.8, "epoch": 0.42698191933240615, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6421478986740112, "kl": 0.04838813366368413, "learning_rate": 6.593125326027689e-07, "loss": 0.0019358258694410324, "reward": 2.994578552246094, "reward_std": 0.44342033863067626, "rewards/IngredientFormatReward/mean": 0.9785788774490356, "rewards/IngredientFormatReward/std": 0.12433337345719338, "rewards/IngredientMatchReward/mean": 0.5929954409599304, "rewards/IngredientMatchReward/std": 0.30798673033714297, "rewards/IngredientQuantityMatchReward/mean": 0.6261291861534118, "rewards/IngredientQuantityMatchReward/std": 0.43534895181655886, "rewards/TotalKcalExactMatchReward/mean": 0.796875, "rewards/TotalKcalExactMatchReward/std": 0.39741336107254027, "step": 1535 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 512.0, "completions/mean_length": 394.778125, "completions/min_length": 273.6, "epoch": 0.4283727399165508, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6390155553817749, "kl": 0.04921357906423509, "learning_rate": 6.571308729594449e-07, "loss": 0.0019685806706547736, "reward": 3.1461612224578857, "reward_std": 0.39753777980804444, "rewards/IngredientFormatReward/mean": 0.9929166674613953, "rewards/IngredientFormatReward/std": 0.07146886438131332, "rewards/IngredientMatchReward/mean": 0.7030505895614624, "rewards/IngredientMatchReward/std": 0.2847248435020447, "rewards/IngredientQuantityMatchReward/mean": 0.6126938581466674, "rewards/IngredientQuantityMatchReward/std": 0.42437596917152404, "rewards/TotalKcalExactMatchReward/mean": 0.8375, "rewards/TotalKcalExactMatchReward/std": 0.3570310711860657, "step": 1540 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 511.4, "completions/mean_length": 390.5375, "completions/min_length": 265.0, "epoch": 0.4297635605006954, "frac_reward_zero_std": 0.1, "grad_norm": 0.6613978147506714, "kl": 0.05156162865459919, "learning_rate": 6.549458888715555e-07, "loss": 0.0020627494901418685, "reward": 3.0349883079528808, "reward_std": 0.4523595035076141, "rewards/IngredientFormatReward/mean": 0.9807142972946167, "rewards/IngredientFormatReward/std": 0.1297312021255493, "rewards/IngredientMatchReward/mean": 0.6188566565513611, "rewards/IngredientMatchReward/std": 0.28720327019691466, "rewards/IngredientQuantityMatchReward/mean": 0.6479173898696899, "rewards/IngredientQuantityMatchReward/std": 0.43776082396507265, "rewards/TotalKcalExactMatchReward/mean": 0.7875, "rewards/TotalKcalExactMatchReward/std": 0.3997353255748749, "step": 1545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0265625, "completions/max_length": 513.0, "completions/mean_length": 395.8234375, "completions/min_length": 270.8, "epoch": 0.43115438108484005, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6673920154571533, "kl": 0.04912932915613055, "learning_rate": 6.527576265671794e-07, "loss": 0.0019652603194117548, "reward": 3.0716404914855957, "reward_std": 0.47040929794311526, "rewards/IngredientFormatReward/mean": 0.9731250047683716, "rewards/IngredientFormatReward/std": 0.1522574841976166, "rewards/IngredientMatchReward/mean": 0.6390248894691467, "rewards/IngredientMatchReward/std": 0.3047917664051056, "rewards/IngredientQuantityMatchReward/mean": 0.6938655853271485, "rewards/IngredientQuantityMatchReward/std": 0.39961657524108884, "rewards/TotalKcalExactMatchReward/mean": 0.765625, "rewards/TotalKcalExactMatchReward/std": 0.41862271428108216, "step": 1550 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 513.0, "completions/mean_length": 400.134375, "completions/min_length": 276.2, "epoch": 0.4325452016689847, "frac_reward_zero_std": 0.0375, "grad_norm": 0.653672993183136, "kl": 0.049115463299676775, "learning_rate": 6.505661323437544e-07, "loss": 0.001964777708053589, "reward": 3.160717725753784, "reward_std": 0.4397841155529022, "rewards/IngredientFormatReward/mean": 0.9830952405929565, "rewards/IngredientFormatReward/std": 0.12367947846651077, "rewards/IngredientMatchReward/mean": 0.6457465529441834, "rewards/IngredientMatchReward/std": 0.3012327909469604, "rewards/IngredientQuantityMatchReward/mean": 0.7271884560585022, "rewards/IngredientQuantityMatchReward/std": 0.38396180272102354, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.37747406363487246, "step": 1555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 509.8, "completions/mean_length": 397.1421875, "completions/min_length": 256.8, "epoch": 0.4339360222531293, "frac_reward_zero_std": 0.0625, "grad_norm": 0.617830753326416, "kl": 0.05066547570750117, "learning_rate": 6.483714525670955e-07, "loss": 0.0020269980654120446, "reward": 3.1229011535644533, "reward_std": 0.40159772634506224, "rewards/IngredientFormatReward/mean": 0.9884374976158142, "rewards/IngredientFormatReward/std": 0.09104880094528198, "rewards/IngredientMatchReward/mean": 0.6592550873756409, "rewards/IngredientMatchReward/std": 0.29240604639053347, "rewards/IngredientQuantityMatchReward/mean": 0.6814584970474243, "rewards/IngredientQuantityMatchReward/std": 0.3873124122619629, "rewards/TotalKcalExactMatchReward/mean": 0.79375, "rewards/TotalKcalExactMatchReward/std": 0.3895157277584076, "step": 1560 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 508.6, "completions/mean_length": 393.6171875, "completions/min_length": 235.4, "epoch": 0.43532684283727396, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6887730360031128, "kl": 0.048044915916398166, "learning_rate": 6.46173633670416e-07, "loss": 0.001921633630990982, "reward": 3.046199083328247, "reward_std": 0.40088788866996766, "rewards/IngredientFormatReward/mean": 0.9917931675910949, "rewards/IngredientFormatReward/std": 0.0687427580356598, "rewards/IngredientMatchReward/mean": 0.6264453172683716, "rewards/IngredientMatchReward/std": 0.30637386441230774, "rewards/IngredientQuantityMatchReward/mean": 0.6310855627059937, "rewards/IngredientQuantityMatchReward/std": 0.434447181224823, "rewards/TotalKcalExactMatchReward/mean": 0.796875, "rewards/TotalKcalExactMatchReward/std": 0.3885793924331665, "step": 1565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 513.0, "completions/mean_length": 398.70625, "completions/min_length": 262.8, "epoch": 0.43671766342141866, "frac_reward_zero_std": 0.05, "grad_norm": 0.7046116590499878, "kl": 0.049918401753529906, "learning_rate": 6.43972722153343e-07, "loss": 0.0019969020038843157, "reward": 2.955770397186279, "reward_std": 0.44179911613464357, "rewards/IngredientFormatReward/mean": 0.9754910826683044, "rewards/IngredientFormatReward/std": 0.1414255291223526, "rewards/IngredientMatchReward/mean": 0.6252684831619263, "rewards/IngredientMatchReward/std": 0.31577845811843874, "rewards/IngredientQuantityMatchReward/mean": 0.6143857598304748, "rewards/IngredientQuantityMatchReward/std": 0.42812495231628417, "rewards/TotalKcalExactMatchReward/mean": 0.740625, "rewards/TotalKcalExactMatchReward/std": 0.4230214834213257, "step": 1570 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 511.4, "completions/mean_length": 396.0625, "completions/min_length": 255.2, "epoch": 0.4381084840055633, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6653915643692017, "kl": 0.07790568340569734, "learning_rate": 6.417687645809358e-07, "loss": 0.0031152090057730676, "reward": 3.0399940490722654, "reward_std": 0.4159983456134796, "rewards/IngredientFormatReward/mean": 0.9914843797683716, "rewards/IngredientFormatReward/std": 0.07876742035150527, "rewards/IngredientMatchReward/mean": 0.6409877419471741, "rewards/IngredientMatchReward/std": 0.2738662242889404, "rewards/IngredientQuantityMatchReward/mean": 0.6075219869613647, "rewards/IngredientQuantityMatchReward/std": 0.41345948576927183, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.3890666842460632, "step": 1575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 513.0, "completions/mean_length": 391.325, "completions/min_length": 271.4, "epoch": 0.43949930458970793, "frac_reward_zero_std": 0.05, "grad_norm": 0.6310789585113525, "kl": 0.051219089329242705, "learning_rate": 6.395618075826987e-07, "loss": 0.002048930525779724, "reward": 3.0662962436676025, "reward_std": 0.4428023874759674, "rewards/IngredientFormatReward/mean": 0.9847916722297668, "rewards/IngredientFormatReward/std": 0.11453506052494049, "rewards/IngredientMatchReward/mean": 0.6471811294555664, "rewards/IngredientMatchReward/std": 0.306696617603302, "rewards/IngredientQuantityMatchReward/mean": 0.6358859181404114, "rewards/IngredientQuantityMatchReward/std": 0.41814287304878234, "rewards/TotalKcalExactMatchReward/mean": 0.7984375, "rewards/TotalKcalExactMatchReward/std": 0.3955422639846802, "step": 1580 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 506.6, "completions/mean_length": 389.2515625, "completions/min_length": 267.4, "epoch": 0.44089012517385257, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7244879007339478, "kl": 0.05323730851523578, "learning_rate": 6.373518978515957e-07, "loss": 0.0021291349083185198, "reward": 3.0412669658660887, "reward_std": 0.44353756308555603, "rewards/IngredientFormatReward/mean": 0.98671875, "rewards/IngredientFormatReward/std": 0.09624509960412979, "rewards/IngredientMatchReward/mean": 0.6482967615127564, "rewards/IngredientMatchReward/std": 0.27869592010974886, "rewards/IngredientQuantityMatchReward/mean": 0.6656264781951904, "rewards/IngredientQuantityMatchReward/std": 0.41459755301475526, "rewards/TotalKcalExactMatchReward/mean": 0.740625, "rewards/TotalKcalExactMatchReward/std": 0.4251006543636322, "step": 1585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 507.4, "completions/mean_length": 392.8515625, "completions/min_length": 272.8, "epoch": 0.4422809457579972, "frac_reward_zero_std": 0.05, "grad_norm": 0.6723690032958984, "kl": 0.048212344851344825, "learning_rate": 6.351390821430626e-07, "loss": 0.0019285336136817932, "reward": 3.0731667995452883, "reward_std": 0.4329515159130096, "rewards/IngredientFormatReward/mean": 0.99140625, "rewards/IngredientFormatReward/std": 0.055959947593510154, "rewards/IngredientMatchReward/mean": 0.6085950136184692, "rewards/IngredientMatchReward/std": 0.3112369000911713, "rewards/IngredientQuantityMatchReward/mean": 0.7184779644012451, "rewards/IngredientQuantityMatchReward/std": 0.376867213845253, "rewards/TotalKcalExactMatchReward/mean": 0.7546875, "rewards/TotalKcalExactMatchReward/std": 0.4265890300273895, "step": 1590 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 503.2, "completions/mean_length": 393.8125, "completions/min_length": 281.4, "epoch": 0.44367176634214184, "frac_reward_zero_std": 0.05, "grad_norm": 0.626299262046814, "kl": 0.046462705451995136, "learning_rate": 6.329234072740169e-07, "loss": 0.0018586689606308938, "reward": 3.0387376308441163, "reward_std": 0.38438517451286314, "rewards/IngredientFormatReward/mean": 0.9895573019981384, "rewards/IngredientFormatReward/std": 0.08768100291490555, "rewards/IngredientMatchReward/mean": 0.6520031332969666, "rewards/IngredientMatchReward/std": 0.29606057405471803, "rewards/IngredientQuantityMatchReward/mean": 0.659677255153656, "rewards/IngredientQuantityMatchReward/std": 0.41896759867668154, "rewards/TotalKcalExactMatchReward/mean": 0.7375, "rewards/TotalKcalExactMatchReward/std": 0.43825141787528993, "step": 1595 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 507.2, "completions/mean_length": 394.828125, "completions/min_length": 272.4, "epoch": 0.44506258692628653, "frac_reward_zero_std": 0.025, "grad_norm": 0.6427936553955078, "kl": 0.046727404044941065, "learning_rate": 6.307049201218683e-07, "loss": 0.001869230531156063, "reward": 3.151887559890747, "reward_std": 0.4053348183631897, "rewards/IngredientFormatReward/mean": 0.9919270753860474, "rewards/IngredientFormatReward/std": 0.055418914556503295, "rewards/IngredientMatchReward/mean": 0.6315675854682923, "rewards/IngredientMatchReward/std": 0.29659752249717714, "rewards/IngredientQuantityMatchReward/mean": 0.7205803394317627, "rewards/IngredientQuantityMatchReward/std": 0.3834276497364044, "rewards/TotalKcalExactMatchReward/mean": 0.8078125, "rewards/TotalKcalExactMatchReward/std": 0.39012913703918456, "step": 1600 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 512.8, "completions/mean_length": 396.3390625, "completions/min_length": 266.4, "epoch": 0.44645340751043117, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6290833950042725, "kl": 0.047926258062943816, "learning_rate": 6.284836676235261e-07, "loss": 0.0019170723855495453, "reward": 3.0292068004608153, "reward_std": 0.40457791090011597, "rewards/IngredientFormatReward/mean": 0.9802083253860474, "rewards/IngredientFormatReward/std": 0.11273946780711412, "rewards/IngredientMatchReward/mean": 0.6371831655502319, "rewards/IngredientMatchReward/std": 0.29279580116271975, "rewards/IngredientQuantityMatchReward/mean": 0.6602527856826782, "rewards/IngredientQuantityMatchReward/std": 0.42310287356376647, "rewards/TotalKcalExactMatchReward/mean": 0.7515625, "rewards/TotalKcalExactMatchReward/std": 0.42987608909606934, "step": 1605 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 511.8, "completions/mean_length": 390.19375, "completions/min_length": 262.6, "epoch": 0.4478442280945758, "frac_reward_zero_std": 0.1, "grad_norm": 0.6547244191169739, "kl": 0.04830891098827124, "learning_rate": 6.262596967744068e-07, "loss": 0.001932542771100998, "reward": 3.120107126235962, "reward_std": 0.38998834490776063, "rewards/IngredientFormatReward/mean": 0.9877306580543518, "rewards/IngredientFormatReward/std": 0.09338119924068451, "rewards/IngredientMatchReward/mean": 0.6428608655929565, "rewards/IngredientMatchReward/std": 0.30790755748748777, "rewards/IngredientQuantityMatchReward/mean": 0.6910781860351562, "rewards/IngredientQuantityMatchReward/std": 0.41570178270339964, "rewards/TotalKcalExactMatchReward/mean": 0.7984375, "rewards/TotalKcalExactMatchReward/std": 0.35660522133111955, "step": 1610 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 507.4, "completions/mean_length": 391.8015625, "completions/min_length": 268.0, "epoch": 0.44923504867872044, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6382926106452942, "kl": 0.04987063794396818, "learning_rate": 6.240330546274393e-07, "loss": 0.0019953008741140366, "reward": 3.0715692043304443, "reward_std": 0.3866082727909088, "rewards/IngredientFormatReward/mean": 0.9904687404632568, "rewards/IngredientFormatReward/std": 0.07038679048418998, "rewards/IngredientMatchReward/mean": 0.6318895101547242, "rewards/IngredientMatchReward/std": 0.30490061044692995, "rewards/IngredientQuantityMatchReward/mean": 0.6007733345031738, "rewards/IngredientQuantityMatchReward/std": 0.41641207933425906, "rewards/TotalKcalExactMatchReward/mean": 0.8484375, "rewards/TotalKcalExactMatchReward/std": 0.3494331300258636, "step": 1615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 507.6, "completions/mean_length": 392.3875, "completions/min_length": 259.4, "epoch": 0.4506258692628651, "frac_reward_zero_std": 0.0375, "grad_norm": 0.654926598072052, "kl": 0.04706335519440472, "learning_rate": 6.218037882920697e-07, "loss": 0.001882902719080448, "reward": 3.0398138999938964, "reward_std": 0.463697224855423, "rewards/IngredientFormatReward/mean": 0.9875, "rewards/IngredientFormatReward/std": 0.08298950344324112, "rewards/IngredientMatchReward/mean": 0.6279631614685058, "rewards/IngredientMatchReward/std": 0.2736996293067932, "rewards/IngredientQuantityMatchReward/mean": 0.6321632027626037, "rewards/IngredientQuantityMatchReward/std": 0.4176937580108643, "rewards/TotalKcalExactMatchReward/mean": 0.7921875, "rewards/TotalKcalExactMatchReward/std": 0.4008431792259216, "step": 1620 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 510.2, "completions/mean_length": 397.5578125, "completions/min_length": 291.6, "epoch": 0.4520166898470097, "frac_reward_zero_std": 0.075, "grad_norm": 0.6824511289596558, "kl": 0.04578695949167013, "learning_rate": 6.195719449332643e-07, "loss": 0.001831657439470291, "reward": 3.0820634365081787, "reward_std": 0.3959769606590271, "rewards/IngredientFormatReward/mean": 0.9825520873069763, "rewards/IngredientFormatReward/std": 0.11164550334215165, "rewards/IngredientMatchReward/mean": 0.6557118058204651, "rewards/IngredientMatchReward/std": 0.28708990216255187, "rewards/IngredientQuantityMatchReward/mean": 0.6781745553016663, "rewards/IngredientQuantityMatchReward/std": 0.4067881166934967, "rewards/TotalKcalExactMatchReward/mean": 0.765625, "rewards/TotalKcalExactMatchReward/std": 0.4194814503192902, "step": 1625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 502.0, "completions/mean_length": 394.9453125, "completions/min_length": 275.8, "epoch": 0.4534075104311544, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7145374417304993, "kl": 0.04789355676621199, "learning_rate": 6.173375717705123e-07, "loss": 0.001915960945188999, "reward": 3.124998092651367, "reward_std": 0.40861119627952575, "rewards/IngredientFormatReward/mean": 0.9889583349227905, "rewards/IngredientFormatReward/std": 0.08167827948927879, "rewards/IngredientMatchReward/mean": 0.6187487602233886, "rewards/IngredientMatchReward/std": 0.2515649914741516, "rewards/IngredientQuantityMatchReward/mean": 0.729790985584259, "rewards/IngredientQuantityMatchReward/std": 0.3765805721282959, "rewards/TotalKcalExactMatchReward/mean": 0.7875, "rewards/TotalKcalExactMatchReward/std": 0.40658249855041506, "step": 1630 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 505.6, "completions/mean_length": 396.221875, "completions/min_length": 280.8, "epoch": 0.45479833101529904, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6588624119758606, "kl": 0.04649864921811968, "learning_rate": 6.151007160768263e-07, "loss": 0.0018601721152663232, "reward": 3.0500646591186524, "reward_std": 0.4239701569080353, "rewards/IngredientFormatReward/mean": 0.9918303489685059, "rewards/IngredientFormatReward/std": 0.05691017508506775, "rewards/IngredientMatchReward/mean": 0.6226277112960815, "rewards/IngredientMatchReward/std": 0.2716664791107178, "rewards/IngredientQuantityMatchReward/mean": 0.6199815273284912, "rewards/IngredientQuantityMatchReward/std": 0.407462352514267, "rewards/TotalKcalExactMatchReward/mean": 0.815625, "rewards/TotalKcalExactMatchReward/std": 0.3732294499874115, "step": 1635 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 513.0, "completions/mean_length": 388.575, "completions/min_length": 264.2, "epoch": 0.4561891515994437, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6433203220367432, "kl": 0.05448652170598507, "learning_rate": 6.128614251777416e-07, "loss": 0.0021792571991682053, "reward": 3.0209889888763426, "reward_std": 0.41428587436676023, "rewards/IngredientFormatReward/mean": 0.9835137724876404, "rewards/IngredientFormatReward/std": 0.10485587120056153, "rewards/IngredientMatchReward/mean": 0.6340167284011841, "rewards/IngredientMatchReward/std": 0.2999426305294037, "rewards/IngredientQuantityMatchReward/mean": 0.6300209879875183, "rewards/IngredientQuantityMatchReward/std": 0.44520034790039065, "rewards/TotalKcalExactMatchReward/mean": 0.7734375, "rewards/TotalKcalExactMatchReward/std": 0.41060879826545715, "step": 1640 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 500.4, "completions/mean_length": 390.2140625, "completions/min_length": 270.2, "epoch": 0.4575799721835883, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6303852200508118, "kl": 0.050624200236052276, "learning_rate": 6.106197464503168e-07, "loss": 0.0020252134650945663, "reward": 3.140613651275635, "reward_std": 0.3557996809482574, "rewards/IngredientFormatReward/mean": 0.9965625047683716, "rewards/IngredientFormatReward/std": 0.03889087215065956, "rewards/IngredientMatchReward/mean": 0.6494605898857116, "rewards/IngredientMatchReward/std": 0.29842119812965395, "rewards/IngredientQuantityMatchReward/mean": 0.6992781519889831, "rewards/IngredientQuantityMatchReward/std": 0.3938143730163574, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.3920258104801178, "step": 1645 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 492.6, "completions/mean_length": 383.9453125, "completions/min_length": 259.4, "epoch": 0.45897079276773295, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7204416394233704, "kl": 0.049180832644924524, "learning_rate": 6.083757273221288e-07, "loss": 0.001967577263712883, "reward": 3.166611337661743, "reward_std": 0.3798985183238983, "rewards/IngredientFormatReward/mean": 0.9952566981315613, "rewards/IngredientFormatReward/std": 0.048683957941830155, "rewards/IngredientMatchReward/mean": 0.6497941613197327, "rewards/IngredientMatchReward/std": 0.28912633061409, "rewards/IngredientQuantityMatchReward/mean": 0.6762479662895202, "rewards/IngredientQuantityMatchReward/std": 0.4136974036693573, "rewards/TotalKcalExactMatchReward/mean": 0.8453125, "rewards/TotalKcalExactMatchReward/std": 0.35293395519256593, "step": 1650 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 507.4, "completions/mean_length": 389.878125, "completions/min_length": 259.0, "epoch": 0.4603616133518776, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6586219072341919, "kl": 0.04915962847881019, "learning_rate": 6.061294152702716e-07, "loss": 0.0019666045904159547, "reward": 3.1720734596252442, "reward_std": 0.39295960068702696, "rewards/IngredientFormatReward/mean": 0.9903497099876404, "rewards/IngredientFormatReward/std": 0.08478823751211166, "rewards/IngredientMatchReward/mean": 0.6524268507957458, "rewards/IngredientMatchReward/std": 0.2759318143129349, "rewards/IngredientQuantityMatchReward/mean": 0.6714844465255737, "rewards/IngredientQuantityMatchReward/std": 0.429459947347641, "rewards/TotalKcalExactMatchReward/mean": 0.8578125, "rewards/TotalKcalExactMatchReward/std": 0.3274064376950264, "step": 1655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 504.6, "completions/mean_length": 389.371875, "completions/min_length": 278.0, "epoch": 0.4617524339360223, "frac_reward_zero_std": 0.05, "grad_norm": 0.6691665649414062, "kl": 0.04552193239796907, "learning_rate": 6.03880857820351e-07, "loss": 0.0018210854381322862, "reward": 3.1165382862091064, "reward_std": 0.39113470911979675, "rewards/IngredientFormatReward/mean": 0.9834821343421936, "rewards/IngredientFormatReward/std": 0.07514195144176483, "rewards/IngredientMatchReward/mean": 0.6446856260299683, "rewards/IngredientMatchReward/std": 0.2858420073986053, "rewards/IngredientQuantityMatchReward/mean": 0.6758704543113708, "rewards/IngredientQuantityMatchReward/std": 0.39811512231826784, "rewards/TotalKcalExactMatchReward/mean": 0.8125, "rewards/TotalKcalExactMatchReward/std": 0.3781652688980103, "step": 1660 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 510.0, "completions/mean_length": 390.2328125, "completions/min_length": 271.8, "epoch": 0.4631432545201669, "frac_reward_zero_std": 0.075, "grad_norm": 0.7427563071250916, "kl": 0.047903071716427804, "learning_rate": 6.016301025454787e-07, "loss": 0.0019164852797985076, "reward": 3.088708591461182, "reward_std": 0.36983243227005, "rewards/IngredientFormatReward/mean": 0.9919270753860474, "rewards/IngredientFormatReward/std": 0.08860928863286972, "rewards/IngredientMatchReward/mean": 0.6738368034362793, "rewards/IngredientMatchReward/std": 0.2914969980716705, "rewards/IngredientQuantityMatchReward/mean": 0.633882200717926, "rewards/IngredientQuantityMatchReward/std": 0.4225495457649231, "rewards/TotalKcalExactMatchReward/mean": 0.7890625, "rewards/TotalKcalExactMatchReward/std": 0.39090106785297396, "step": 1665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.2, "completions/mean_length": 394.36875, "completions/min_length": 286.6, "epoch": 0.46453407510431155, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6977032423019409, "kl": 0.048976365476846695, "learning_rate": 5.99377197065266e-07, "loss": 0.001959040015935898, "reward": 3.0245996475219727, "reward_std": 0.39999868273735045, "rewards/IngredientFormatReward/mean": 0.991223955154419, "rewards/IngredientFormatReward/std": 0.06987991854548455, "rewards/IngredientMatchReward/mean": 0.6229538559913635, "rewards/IngredientMatchReward/std": 0.29490872621536257, "rewards/IngredientQuantityMatchReward/mean": 0.6401092886924744, "rewards/IngredientQuantityMatchReward/std": 0.42952102422714233, "rewards/TotalKcalExactMatchReward/mean": 0.7703125, "rewards/TotalKcalExactMatchReward/std": 0.38621187806129453, "step": 1670 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 509.6, "completions/mean_length": 394.971875, "completions/min_length": 265.2, "epoch": 0.4659248956884562, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7027736306190491, "kl": 0.05804067454300821, "learning_rate": 5.971221890448175e-07, "loss": 0.002321036159992218, "reward": 3.1033143043518066, "reward_std": 0.39811227917671205, "rewards/IngredientFormatReward/mean": 0.9859375, "rewards/IngredientFormatReward/std": 0.10241568833589554, "rewards/IngredientMatchReward/mean": 0.6516257405281067, "rewards/IngredientMatchReward/std": 0.2847370356321335, "rewards/IngredientQuantityMatchReward/mean": 0.6501261115074157, "rewards/IngredientQuantityMatchReward/std": 0.4035037100315094, "rewards/TotalKcalExactMatchReward/mean": 0.815625, "rewards/TotalKcalExactMatchReward/std": 0.383716356754303, "step": 1675 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 505.8, "completions/mean_length": 394.65, "completions/min_length": 272.0, "epoch": 0.4673157162726008, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6062320470809937, "kl": 0.04617527783848345, "learning_rate": 5.948651261937202e-07, "loss": 0.0018472731113433837, "reward": 3.1725085735321046, "reward_std": 0.3940969526767731, "rewards/IngredientFormatReward/mean": 0.9842708349227905, "rewards/IngredientFormatReward/std": 0.09453187361359597, "rewards/IngredientMatchReward/mean": 0.6568180322647095, "rewards/IngredientMatchReward/std": 0.2938332736492157, "rewards/IngredientQuantityMatchReward/mean": 0.6704821944236755, "rewards/IngredientQuantityMatchReward/std": 0.41570605635643004, "rewards/TotalKcalExactMatchReward/mean": 0.8609375, "rewards/TotalKcalExactMatchReward/std": 0.3382291287183762, "step": 1680 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 507.0, "completions/mean_length": 387.928125, "completions/min_length": 264.2, "epoch": 0.46870653685674546, "frac_reward_zero_std": 0.05, "grad_norm": 0.7044047117233276, "kl": 0.048976872209459545, "learning_rate": 5.926060562650365e-07, "loss": 0.001959339529275894, "reward": 2.9777952671051025, "reward_std": 0.4341627299785614, "rewards/IngredientFormatReward/mean": 0.983802080154419, "rewards/IngredientFormatReward/std": 0.10478676557540893, "rewards/IngredientMatchReward/mean": 0.6123015999794006, "rewards/IngredientMatchReward/std": 0.2814868152141571, "rewards/IngredientQuantityMatchReward/mean": 0.627004086971283, "rewards/IngredientQuantityMatchReward/std": 0.41599141955375674, "rewards/TotalKcalExactMatchReward/mean": 0.7546875, "rewards/TotalKcalExactMatchReward/std": 0.4197186291217804, "step": 1685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 502.6, "completions/mean_length": 389.990625, "completions/min_length": 255.4, "epoch": 0.47009735744089015, "frac_reward_zero_std": 0.125, "grad_norm": 0.644118070602417, "kl": 0.04798829292412847, "learning_rate": 5.903450270542924e-07, "loss": 0.0019195247441530228, "reward": 3.198506736755371, "reward_std": 0.41594612002372744, "rewards/IngredientFormatReward/mean": 0.986517870426178, "rewards/IngredientFormatReward/std": 0.09814060274511575, "rewards/IngredientMatchReward/mean": 0.6614857912063599, "rewards/IngredientMatchReward/std": 0.2818374246358871, "rewards/IngredientQuantityMatchReward/mean": 0.6973781704902648, "rewards/IngredientQuantityMatchReward/std": 0.40497069954872134, "rewards/TotalKcalExactMatchReward/mean": 0.853125, "rewards/TotalKcalExactMatchReward/std": 0.34283066987991334, "step": 1690 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 512.2, "completions/mean_length": 394.8625, "completions/min_length": 287.2, "epoch": 0.4714881780250348, "frac_reward_zero_std": 0.05, "grad_norm": 0.7088591456413269, "kl": 0.04952654310036451, "learning_rate": 5.880820863984671e-07, "loss": 0.001980846002697945, "reward": 2.948525238037109, "reward_std": 0.4502517580986023, "rewards/IngredientFormatReward/mean": 0.9842187523841858, "rewards/IngredientFormatReward/std": 0.08936193883419037, "rewards/IngredientMatchReward/mean": 0.5951304972171784, "rewards/IngredientMatchReward/std": 0.28045451641082764, "rewards/IngredientQuantityMatchReward/mean": 0.5848009943962097, "rewards/IngredientQuantityMatchReward/std": 0.43246378302574157, "rewards/TotalKcalExactMatchReward/mean": 0.784375, "rewards/TotalKcalExactMatchReward/std": 0.39880833625793455, "step": 1695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 499.2, "completions/mean_length": 384.971875, "completions/min_length": 279.4, "epoch": 0.4728789986091794, "frac_reward_zero_std": 0.1125, "grad_norm": 0.6974106431007385, "kl": 0.048825663654133676, "learning_rate": 5.858172821749804e-07, "loss": 0.0019530490040779113, "reward": 3.026969051361084, "reward_std": 0.3304957687854767, "rewards/IngredientFormatReward/mean": 0.9939843773841858, "rewards/IngredientFormatReward/std": 0.03864100463688373, "rewards/IngredientMatchReward/mean": 0.5842813670635223, "rewards/IngredientMatchReward/std": 0.2920273244380951, "rewards/IngredientQuantityMatchReward/mean": 0.6471408128738403, "rewards/IngredientQuantityMatchReward/std": 0.41217195987701416, "rewards/TotalKcalExactMatchReward/mean": 0.8015625, "rewards/TotalKcalExactMatchReward/std": 0.3969415783882141, "step": 1700 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 502.8, "completions/mean_length": 386.540625, "completions/min_length": 265.0, "epoch": 0.47426981919332406, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6712289452552795, "kl": 0.05012807142920792, "learning_rate": 5.835506623006797e-07, "loss": 0.002004893682897091, "reward": 3.0048501968383787, "reward_std": 0.4382961392402649, "rewards/IngredientFormatReward/mean": 0.9930059671401977, "rewards/IngredientFormatReward/std": 0.051491785980761054, "rewards/IngredientMatchReward/mean": 0.6337772965431213, "rewards/IngredientMatchReward/std": 0.305075603723526, "rewards/IngredientQuantityMatchReward/mean": 0.6108795642852783, "rewards/IngredientQuantityMatchReward/std": 0.43121986389160155, "rewards/TotalKcalExactMatchReward/mean": 0.7671875, "rewards/TotalKcalExactMatchReward/std": 0.4157248854637146, "step": 1705 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 499.2, "completions/mean_length": 387.99375, "completions/min_length": 263.2, "epoch": 0.4756606397774687, "frac_reward_zero_std": 0.1, "grad_norm": 0.6293528079986572, "kl": 0.048464122042059896, "learning_rate": 5.812822747308267e-07, "loss": 0.0019388705492019652, "reward": 3.175932788848877, "reward_std": 0.3702836275100708, "rewards/IngredientFormatReward/mean": 0.9876116037368774, "rewards/IngredientFormatReward/std": 0.07741203494369983, "rewards/IngredientMatchReward/mean": 0.659701144695282, "rewards/IngredientMatchReward/std": 0.29684895277023315, "rewards/IngredientQuantityMatchReward/mean": 0.7348701000213623, "rewards/IngredientQuantityMatchReward/std": 0.37621399760246277, "rewards/TotalKcalExactMatchReward/mean": 0.79375, "rewards/TotalKcalExactMatchReward/std": 0.397750324010849, "step": 1710 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 504.0, "completions/mean_length": 389.025, "completions/min_length": 238.4, "epoch": 0.47705146036161333, "frac_reward_zero_std": 0.025, "grad_norm": 0.6771330833435059, "kl": 0.049341540411114694, "learning_rate": 5.790121674580825e-07, "loss": 0.0019737355411052703, "reward": 2.981400489807129, "reward_std": 0.3796704918146133, "rewards/IngredientFormatReward/mean": 0.9787500023841857, "rewards/IngredientFormatReward/std": 0.10464536398649216, "rewards/IngredientMatchReward/mean": 0.6600198388099671, "rewards/IngredientMatchReward/std": 0.2891321390867233, "rewards/IngredientQuantityMatchReward/mean": 0.5941931962966919, "rewards/IngredientQuantityMatchReward/std": 0.40178399682044985, "rewards/TotalKcalExactMatchReward/mean": 0.7484375, "rewards/TotalKcalExactMatchReward/std": 0.4310473442077637, "step": 1715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 499.2, "completions/mean_length": 387.9921875, "completions/min_length": 253.6, "epoch": 0.47844228094575797, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6627811789512634, "kl": 0.04793977346271276, "learning_rate": 5.767403885114922e-07, "loss": 0.0019301587715744972, "reward": 3.085385274887085, "reward_std": 0.36830160617828367, "rewards/IngredientFormatReward/mean": 0.9941145777702332, "rewards/IngredientFormatReward/std": 0.04807176664471626, "rewards/IngredientMatchReward/mean": 0.6792026281356811, "rewards/IngredientMatchReward/std": 0.2884396731853485, "rewards/IngredientQuantityMatchReward/mean": 0.6370681166648865, "rewards/IngredientQuantityMatchReward/std": 0.4236335575580597, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.40007728040218354, "step": 1720 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 512.6, "completions/mean_length": 395.703125, "completions/min_length": 256.8, "epoch": 0.47983310152990266, "frac_reward_zero_std": 0.025, "grad_norm": 0.6980003118515015, "kl": 0.051507193874567744, "learning_rate": 5.744669859554688e-07, "loss": 0.0020601864904165267, "reward": 2.9863351345062257, "reward_std": 0.4324962854385376, "rewards/IngredientFormatReward/mean": 0.984587037563324, "rewards/IngredientFormatReward/std": 0.0999862365424633, "rewards/IngredientMatchReward/mean": 0.6219358801841736, "rewards/IngredientMatchReward/std": 0.29111440777778624, "rewards/IngredientQuantityMatchReward/mean": 0.643874716758728, "rewards/IngredientQuantityMatchReward/std": 0.41615907549858094, "rewards/TotalKcalExactMatchReward/mean": 0.7359375, "rewards/TotalKcalExactMatchReward/std": 0.43581517934799197, "step": 1725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 507.4, "completions/mean_length": 393.6796875, "completions/min_length": 257.6, "epoch": 0.4812239221140473, "frac_reward_zero_std": 0.075, "grad_norm": 0.652874231338501, "kl": 0.04963922812603414, "learning_rate": 5.721920078887763e-07, "loss": 0.0019854392856359484, "reward": 3.042120361328125, "reward_std": 0.3817890167236328, "rewards/IngredientFormatReward/mean": 0.9817125558853149, "rewards/IngredientFormatReward/std": 0.11481560580432415, "rewards/IngredientMatchReward/mean": 0.6558786153793335, "rewards/IngredientMatchReward/std": 0.2739634424448013, "rewards/IngredientQuantityMatchReward/mean": 0.6482792258262634, "rewards/IngredientQuantityMatchReward/std": 0.4263168752193451, "rewards/TotalKcalExactMatchReward/mean": 0.75625, "rewards/TotalKcalExactMatchReward/std": 0.41720872521400454, "step": 1730 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 503.8, "completions/mean_length": 395.4546875, "completions/min_length": 266.4, "epoch": 0.48261474269819193, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6124057769775391, "kl": 0.045404232246801256, "learning_rate": 5.699155024435123e-07, "loss": 0.0018159780651330949, "reward": 3.0864189147949217, "reward_std": 0.34279492497444153, "rewards/IngredientFormatReward/mean": 0.9932775378227234, "rewards/IngredientFormatReward/std": 0.05568651407957077, "rewards/IngredientMatchReward/mean": 0.6521255016326905, "rewards/IngredientMatchReward/std": 0.2828336536884308, "rewards/IngredientQuantityMatchReward/mean": 0.622265899181366, "rewards/IngredientQuantityMatchReward/std": 0.4201075792312622, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.38487215638160704, "step": 1735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 512.8, "completions/mean_length": 394.9109375, "completions/min_length": 261.4, "epoch": 0.48400556328233657, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6815362572669983, "kl": 0.04877625331282616, "learning_rate": 5.676375177840886e-07, "loss": 0.0019514326006174088, "reward": 3.0187916278839113, "reward_std": 0.42291984558105467, "rewards/IngredientFormatReward/mean": 0.9903645753860474, "rewards/IngredientFormatReward/std": 0.08810437638312578, "rewards/IngredientMatchReward/mean": 0.6148133635520935, "rewards/IngredientMatchReward/std": 0.2883909821510315, "rewards/IngredientQuantityMatchReward/mean": 0.6151761889457703, "rewards/IngredientQuantityMatchReward/std": 0.4394866168498993, "rewards/TotalKcalExactMatchReward/mean": 0.7984375, "rewards/TotalKcalExactMatchReward/std": 0.39994245767593384, "step": 1740 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 501.6, "completions/mean_length": 390.396875, "completions/min_length": 268.4, "epoch": 0.4853963838664812, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6231136918067932, "kl": 0.047634016210213306, "learning_rate": 5.653581021062138e-07, "loss": 0.0019053995609283447, "reward": 3.1004568576812743, "reward_std": 0.4429864943027496, "rewards/IngredientFormatReward/mean": 0.9809375047683716, "rewards/IngredientFormatReward/std": 0.08808905854821206, "rewards/IngredientMatchReward/mean": 0.6266797065734864, "rewards/IngredientMatchReward/std": 0.27853635549545286, "rewards/IngredientQuantityMatchReward/mean": 0.6850271940231323, "rewards/IngredientQuantityMatchReward/std": 0.41117426156997683, "rewards/TotalKcalExactMatchReward/mean": 0.8078125, "rewards/TotalKcalExactMatchReward/std": 0.3709612190723419, "step": 1745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 512.8, "completions/mean_length": 400.534375, "completions/min_length": 275.2, "epoch": 0.48678720445062584, "frac_reward_zero_std": 0.05, "grad_norm": 0.6302758455276489, "kl": 0.055369544448331, "learning_rate": 5.630773036358727e-07, "loss": 0.0022154580801725388, "reward": 3.1487196922302245, "reward_std": 0.43740108609199524, "rewards/IngredientFormatReward/mean": 0.9723288774490356, "rewards/IngredientFormatReward/std": 0.13584012035280466, "rewards/IngredientMatchReward/mean": 0.6218018412590027, "rewards/IngredientMatchReward/std": 0.29499236941337587, "rewards/IngredientQuantityMatchReward/mean": 0.715526533126831, "rewards/IngredientQuantityMatchReward/std": 0.3941978871822357, "rewards/TotalKcalExactMatchReward/mean": 0.8390625, "rewards/TotalKcalExactMatchReward/std": 0.3547331839799881, "step": 1750 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 502.8, "completions/mean_length": 393.821875, "completions/min_length": 278.6, "epoch": 0.48817802503477054, "frac_reward_zero_std": 0.0125, "grad_norm": 0.655293345451355, "kl": 0.046050224895589055, "learning_rate": 5.607951706283055e-07, "loss": 0.0018420252948999406, "reward": 3.1131840705871583, "reward_std": 0.39082067608833315, "rewards/IngredientFormatReward/mean": 0.9950520753860473, "rewards/IngredientFormatReward/std": 0.04335647970438004, "rewards/IngredientMatchReward/mean": 0.6560224533081055, "rewards/IngredientMatchReward/std": 0.2750636339187622, "rewards/IngredientQuantityMatchReward/mean": 0.6605471014976502, "rewards/IngredientQuantityMatchReward/std": 0.4154274582862854, "rewards/TotalKcalExactMatchReward/mean": 0.8015625, "rewards/TotalKcalExactMatchReward/std": 0.39623318910598754, "step": 1755 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 508.0, "completions/mean_length": 389.0453125, "completions/min_length": 257.4, "epoch": 0.48956884561891517, "frac_reward_zero_std": 0.025, "grad_norm": 0.7058917880058289, "kl": 0.05053422078490257, "learning_rate": 5.585117513669882e-07, "loss": 0.002021782286465168, "reward": 3.0528096675872805, "reward_std": 0.46687796115875246, "rewards/IngredientFormatReward/mean": 0.9771986722946167, "rewards/IngredientFormatReward/std": 0.12420394718647003, "rewards/IngredientMatchReward/mean": 0.6485894083976745, "rewards/IngredientMatchReward/std": 0.29609904885292054, "rewards/IngredientQuantityMatchReward/mean": 0.6395214915275573, "rewards/IngredientQuantityMatchReward/std": 0.4058083415031433, "rewards/TotalKcalExactMatchReward/mean": 0.7875, "rewards/TotalKcalExactMatchReward/std": 0.3928277313709259, "step": 1760 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 509.6, "completions/mean_length": 401.221875, "completions/min_length": 272.6, "epoch": 0.4909596662030598, "frac_reward_zero_std": 0.0125, "grad_norm": 0.659774899482727, "kl": 0.04758750724140555, "learning_rate": 5.562270941626098e-07, "loss": 0.001903420127928257, "reward": 3.078034543991089, "reward_std": 0.4054509222507477, "rewards/IngredientFormatReward/mean": 0.9913541555404664, "rewards/IngredientFormatReward/std": 0.06280860994011164, "rewards/IngredientMatchReward/mean": 0.6158724069595337, "rewards/IngredientMatchReward/std": 0.273968768119812, "rewards/IngredientQuantityMatchReward/mean": 0.6911204814910888, "rewards/IngredientQuantityMatchReward/std": 0.3966470003128052, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.4132148861885071, "step": 1765 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 513.0, "completions/mean_length": 396.8234375, "completions/min_length": 256.2, "epoch": 0.49235048678720444, "frac_reward_zero_std": 0.05, "grad_norm": 4.437183380126953, "kl": 0.06853532809764147, "learning_rate": 5.539412473520507e-07, "loss": 0.0027404358610510827, "reward": 3.176833152770996, "reward_std": 0.3862785041332245, "rewards/IngredientFormatReward/mean": 0.9880208253860474, "rewards/IngredientFormatReward/std": 0.1034691423177719, "rewards/IngredientMatchReward/mean": 0.6355412840843201, "rewards/IngredientMatchReward/std": 0.2928665101528168, "rewards/IngredientQuantityMatchReward/mean": 0.6907709836959839, "rewards/IngredientQuantityMatchReward/std": 0.3815356969833374, "rewards/TotalKcalExactMatchReward/mean": 0.8625, "rewards/TotalKcalExactMatchReward/std": 0.3310780256986618, "step": 1770 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 505.8, "completions/mean_length": 399.0546875, "completions/min_length": 283.2, "epoch": 0.4937413073713491, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6291945576667786, "kl": 0.047819344280287626, "learning_rate": 5.516542592973603e-07, "loss": 0.0019129764288663865, "reward": 2.977151393890381, "reward_std": 0.47986037731170655, "rewards/IngredientFormatReward/mean": 0.9804017782211304, "rewards/IngredientFormatReward/std": 0.11696080565452575, "rewards/IngredientMatchReward/mean": 0.6131088972091675, "rewards/IngredientMatchReward/std": 0.3076077878475189, "rewards/IngredientQuantityMatchReward/mean": 0.635203194618225, "rewards/IngredientQuantityMatchReward/std": 0.4294419288635254, "rewards/TotalKcalExactMatchReward/mean": 0.7484375, "rewards/TotalKcalExactMatchReward/std": 0.41134634613990784, "step": 1775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 511.8, "completions/mean_length": 394.38125, "completions/min_length": 256.8, "epoch": 0.4951321279554937, "frac_reward_zero_std": 0.075, "grad_norm": 0.6367654204368591, "kl": 0.04918671539053321, "learning_rate": 5.49366178384733e-07, "loss": 0.0019677009433507918, "reward": 3.022269535064697, "reward_std": 0.42463927865028384, "rewards/IngredientFormatReward/mean": 0.9825520753860474, "rewards/IngredientFormatReward/std": 0.11480289716273546, "rewards/IngredientMatchReward/mean": 0.6272984862327575, "rewards/IngredientMatchReward/std": 0.3080913841724396, "rewards/IngredientQuantityMatchReward/mean": 0.6374189138412476, "rewards/IngredientQuantityMatchReward/std": 0.44004141092300414, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.4037832021713257, "step": 1780 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 511.2, "completions/mean_length": 395.1953125, "completions/min_length": 271.6, "epoch": 0.4965229485396384, "frac_reward_zero_std": 0.05, "grad_norm": 0.6336233615875244, "kl": 0.05262123909778893, "learning_rate": 5.470770530234855e-07, "loss": 0.002105065435171127, "reward": 3.0902578353881838, "reward_std": 0.3941184222698212, "rewards/IngredientFormatReward/mean": 0.9919345259666443, "rewards/IngredientFormatReward/std": 0.0732752576470375, "rewards/IngredientMatchReward/mean": 0.644134521484375, "rewards/IngredientMatchReward/std": 0.282639217376709, "rewards/IngredientQuantityMatchReward/mean": 0.6432512640953064, "rewards/IngredientQuantityMatchReward/std": 0.41275890469551085, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.3679554224014282, "step": 1785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 505.0, "completions/mean_length": 400.9296875, "completions/min_length": 299.4, "epoch": 0.49791376912378305, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6558573842048645, "kl": 0.04567183582112193, "learning_rate": 5.447869316450318e-07, "loss": 0.0018267896026372909, "reward": 3.1581130981445313, "reward_std": 0.3598249554634094, "rewards/IngredientFormatReward/mean": 0.9941666841506958, "rewards/IngredientFormatReward/std": 0.04706086441874504, "rewards/IngredientMatchReward/mean": 0.6860336184501648, "rewards/IngredientMatchReward/std": 0.2729302167892456, "rewards/IngredientQuantityMatchReward/mean": 0.6732252895832062, "rewards/IngredientQuantityMatchReward/std": 0.3966924250125885, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.3842876791954041, "step": 1790 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.6, "completions/mean_length": 396.1359375, "completions/min_length": 287.4, "epoch": 0.4993045897079277, "frac_reward_zero_std": 0.1, "grad_norm": 0.5723157525062561, "kl": 0.04732802438084036, "learning_rate": 5.424958627018586e-07, "loss": 0.001893388107419014, "reward": 3.118370294570923, "reward_std": 0.37871933579444883, "rewards/IngredientFormatReward/mean": 0.98984375, "rewards/IngredientFormatReward/std": 0.08600255399942398, "rewards/IngredientMatchReward/mean": 0.681352949142456, "rewards/IngredientMatchReward/std": 0.27151423394680024, "rewards/IngredientQuantityMatchReward/mean": 0.6377986788749694, "rewards/IngredientQuantityMatchReward/std": 0.4133506715297699, "rewards/TotalKcalExactMatchReward/mean": 0.809375, "rewards/TotalKcalExactMatchReward/std": 0.3859631597995758, "step": 1795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 509.8, "completions/mean_length": 392.56875, "completions/min_length": 273.8, "epoch": 0.5006954102920723, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6640133261680603, "kl": 0.05193311013281345, "learning_rate": 5.402038946665011e-07, "loss": 0.0020771391689777376, "reward": 3.1144729137420653, "reward_std": 0.4295255720615387, "rewards/IngredientFormatReward/mean": 0.9897916555404663, "rewards/IngredientFormatReward/std": 0.07800195086747408, "rewards/IngredientMatchReward/mean": 0.6456776976585388, "rewards/IngredientMatchReward/std": 0.30382000207901, "rewards/IngredientQuantityMatchReward/mean": 0.6586910247802734, "rewards/IngredientQuantityMatchReward/std": 0.4264285624027252, "rewards/TotalKcalExactMatchReward/mean": 0.8203125, "rewards/TotalKcalExactMatchReward/std": 0.3754447340965271, "step": 1800 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 510.8, "completions/mean_length": 396.7265625, "completions/min_length": 285.0, "epoch": 0.502086230876217, "frac_reward_zero_std": 0.05, "grad_norm": 0.612971305847168, "kl": 0.052038407209329306, "learning_rate": 5.37911076030516e-07, "loss": 0.0020814632996916773, "reward": 2.9839890956878663, "reward_std": 0.4456018149852753, "rewards/IngredientFormatReward/mean": 0.9828125, "rewards/IngredientFormatReward/std": 0.10817223787307739, "rewards/IngredientMatchReward/mean": 0.5959939360618591, "rewards/IngredientMatchReward/std": 0.3077159345149994, "rewards/IngredientQuantityMatchReward/mean": 0.6067451357841491, "rewards/IngredientQuantityMatchReward/std": 0.41565820574760437, "rewards/TotalKcalExactMatchReward/mean": 0.7984375, "rewards/TotalKcalExactMatchReward/std": 0.38725520074367525, "step": 1805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 506.2, "completions/mean_length": 395.128125, "completions/min_length": 252.2, "epoch": 0.5034770514603616, "frac_reward_zero_std": 0.075, "grad_norm": 0.6170816421508789, "kl": 0.04653078094124794, "learning_rate": 5.356174553034564e-07, "loss": 0.0018611878156661987, "reward": 3.077640676498413, "reward_std": 0.41207525730133054, "rewards/IngredientFormatReward/mean": 0.9891145944595336, "rewards/IngredientFormatReward/std": 0.08445280343294144, "rewards/IngredientMatchReward/mean": 0.6564167976379395, "rewards/IngredientMatchReward/std": 0.2918438494205475, "rewards/IngredientQuantityMatchReward/mean": 0.658671748638153, "rewards/IngredientQuantityMatchReward/std": 0.4138647198677063, "rewards/TotalKcalExactMatchReward/mean": 0.7734375, "rewards/TotalKcalExactMatchReward/std": 0.4117767035961151, "step": 1810 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 512.0, "completions/mean_length": 400.0734375, "completions/min_length": 255.0, "epoch": 0.5048678720445062, "frac_reward_zero_std": 0.025, "grad_norm": 0.6442963480949402, "kl": 0.056444175005890426, "learning_rate": 5.333230810118461e-07, "loss": 0.0022573621943593027, "reward": 3.116063928604126, "reward_std": 0.4383660852909088, "rewards/IngredientFormatReward/mean": 0.9827827334403991, "rewards/IngredientFormatReward/std": 0.11086281388998032, "rewards/IngredientMatchReward/mean": 0.6368998050689697, "rewards/IngredientMatchReward/std": 0.2972823202610016, "rewards/IngredientQuantityMatchReward/mean": 0.6682563424110413, "rewards/IngredientQuantityMatchReward/std": 0.3951943576335907, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.37783130407333376, "step": 1815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 505.4, "completions/mean_length": 387.6359375, "completions/min_length": 251.8, "epoch": 0.5062586926286509, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6564891934394836, "kl": 0.05077703422866762, "learning_rate": 5.310280016981512e-07, "loss": 0.002030998095870018, "reward": 3.0581398963928224, "reward_std": 0.4139727771282196, "rewards/IngredientFormatReward/mean": 0.9908854246139527, "rewards/IngredientFormatReward/std": 0.07461793459951878, "rewards/IngredientMatchReward/mean": 0.6378273844718934, "rewards/IngredientMatchReward/std": 0.28663045167922974, "rewards/IngredientQuantityMatchReward/mean": 0.6044270992279053, "rewards/IngredientQuantityMatchReward/std": 0.42292520403862, "rewards/TotalKcalExactMatchReward/mean": 0.825, "rewards/TotalKcalExactMatchReward/std": 0.37856464385986327, "step": 1820 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 512.6, "completions/mean_length": 394.4625, "completions/min_length": 240.6, "epoch": 0.5076495132127955, "frac_reward_zero_std": 0.05, "grad_norm": 8.344937324523926, "kl": 0.08002238939516246, "learning_rate": 5.287322659197547e-07, "loss": 0.0032052285969257355, "reward": 2.9394122123718263, "reward_std": 0.4769141972064972, "rewards/IngredientFormatReward/mean": 0.9648660778999328, "rewards/IngredientFormatReward/std": 0.17725641131401063, "rewards/IngredientMatchReward/mean": 0.6225856423377991, "rewards/IngredientMatchReward/std": 0.29008139967918395, "rewards/IngredientQuantityMatchReward/mean": 0.6519605159759522, "rewards/IngredientQuantityMatchReward/std": 0.42122418284416197, "rewards/TotalKcalExactMatchReward/mean": 0.7, "rewards/TotalKcalExactMatchReward/std": 0.44271482825279235, "step": 1825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 513.0, "completions/mean_length": 395.7015625, "completions/min_length": 277.0, "epoch": 0.5090403337969402, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6441317796707153, "kl": 0.05329477796331048, "learning_rate": 5.264359222479284e-07, "loss": 0.0021320804953575134, "reward": 2.9984073638916016, "reward_std": 0.3900154232978821, "rewards/IngredientFormatReward/mean": 0.9801413774490356, "rewards/IngredientFormatReward/std": 0.13333013355731965, "rewards/IngredientMatchReward/mean": 0.6038113951683044, "rewards/IngredientMatchReward/std": 0.2904983043670654, "rewards/IngredientQuantityMatchReward/mean": 0.5894545555114746, "rewards/IngredientQuantityMatchReward/std": 0.4164754033088684, "rewards/TotalKcalExactMatchReward/mean": 0.825, "rewards/TotalKcalExactMatchReward/std": 0.3774557411670685, "step": 1830 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 506.0, "completions/mean_length": 394.346875, "completions/min_length": 231.2, "epoch": 0.5104311543810849, "frac_reward_zero_std": 0.05, "grad_norm": 0.571604311466217, "kl": 0.048023554123938085, "learning_rate": 5.241390192668053e-07, "loss": 0.0019212931394577027, "reward": 3.1547540187835694, "reward_std": 0.41128666400909425, "rewards/IngredientFormatReward/mean": 0.9815755128860474, "rewards/IngredientFormatReward/std": 0.1034016104415059, "rewards/IngredientMatchReward/mean": 0.662100088596344, "rewards/IngredientMatchReward/std": 0.2865631103515625, "rewards/IngredientQuantityMatchReward/mean": 0.7142034649848938, "rewards/IngredientQuantityMatchReward/std": 0.3982077479362488, "rewards/TotalKcalExactMatchReward/mean": 0.796875, "rewards/TotalKcalExactMatchReward/std": 0.39826048016548155, "step": 1835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.040625, "completions/max_length": 513.0, "completions/mean_length": 401.8265625, "completions/min_length": 278.4, "epoch": 0.5118219749652295, "frac_reward_zero_std": 0.025, "grad_norm": 0.6243442893028259, "kl": 0.049475864064879714, "learning_rate": 5.218416055723517e-07, "loss": 0.0019794130697846414, "reward": 3.047327184677124, "reward_std": 0.42559523582458497, "rewards/IngredientFormatReward/mean": 0.9601413607597351, "rewards/IngredientFormatReward/std": 0.18492761999368668, "rewards/IngredientMatchReward/mean": 0.6405964970588685, "rewards/IngredientMatchReward/std": 0.2941777765750885, "rewards/IngredientQuantityMatchReward/mean": 0.6387767672538758, "rewards/IngredientQuantityMatchReward/std": 0.4147272765636444, "rewards/TotalKcalExactMatchReward/mean": 0.8078125, "rewards/TotalKcalExactMatchReward/std": 0.3916107892990112, "step": 1840 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 509.4, "completions/mean_length": 398.18125, "completions/min_length": 289.2, "epoch": 0.5132127955493742, "frac_reward_zero_std": 0.025, "grad_norm": 0.6171333193778992, "kl": 0.048615764314308765, "learning_rate": 5.195437297713396e-07, "loss": 0.001944425329566002, "reward": 3.102466917037964, "reward_std": 0.4576135456562042, "rewards/IngredientFormatReward/mean": 0.9871726155281066, "rewards/IngredientFormatReward/std": 0.08229567054659129, "rewards/IngredientMatchReward/mean": 0.6418576717376709, "rewards/IngredientMatchReward/std": 0.2877155005931854, "rewards/IngredientQuantityMatchReward/mean": 0.61249920129776, "rewards/IngredientQuantityMatchReward/std": 0.41496188640594484, "rewards/TotalKcalExactMatchReward/mean": 0.8609375, "rewards/TotalKcalExactMatchReward/std": 0.3430387437343597, "step": 1845 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 500.4, "completions/mean_length": 393.8796875, "completions/min_length": 265.2, "epoch": 0.5146036161335188, "frac_reward_zero_std": 0.075, "grad_norm": 0.5736840963363647, "kl": 0.04830085132271052, "learning_rate": 5.172454404803176e-07, "loss": 0.0019320830702781676, "reward": 3.0672821044921874, "reward_std": 0.4006498992443085, "rewards/IngredientFormatReward/mean": 0.9946875095367431, "rewards/IngredientFormatReward/std": 0.04643735066056252, "rewards/IngredientMatchReward/mean": 0.635732901096344, "rewards/IngredientMatchReward/std": 0.29260424375534055, "rewards/IngredientQuantityMatchReward/mean": 0.6556117177009583, "rewards/IngredientQuantityMatchReward/std": 0.4205364763736725, "rewards/TotalKcalExactMatchReward/mean": 0.78125, "rewards/TotalKcalExactMatchReward/std": 0.4017998993396759, "step": 1850 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 509.4, "completions/mean_length": 396.253125, "completions/min_length": 264.8, "epoch": 0.5159944367176634, "frac_reward_zero_std": 0.025, "grad_norm": 0.6130569577217102, "kl": 0.049074625875800845, "learning_rate": 5.149467863245823e-07, "loss": 0.0019630441442131997, "reward": 3.0270339012145997, "reward_std": 0.4051498770713806, "rewards/IngredientFormatReward/mean": 0.9796875, "rewards/IngredientFormatReward/std": 0.08117598444223403, "rewards/IngredientMatchReward/mean": 0.627546489238739, "rewards/IngredientMatchReward/std": 0.27951192259788515, "rewards/IngredientQuantityMatchReward/mean": 0.6651124238967896, "rewards/IngredientQuantityMatchReward/std": 0.41736636161804197, "rewards/TotalKcalExactMatchReward/mean": 0.7546875, "rewards/TotalKcalExactMatchReward/std": 0.42597426772117614, "step": 1855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 513.0, "completions/mean_length": 395.4453125, "completions/min_length": 284.8, "epoch": 0.5173852573018081, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6463583707809448, "kl": 0.045586379477754235, "learning_rate": 5.126478159371503e-07, "loss": 0.0018231194466352464, "reward": 3.189420700073242, "reward_std": 0.36946871876716614, "rewards/IngredientFormatReward/mean": 0.9851041674613953, "rewards/IngredientFormatReward/std": 0.09958937615156174, "rewards/IngredientMatchReward/mean": 0.6497030019760132, "rewards/IngredientMatchReward/std": 0.29318724274635316, "rewards/IngredientQuantityMatchReward/mean": 0.7218011021614075, "rewards/IngredientQuantityMatchReward/std": 0.36770058870315553, "rewards/TotalKcalExactMatchReward/mean": 0.8328125, "rewards/TotalKcalExactMatchReward/std": 0.37152652740478515, "step": 1860 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 513.0, "completions/mean_length": 401.04375, "completions/min_length": 285.0, "epoch": 0.5187760778859527, "frac_reward_zero_std": 0.05, "grad_norm": 0.65671706199646, "kl": 0.051340340590104463, "learning_rate": 5.103485779577284e-07, "loss": 0.002053655683994293, "reward": 3.0907777309417725, "reward_std": 0.4277846872806549, "rewards/IngredientFormatReward/mean": 0.9821614503860474, "rewards/IngredientFormatReward/std": 0.1232692077755928, "rewards/IngredientMatchReward/mean": 0.6434660196304322, "rewards/IngredientMatchReward/std": 0.29241865277290346, "rewards/IngredientQuantityMatchReward/mean": 0.6682752728462219, "rewards/IngredientQuantityMatchReward/std": 0.40513462424278257, "rewards/TotalKcalExactMatchReward/mean": 0.796875, "rewards/TotalKcalExactMatchReward/std": 0.4023512899875641, "step": 1865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0265625, "completions/max_length": 513.0, "completions/mean_length": 388.9578125, "completions/min_length": 270.4, "epoch": 0.5201668984700973, "frac_reward_zero_std": 0.025, "grad_norm": 0.7248250246047974, "kl": 0.05152199740987271, "learning_rate": 5.08049121031685e-07, "loss": 0.002060762792825699, "reward": 3.028227710723877, "reward_std": 0.46579166650772097, "rewards/IngredientFormatReward/mean": 0.9721874952316284, "rewards/IngredientFormatReward/std": 0.15746290534734725, "rewards/IngredientMatchReward/mean": 0.6362196326255798, "rewards/IngredientMatchReward/std": 0.306441468000412, "rewards/IngredientQuantityMatchReward/mean": 0.6495080471038819, "rewards/IngredientQuantityMatchReward/std": 0.4341913163661957, "rewards/TotalKcalExactMatchReward/mean": 0.7703125, "rewards/TotalKcalExactMatchReward/std": 0.41461523771286013, "step": 1870 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 506.6, "completions/mean_length": 395.9140625, "completions/min_length": 284.8, "epoch": 0.521557719054242, "frac_reward_zero_std": 0.075, "grad_norm": 0.616625964641571, "kl": 0.04834558917209506, "learning_rate": 5.057494938090211e-07, "loss": 0.001933823898434639, "reward": 3.110442638397217, "reward_std": 0.41063175201416013, "rewards/IngredientFormatReward/mean": 0.9895089268684387, "rewards/IngredientFormatReward/std": 0.07405889481306076, "rewards/IngredientMatchReward/mean": 0.6785714387893677, "rewards/IngredientMatchReward/std": 0.2774973541498184, "rewards/IngredientQuantityMatchReward/mean": 0.6517373204231263, "rewards/IngredientQuantityMatchReward/std": 0.39638755321502683, "rewards/TotalKcalExactMatchReward/mean": 0.790625, "rewards/TotalKcalExactMatchReward/std": 0.39762794971466064, "step": 1875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 506.6, "completions/mean_length": 391.709375, "completions/min_length": 276.4, "epoch": 0.5229485396383866, "frac_reward_zero_std": 0.0125, "grad_norm": 0.665547251701355, "kl": 0.04781521300319582, "learning_rate": 5.034497449433401e-07, "loss": 0.0019125932827591895, "reward": 3.0387969493865965, "reward_std": 0.42612879276275634, "rewards/IngredientFormatReward/mean": 0.9903125047683716, "rewards/IngredientFormatReward/std": 0.07061336636543274, "rewards/IngredientMatchReward/mean": 0.6220145106315613, "rewards/IngredientMatchReward/std": 0.2822760850191116, "rewards/IngredientQuantityMatchReward/mean": 0.6280323743820191, "rewards/IngredientQuantityMatchReward/std": 0.4209755420684814, "rewards/TotalKcalExactMatchReward/mean": 0.7984375, "rewards/TotalKcalExactMatchReward/std": 0.37942943572998045, "step": 1880 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 511.8, "completions/mean_length": 385.7609375, "completions/min_length": 254.0, "epoch": 0.5243393602225312, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6137979030609131, "kl": 0.05282691507600248, "learning_rate": 5.011499230908194e-07, "loss": 0.002113470435142517, "reward": 3.118752431869507, "reward_std": 0.41371476650238037, "rewards/IngredientFormatReward/mean": 0.9895312547683716, "rewards/IngredientFormatReward/std": 0.08896933421492577, "rewards/IngredientMatchReward/mean": 0.646869421005249, "rewards/IngredientMatchReward/std": 0.285538986325264, "rewards/IngredientQuantityMatchReward/mean": 0.6636016845703125, "rewards/IngredientQuantityMatchReward/std": 0.4197295904159546, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.37991697788238527, "step": 1885 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 509.8, "completions/mean_length": 396.703125, "completions/min_length": 287.0, "epoch": 0.5257301808066759, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6123167276382446, "kl": 0.04784087296575308, "learning_rate": 4.988500769091808e-07, "loss": 0.0019137132912874222, "reward": 2.978975248336792, "reward_std": 0.41777234673500063, "rewards/IngredientFormatReward/mean": 0.9920833110809326, "rewards/IngredientFormatReward/std": 0.07661278918385506, "rewards/IngredientMatchReward/mean": 0.668877112865448, "rewards/IngredientMatchReward/std": 0.2592278867959976, "rewards/IngredientQuantityMatchReward/mean": 0.5773897886276245, "rewards/IngredientQuantityMatchReward/std": 0.4320713520050049, "rewards/TotalKcalExactMatchReward/mean": 0.740625, "rewards/TotalKcalExactMatchReward/std": 0.4287377059459686, "step": 1890 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 510.0, "completions/mean_length": 394.1625, "completions/min_length": 252.8, "epoch": 0.5271210013908206, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6603817343711853, "kl": 0.04811024349182844, "learning_rate": 4.9655025505666e-07, "loss": 0.0019248005002737046, "reward": 3.117777681350708, "reward_std": 0.42486504316329954, "rewards/IngredientFormatReward/mean": 0.9875, "rewards/IngredientFormatReward/std": 0.08148004412651062, "rewards/IngredientMatchReward/mean": 0.6686545133590698, "rewards/IngredientMatchReward/std": 0.2971228897571564, "rewards/IngredientQuantityMatchReward/mean": 0.6897481679916382, "rewards/IngredientQuantityMatchReward/std": 0.39132343530654906, "rewards/TotalKcalExactMatchReward/mean": 0.771875, "rewards/TotalKcalExactMatchReward/std": 0.4117573142051697, "step": 1895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 499.6, "completions/mean_length": 390.7015625, "completions/min_length": 272.6, "epoch": 0.5285118219749653, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6234140992164612, "kl": 0.05170197752304375, "learning_rate": 4.942505061909789e-07, "loss": 0.0020677197724580767, "reward": 3.123731803894043, "reward_std": 0.38712977766990664, "rewards/IngredientFormatReward/mean": 0.992968738079071, "rewards/IngredientFormatReward/std": 0.05120236147195101, "rewards/IngredientMatchReward/mean": 0.6404768109321595, "rewards/IngredientMatchReward/std": 0.2947970926761627, "rewards/IngredientQuantityMatchReward/mean": 0.6902862906455993, "rewards/IngredientQuantityMatchReward/std": 0.41358540654182435, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.39628866314888, "step": 1900 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 513.0, "completions/mean_length": 396.00625, "completions/min_length": 273.0, "epoch": 0.5299026425591099, "frac_reward_zero_std": 0.05, "grad_norm": 0.6213961839675903, "kl": 0.051998325670138, "learning_rate": 4.919508789683148e-07, "loss": 0.002079806849360466, "reward": 2.9941901683807375, "reward_std": 0.44365312457084655, "rewards/IngredientFormatReward/mean": 0.98125, "rewards/IngredientFormatReward/std": 0.12953428775072098, "rewards/IngredientMatchReward/mean": 0.6144717335700989, "rewards/IngredientMatchReward/std": 0.27872682809829713, "rewards/IngredientQuantityMatchReward/mean": 0.6562810420989991, "rewards/IngredientQuantityMatchReward/std": 0.42560343742370604, "rewards/TotalKcalExactMatchReward/mean": 0.7421875, "rewards/TotalKcalExactMatchReward/std": 0.4294995665550232, "step": 1905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 505.8, "completions/mean_length": 397.371875, "completions/min_length": 258.6, "epoch": 0.5312934631432545, "frac_reward_zero_std": 0.025, "grad_norm": 0.6720630526542664, "kl": 0.051461770036257806, "learning_rate": 4.896514220422716e-07, "loss": 0.0020584441721439362, "reward": 3.0832706928253173, "reward_std": 0.43686205744743345, "rewards/IngredientFormatReward/mean": 0.9894661426544189, "rewards/IngredientFormatReward/std": 0.07865497097373009, "rewards/IngredientMatchReward/mean": 0.6534127116203308, "rewards/IngredientMatchReward/std": 0.2737223029136658, "rewards/IngredientQuantityMatchReward/mean": 0.6482043623924255, "rewards/IngredientQuantityMatchReward/std": 0.393373167514801, "rewards/TotalKcalExactMatchReward/mean": 0.7921875, "rewards/TotalKcalExactMatchReward/std": 0.39381408095359804, "step": 1910 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 505.8, "completions/mean_length": 390.590625, "completions/min_length": 258.4, "epoch": 0.5326842837273992, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6541943550109863, "kl": 0.04684328753501177, "learning_rate": 4.873521840628498e-07, "loss": 0.0018735647201538086, "reward": 3.1317606925964356, "reward_std": 0.3644884526729584, "rewards/IngredientFormatReward/mean": 0.9934895753860473, "rewards/IngredientFormatReward/std": 0.052749037928879264, "rewards/IngredientMatchReward/mean": 0.6564211845397949, "rewards/IngredientMatchReward/std": 0.2947027921676636, "rewards/IngredientQuantityMatchReward/mean": 0.6568499028682708, "rewards/IngredientQuantityMatchReward/std": 0.4014035701751709, "rewards/TotalKcalExactMatchReward/mean": 0.825, "rewards/TotalKcalExactMatchReward/std": 0.37870767116546633, "step": 1915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 508.8, "completions/mean_length": 387.7453125, "completions/min_length": 261.0, "epoch": 0.5340751043115438, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6302548050880432, "kl": 0.04872983023524284, "learning_rate": 4.850532136754177e-07, "loss": 0.0019491085782647132, "reward": 3.096049499511719, "reward_std": 0.3624439001083374, "rewards/IngredientFormatReward/mean": 0.9915624976158142, "rewards/IngredientFormatReward/std": 0.06838876903057098, "rewards/IngredientMatchReward/mean": 0.6316896080970764, "rewards/IngredientMatchReward/std": 0.29681622982025146, "rewards/IngredientQuantityMatchReward/mean": 0.6462348580360413, "rewards/IngredientQuantityMatchReward/std": 0.40400696396827696, "rewards/TotalKcalExactMatchReward/mean": 0.8265625, "rewards/TotalKcalExactMatchReward/std": 0.3677105069160461, "step": 1920 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 504.8, "completions/mean_length": 392.2953125, "completions/min_length": 272.6, "epoch": 0.5354659248956884, "frac_reward_zero_std": 0.075, "grad_norm": 0.7258259654045105, "kl": 0.05029868185520172, "learning_rate": 4.827545595196824e-07, "loss": 0.002012159302830696, "reward": 3.0869573593139648, "reward_std": 0.3494667589664459, "rewards/IngredientFormatReward/mean": 0.9827585577964782, "rewards/IngredientFormatReward/std": 0.09563635978847743, "rewards/IngredientMatchReward/mean": 0.6535475254058838, "rewards/IngredientMatchReward/std": 0.29701568484306334, "rewards/IngredientQuantityMatchReward/mean": 0.6256512045860291, "rewards/IngredientQuantityMatchReward/std": 0.43892085552215576, "rewards/TotalKcalExactMatchReward/mean": 0.825, "rewards/TotalKcalExactMatchReward/std": 0.37521823644638064, "step": 1925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 496.2, "completions/mean_length": 391.184375, "completions/min_length": 274.4, "epoch": 0.5368567454798331, "frac_reward_zero_std": 0.075, "grad_norm": 0.6211183071136475, "kl": 0.04746301844716072, "learning_rate": 4.804562702286604e-07, "loss": 0.001898600347340107, "reward": 3.1789096355438233, "reward_std": 0.349788224697113, "rewards/IngredientFormatReward/mean": 0.99296875, "rewards/IngredientFormatReward/std": 0.05298392586410046, "rewards/IngredientMatchReward/mean": 0.6501227736473083, "rewards/IngredientMatchReward/std": 0.2874899566173553, "rewards/IngredientQuantityMatchReward/mean": 0.6701931238174439, "rewards/IngredientQuantityMatchReward/std": 0.4015944302082062, "rewards/TotalKcalExactMatchReward/mean": 0.865625, "rewards/TotalKcalExactMatchReward/std": 0.3344068914651871, "step": 1930 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 508.8, "completions/mean_length": 396.7921875, "completions/min_length": 276.6, "epoch": 0.5382475660639777, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6513813734054565, "kl": 0.04723638026043773, "learning_rate": 4.781583944276484e-07, "loss": 0.001889600232243538, "reward": 3.0589586734771728, "reward_std": 0.3765470445156097, "rewards/IngredientFormatReward/mean": 0.9848958373069763, "rewards/IngredientFormatReward/std": 0.10098038241267204, "rewards/IngredientMatchReward/mean": 0.6359009265899658, "rewards/IngredientMatchReward/std": 0.2934915781021118, "rewards/IngredientQuantityMatchReward/mean": 0.6553495287895202, "rewards/IngredientQuantityMatchReward/std": 0.407476407289505, "rewards/TotalKcalExactMatchReward/mean": 0.7828125, "rewards/TotalKcalExactMatchReward/std": 0.4078086793422699, "step": 1935 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.4, "completions/mean_length": 396.7640625, "completions/min_length": 264.8, "epoch": 0.5396383866481224, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6842653155326843, "kl": 0.05088149183429778, "learning_rate": 4.7586098073319477e-07, "loss": 0.002035403996706009, "reward": 3.08815450668335, "reward_std": 0.4408495545387268, "rewards/IngredientFormatReward/mean": 0.9842559576034546, "rewards/IngredientFormatReward/std": 0.11024485677480697, "rewards/IngredientMatchReward/mean": 0.6107713341712951, "rewards/IngredientMatchReward/std": 0.26985863745212557, "rewards/IngredientQuantityMatchReward/mean": 0.6790647387504578, "rewards/IngredientQuantityMatchReward/std": 0.40906696319580077, "rewards/TotalKcalExactMatchReward/mean": 0.8140625, "rewards/TotalKcalExactMatchReward/std": 0.38226786255836487, "step": 1940 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 508.0, "completions/mean_length": 387.1421875, "completions/min_length": 241.0, "epoch": 0.541029207232267, "frac_reward_zero_std": 0.025, "grad_norm": 0.6372060775756836, "kl": 0.051733718090690675, "learning_rate": 4.735640777520716e-07, "loss": 0.0020697489380836487, "reward": 3.1369984626770018, "reward_std": 0.41252625584602354, "rewards/IngredientFormatReward/mean": 0.9944010376930237, "rewards/IngredientFormatReward/std": 0.051148696616292, "rewards/IngredientMatchReward/mean": 0.6390866875648499, "rewards/IngredientMatchReward/std": 0.29094711542129514, "rewards/IngredientQuantityMatchReward/mean": 0.6597607374191284, "rewards/IngredientQuantityMatchReward/std": 0.40805192589759826, "rewards/TotalKcalExactMatchReward/mean": 0.84375, "rewards/TotalKcalExactMatchReward/std": 0.3470917284488678, "step": 1945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 504.8, "completions/mean_length": 390.95625, "completions/min_length": 248.2, "epoch": 0.5424200278164116, "frac_reward_zero_std": 0.075, "grad_norm": 16.270816802978516, "kl": 0.11754101505503059, "learning_rate": 4.7126773408024536e-07, "loss": 0.004704966396093369, "reward": 3.0708155155181887, "reward_std": 0.3849705934524536, "rewards/IngredientFormatReward/mean": 0.9884895801544189, "rewards/IngredientFormatReward/std": 0.08361713178455829, "rewards/IngredientMatchReward/mean": 0.6275595426559448, "rewards/IngredientMatchReward/std": 0.30697761178016664, "rewards/IngredientQuantityMatchReward/mean": 0.6485163748264313, "rewards/IngredientQuantityMatchReward/std": 0.41862033009529115, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.39109028577804567, "step": 1950 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 500.2, "completions/mean_length": 396.546875, "completions/min_length": 267.6, "epoch": 0.5438108484005564, "frac_reward_zero_std": 0.0625, "grad_norm": 26.406909942626953, "kl": 0.10048759747296572, "learning_rate": 4.689719983018488e-07, "loss": 0.004031499475240707, "reward": 3.0238998413085936, "reward_std": 0.37822062373161314, "rewards/IngredientFormatReward/mean": 0.9881770849227905, "rewards/IngredientFormatReward/std": 0.07832716181874275, "rewards/IngredientMatchReward/mean": 0.6181553840637207, "rewards/IngredientMatchReward/std": 0.29799574613571167, "rewards/IngredientQuantityMatchReward/mean": 0.5863174438476563, "rewards/IngredientQuantityMatchReward/std": 0.4404876172542572, "rewards/TotalKcalExactMatchReward/mean": 0.83125, "rewards/TotalKcalExactMatchReward/std": 0.3680418610572815, "step": 1955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 511.0, "completions/mean_length": 401.3109375, "completions/min_length": 271.8, "epoch": 0.545201668984701, "frac_reward_zero_std": 0.05, "grad_norm": 0.6423550248146057, "kl": 0.047905878722667695, "learning_rate": 4.666769189881539e-07, "loss": 0.0019165009260177612, "reward": 3.1259632110595703, "reward_std": 0.46272686719894407, "rewards/IngredientFormatReward/mean": 0.9774181604385376, "rewards/IngredientFormatReward/std": 0.1268059030175209, "rewards/IngredientMatchReward/mean": 0.6502027630805969, "rewards/IngredientMatchReward/std": 0.29270836114883425, "rewards/IngredientQuantityMatchReward/mean": 0.6874048829078674, "rewards/IngredientQuantityMatchReward/std": 0.40191797018051145, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.39044166207313535, "step": 1960 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 497.2, "completions/mean_length": 383.03125, "completions/min_length": 239.2, "epoch": 0.5465924895688457, "frac_reward_zero_std": 0.025, "grad_norm": 0.7022526860237122, "kl": 0.051917689852416515, "learning_rate": 4.6438254469654346e-07, "loss": 0.0020762814208865167, "reward": 3.08801589012146, "reward_std": 0.39448719620704653, "rewards/IngredientFormatReward/mean": 0.9934895753860473, "rewards/IngredientFormatReward/std": 0.05592806488275528, "rewards/IngredientMatchReward/mean": 0.6122198224067688, "rewards/IngredientMatchReward/std": 0.29717632532119753, "rewards/IngredientQuantityMatchReward/mean": 0.6713690161705017, "rewards/IngredientQuantityMatchReward/std": 0.4097886860370636, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.3882238745689392, "step": 1965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 512.8, "completions/mean_length": 390.5203125, "completions/min_length": 262.6, "epoch": 0.5479833101529903, "frac_reward_zero_std": 0.0625, "grad_norm": 0.702995777130127, "kl": 0.052073144866153595, "learning_rate": 4.6208892396948415e-07, "loss": 0.0020829498767852782, "reward": 3.050513982772827, "reward_std": 0.42421574592590333, "rewards/IngredientFormatReward/mean": 0.9853236675262451, "rewards/IngredientFormatReward/std": 0.10791139267385005, "rewards/IngredientMatchReward/mean": 0.6532893180847168, "rewards/IngredientMatchReward/std": 0.28823532462120055, "rewards/IngredientQuantityMatchReward/mean": 0.6759635448455811, "rewards/IngredientQuantityMatchReward/std": 0.41645088195800783, "rewards/TotalKcalExactMatchReward/mean": 0.7359375, "rewards/TotalKcalExactMatchReward/std": 0.40241574943065644, "step": 1970 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 511.6, "completions/mean_length": 391.7546875, "completions/min_length": 266.2, "epoch": 0.5493741307371349, "frac_reward_zero_std": 0.05, "grad_norm": 0.5921576619148254, "kl": 0.04812497249804437, "learning_rate": 4.59796105333499e-07, "loss": 0.0019246641546487807, "reward": 3.038289928436279, "reward_std": 0.3981735408306122, "rewards/IngredientFormatReward/mean": 0.9900000095367432, "rewards/IngredientFormatReward/std": 0.07664789929986, "rewards/IngredientMatchReward/mean": 0.625024163722992, "rewards/IngredientMatchReward/std": 0.29402685165405273, "rewards/IngredientQuantityMatchReward/mean": 0.6451407432556152, "rewards/IngredientQuantityMatchReward/std": 0.42612059116363527, "rewards/TotalKcalExactMatchReward/mean": 0.778125, "rewards/TotalKcalExactMatchReward/std": 0.40204805731773374, "step": 1975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 508.8, "completions/mean_length": 402.7, "completions/min_length": 276.0, "epoch": 0.5507649513212796, "frac_reward_zero_std": 0.05, "grad_norm": 0.6542710661888123, "kl": 0.04761419745627791, "learning_rate": 4.5750413729814134e-07, "loss": 0.0019045617431402207, "reward": 3.0831342220306395, "reward_std": 0.4208559811115265, "rewards/IngredientFormatReward/mean": 0.9742559432983399, "rewards/IngredientFormatReward/std": 0.12943890187889337, "rewards/IngredientMatchReward/mean": 0.654008400440216, "rewards/IngredientMatchReward/std": 0.2880619168281555, "rewards/IngredientQuantityMatchReward/mean": 0.6736199378967285, "rewards/IngredientQuantityMatchReward/std": 0.37951394319534304, "rewards/TotalKcalExactMatchReward/mean": 0.78125, "rewards/TotalKcalExactMatchReward/std": 0.4055814027786255, "step": 1980 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 511.8, "completions/mean_length": 396.68125, "completions/min_length": 259.6, "epoch": 0.5521557719054242, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6633925437927246, "kl": 0.09954718144144863, "learning_rate": 4.552130683549682e-07, "loss": 0.003990843892097473, "reward": 3.1635501384735107, "reward_std": 0.4512316048145294, "rewards/IngredientFormatReward/mean": 0.9836718797683716, "rewards/IngredientFormatReward/std": 0.10353007614612579, "rewards/IngredientMatchReward/mean": 0.7020808815956116, "rewards/IngredientMatchReward/std": 0.2649115890264511, "rewards/IngredientQuantityMatchReward/mean": 0.6449849724769592, "rewards/IngredientQuantityMatchReward/std": 0.42183072566986085, "rewards/TotalKcalExactMatchReward/mean": 0.8328125, "rewards/TotalKcalExactMatchReward/std": 0.3696657299995422, "step": 1985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 512.8, "completions/mean_length": 393.4203125, "completions/min_length": 262.8, "epoch": 0.5535465924895688, "frac_reward_zero_std": 0.0, "grad_norm": 0.699221134185791, "kl": 0.05136008146218955, "learning_rate": 4.529229469765147e-07, "loss": 0.002054634690284729, "reward": 3.000317668914795, "reward_std": 0.4171613872051239, "rewards/IngredientFormatReward/mean": 0.983411455154419, "rewards/IngredientFormatReward/std": 0.1190970316529274, "rewards/IngredientMatchReward/mean": 0.6175824522972106, "rewards/IngredientMatchReward/std": 0.2949405133724213, "rewards/IngredientQuantityMatchReward/mean": 0.6352612376213074, "rewards/IngredientQuantityMatchReward/std": 0.4082452058792114, "rewards/TotalKcalExactMatchReward/mean": 0.7640625, "rewards/TotalKcalExactMatchReward/std": 0.40400839447975156, "step": 1990 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 500.6, "completions/mean_length": 394.9859375, "completions/min_length": 268.6, "epoch": 0.5549374130737135, "frac_reward_zero_std": 0.0625, "grad_norm": 1.7067408561706543, "kl": 0.05321628218516707, "learning_rate": 4.50633821615267e-07, "loss": 0.0021282849833369256, "reward": 3.1403940677642823, "reward_std": 0.3678104758262634, "rewards/IngredientFormatReward/mean": 0.9927938938140869, "rewards/IngredientFormatReward/std": 0.05982886739075184, "rewards/IngredientMatchReward/mean": 0.6677821159362793, "rewards/IngredientMatchReward/std": 0.2782954514026642, "rewards/IngredientQuantityMatchReward/mean": 0.6579430580139161, "rewards/IngredientQuantityMatchReward/std": 0.40156732201576234, "rewards/TotalKcalExactMatchReward/mean": 0.821875, "rewards/TotalKcalExactMatchReward/std": 0.3716013550758362, "step": 1995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 509.6, "completions/mean_length": 391.5359375, "completions/min_length": 244.0, "epoch": 0.5563282336578581, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6526069045066833, "kl": 0.04948800450656563, "learning_rate": 4.4834574070263974e-07, "loss": 0.001979481242597103, "reward": 3.083600330352783, "reward_std": 0.4555691659450531, "rewards/IngredientFormatReward/mean": 0.9757143020629883, "rewards/IngredientFormatReward/std": 0.13633684441447258, "rewards/IngredientMatchReward/mean": 0.6698400378227234, "rewards/IngredientMatchReward/std": 0.29061307311058043, "rewards/IngredientQuantityMatchReward/mean": 0.6161710500717164, "rewards/IngredientQuantityMatchReward/std": 0.4445948779582977, "rewards/TotalKcalExactMatchReward/mean": 0.821875, "rewards/TotalKcalExactMatchReward/std": 0.3704223304986954, "step": 2000 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 499.6, "completions/mean_length": 397.13125, "completions/min_length": 281.2, "epoch": 0.5577190542420027, "frac_reward_zero_std": 0.0375, "grad_norm": 0.640072226524353, "kl": 0.05212169520091266, "learning_rate": 4.460587526479492e-07, "loss": 0.002084844559431076, "reward": 3.1748961925506594, "reward_std": 0.3355372905731201, "rewards/IngredientFormatReward/mean": 0.9914843797683716, "rewards/IngredientFormatReward/std": 0.05400210823863745, "rewards/IngredientMatchReward/mean": 0.6208835721015931, "rewards/IngredientMatchReward/std": 0.27205938696861265, "rewards/IngredientQuantityMatchReward/mean": 0.7047157406806945, "rewards/IngredientQuantityMatchReward/std": 0.3964652955532074, "rewards/TotalKcalExactMatchReward/mean": 0.8578125, "rewards/TotalKcalExactMatchReward/std": 0.344100821018219, "step": 2005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 509.0, "completions/mean_length": 394.5671875, "completions/min_length": 266.0, "epoch": 0.5591098748261474, "frac_reward_zero_std": 0.0875, "grad_norm": 0.5653246641159058, "kl": 0.04876701077446342, "learning_rate": 4.4377290583739023e-07, "loss": 0.0019505947828292846, "reward": 3.094419002532959, "reward_std": 0.43140562176704406, "rewards/IngredientFormatReward/mean": 0.9779278397560119, "rewards/IngredientFormatReward/std": 0.13397145569324492, "rewards/IngredientMatchReward/mean": 0.6553794503211975, "rewards/IngredientMatchReward/std": 0.30059259533882143, "rewards/IngredientQuantityMatchReward/mean": 0.6423617362976074, "rewards/IngredientQuantityMatchReward/std": 0.4208782434463501, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.36920258700847625, "step": 2010 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 506.2, "completions/mean_length": 387.225, "completions/min_length": 254.6, "epoch": 0.5605006954102921, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7243109941482544, "kl": 0.0509596262825653, "learning_rate": 4.414882486330118e-07, "loss": 0.0020386766642332076, "reward": 2.980712413787842, "reward_std": 0.4302122414112091, "rewards/IngredientFormatReward/mean": 0.9727083325386048, "rewards/IngredientFormatReward/std": 0.12047353163361549, "rewards/IngredientMatchReward/mean": 0.6474566102027893, "rewards/IngredientMatchReward/std": 0.28137449622154237, "rewards/IngredientQuantityMatchReward/mean": 0.5964849531650543, "rewards/IngredientQuantityMatchReward/std": 0.43391476273536683, "rewards/TotalKcalExactMatchReward/mean": 0.7640625, "rewards/TotalKcalExactMatchReward/std": 0.41384188532829286, "step": 2015 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.8, "completions/mean_length": 388.784375, "completions/min_length": 259.4, "epoch": 0.5618915159944368, "frac_reward_zero_std": 0.075, "grad_norm": 0.7031833529472351, "kl": 0.04954910618253052, "learning_rate": 4.392048293716945e-07, "loss": 0.0019819369539618493, "reward": 3.0167937755584715, "reward_std": 0.42304224967956544, "rewards/IngredientFormatReward/mean": 0.9863020658493042, "rewards/IngredientFormatReward/std": 0.09604550898075104, "rewards/IngredientMatchReward/mean": 0.6214434504508972, "rewards/IngredientMatchReward/std": 0.30990185141563414, "rewards/IngredientQuantityMatchReward/mean": 0.6402982592582702, "rewards/IngredientQuantityMatchReward/std": 0.42372643351554873, "rewards/TotalKcalExactMatchReward/mean": 0.76875, "rewards/TotalKcalExactMatchReward/std": 0.42207556366920473, "step": 2020 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0296875, "completions/max_length": 512.8, "completions/mean_length": 395.628125, "completions/min_length": 260.6, "epoch": 0.5632823365785814, "frac_reward_zero_std": 0.075, "grad_norm": 0.5899977684020996, "kl": 0.05259674787521362, "learning_rate": 4.3692269636412735e-07, "loss": 0.0021039582788944244, "reward": 3.0533965110778807, "reward_std": 0.44099304676055906, "rewards/IngredientFormatReward/mean": 0.9689062356948852, "rewards/IngredientFormatReward/std": 0.15842351466417312, "rewards/IngredientMatchReward/mean": 0.663708484172821, "rewards/IngredientMatchReward/std": 0.2866365730762482, "rewards/IngredientQuantityMatchReward/mean": 0.6770317912101745, "rewards/IngredientQuantityMatchReward/std": 0.40612250566482544, "rewards/TotalKcalExactMatchReward/mean": 0.74375, "rewards/TotalKcalExactMatchReward/std": 0.4267735958099365, "step": 2025 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 502.8, "completions/mean_length": 399.5, "completions/min_length": 294.4, "epoch": 0.564673157162726, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6661276817321777, "kl": 0.0543745540548116, "learning_rate": 4.3464189789378623e-07, "loss": 0.0021728880703449248, "reward": 3.174680805206299, "reward_std": 0.3439257264137268, "rewards/IngredientFormatReward/mean": 0.9936979055404663, "rewards/IngredientFormatReward/std": 0.05086542461067438, "rewards/IngredientMatchReward/mean": 0.6711421132087707, "rewards/IngredientMatchReward/std": 0.2759589582681656, "rewards/IngredientQuantityMatchReward/mean": 0.7098408579826355, "rewards/IngredientQuantityMatchReward/std": 0.38981047868728635, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.39488060474395753, "step": 2030 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 509.2, "completions/mean_length": 387.4421875, "completions/min_length": 256.0, "epoch": 0.5660639777468707, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7035443782806396, "kl": 0.05331933759152889, "learning_rate": 4.3236248221591155e-07, "loss": 0.0021326553076505663, "reward": 3.0655640602111816, "reward_std": 0.43879054188728334, "rewards/IngredientFormatReward/mean": 0.9842187523841858, "rewards/IngredientFormatReward/std": 0.0974664755165577, "rewards/IngredientMatchReward/mean": 0.6237977385520935, "rewards/IngredientMatchReward/std": 0.26321753561496736, "rewards/IngredientQuantityMatchReward/mean": 0.6575475811958313, "rewards/IngredientQuantityMatchReward/std": 0.4133492708206177, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.39513932466506957, "step": 2035 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 511.0, "completions/mean_length": 394.975, "completions/min_length": 254.4, "epoch": 0.5674547983310153, "frac_reward_zero_std": 0.1, "grad_norm": 0.7625192403793335, "kl": 0.051801764173433185, "learning_rate": 4.300844975564878e-07, "loss": 0.0020724233239889146, "reward": 3.065888023376465, "reward_std": 0.40156723856925963, "rewards/IngredientFormatReward/mean": 0.9851637005805969, "rewards/IngredientFormatReward/std": 0.09742565155029297, "rewards/IngredientMatchReward/mean": 0.6349423289299011, "rewards/IngredientMatchReward/std": 0.28503390550613406, "rewards/IngredientQuantityMatchReward/mean": 0.6426570415496826, "rewards/IngredientQuantityMatchReward/std": 0.42421199679374694, "rewards/TotalKcalExactMatchReward/mean": 0.803125, "rewards/TotalKcalExactMatchReward/std": 0.3835489392280579, "step": 2040 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 499.6, "completions/mean_length": 390.475, "completions/min_length": 267.0, "epoch": 0.56884561891516, "frac_reward_zero_std": 0.075, "grad_norm": 0.6383020281791687, "kl": 0.04932130335364491, "learning_rate": 4.278079921112235e-07, "loss": 0.0019727183505892755, "reward": 3.1172127723693848, "reward_std": 0.3688182234764099, "rewards/IngredientFormatReward/mean": 0.9934895753860473, "rewards/IngredientFormatReward/std": 0.04815645068883896, "rewards/IngredientMatchReward/mean": 0.6810300469398498, "rewards/IngredientMatchReward/std": 0.28476625084877016, "rewards/IngredientQuantityMatchReward/mean": 0.6755056738853454, "rewards/IngredientQuantityMatchReward/std": 0.40387414693832396, "rewards/TotalKcalExactMatchReward/mean": 0.7671875, "rewards/TotalKcalExactMatchReward/std": 0.41967945694923403, "step": 2045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 509.6, "completions/mean_length": 394.609375, "completions/min_length": 254.4, "epoch": 0.5702364394993046, "frac_reward_zero_std": 0.0, "grad_norm": 0.6549330353736877, "kl": 0.05572653063572943, "learning_rate": 4.2553301404453117e-07, "loss": 0.0022291053086519242, "reward": 2.9725974559783936, "reward_std": 0.4430975794792175, "rewards/IngredientFormatReward/mean": 0.9778125047683716, "rewards/IngredientFormatReward/std": 0.1242400899529457, "rewards/IngredientMatchReward/mean": 0.5863467276096344, "rewards/IngredientMatchReward/std": 0.2854700267314911, "rewards/IngredientQuantityMatchReward/mean": 0.5896882534027099, "rewards/IngredientQuantityMatchReward/std": 0.43062989711761473, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.37587525844573977, "step": 2050 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 508.2, "completions/mean_length": 393.278125, "completions/min_length": 278.6, "epoch": 0.5716272600834492, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6046156883239746, "kl": 0.0488880162127316, "learning_rate": 4.232596114885078e-07, "loss": 0.0019555903971195223, "reward": 3.081752300262451, "reward_std": 0.37385761737823486, "rewards/IngredientFormatReward/mean": 0.9907775282859802, "rewards/IngredientFormatReward/std": 0.07260209694504738, "rewards/IngredientMatchReward/mean": 0.5988665580749511, "rewards/IngredientMatchReward/std": 0.2849681168794632, "rewards/IngredientQuantityMatchReward/mean": 0.679608142375946, "rewards/IngredientQuantityMatchReward/std": 0.4193470299243927, "rewards/TotalKcalExactMatchReward/mean": 0.8125, "rewards/TotalKcalExactMatchReward/std": 0.3588409751653671, "step": 2055 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 503.8, "completions/mean_length": 385.078125, "completions/min_length": 271.2, "epoch": 0.5730180806675939, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6553982496261597, "kl": 0.05303655038587749, "learning_rate": 4.209878325419175e-07, "loss": 0.0021214175969362257, "reward": 2.986969470977783, "reward_std": 0.4297593176364899, "rewards/IngredientFormatReward/mean": 0.9858854174613952, "rewards/IngredientFormatReward/std": 0.07832327783107758, "rewards/IngredientMatchReward/mean": 0.6215869665145874, "rewards/IngredientMatchReward/std": 0.3030324518680573, "rewards/IngredientQuantityMatchReward/mean": 0.6326221823692322, "rewards/IngredientQuantityMatchReward/std": 0.4269497454166412, "rewards/TotalKcalExactMatchReward/mean": 0.746875, "rewards/TotalKcalExactMatchReward/std": 0.4260698974132538, "step": 2060 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 508.8, "completions/mean_length": 398.115625, "completions/min_length": 281.0, "epoch": 0.5744089012517385, "frac_reward_zero_std": 0.075, "grad_norm": 0.6358456015586853, "kl": 0.04783306939061731, "learning_rate": 4.187177252691733e-07, "loss": 0.0019132602959871293, "reward": 2.9515925884246825, "reward_std": 0.40451722145080565, "rewards/IngredientFormatReward/mean": 0.9834486603736877, "rewards/IngredientFormatReward/std": 0.09680206961929798, "rewards/IngredientMatchReward/mean": 0.6217875838279724, "rewards/IngredientMatchReward/std": 0.2716970294713974, "rewards/IngredientQuantityMatchReward/mean": 0.5947938621044159, "rewards/IngredientQuantityMatchReward/std": 0.42498272061347964, "rewards/TotalKcalExactMatchReward/mean": 0.7515625, "rewards/TotalKcalExactMatchReward/std": 0.42524816393852233, "step": 2065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 488.6, "completions/mean_length": 386.153125, "completions/min_length": 233.2, "epoch": 0.5757997218358831, "frac_reward_zero_std": 0.05, "grad_norm": 0.6399543881416321, "kl": 0.04871748448349535, "learning_rate": 4.164493376993204e-07, "loss": 0.0019484382122755052, "reward": 3.167618465423584, "reward_std": 0.39762314558029177, "rewards/IngredientFormatReward/mean": 0.9763392925262451, "rewards/IngredientFormatReward/std": 0.08031712025403977, "rewards/IngredientMatchReward/mean": 0.6643613576889038, "rewards/IngredientMatchReward/std": 0.27025269269943236, "rewards/IngredientQuantityMatchReward/mean": 0.689417815208435, "rewards/IngredientQuantityMatchReward/std": 0.3921951472759247, "rewards/TotalKcalExactMatchReward/mean": 0.8375, "rewards/TotalKcalExactMatchReward/std": 0.3567990481853485, "step": 2070 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 510.0, "completions/mean_length": 392.71875, "completions/min_length": 275.6, "epoch": 0.5771905424200278, "frac_reward_zero_std": 0.025, "grad_norm": 0.6949378252029419, "kl": 0.04844976761378348, "learning_rate": 4.141827178250197e-07, "loss": 0.001937885209918022, "reward": 3.137814426422119, "reward_std": 0.3677864849567413, "rewards/IngredientFormatReward/mean": 0.9881119847297668, "rewards/IngredientFormatReward/std": 0.09051166772842408, "rewards/IngredientMatchReward/mean": 0.6526115655899047, "rewards/IngredientMatchReward/std": 0.2787861585617065, "rewards/IngredientQuantityMatchReward/mean": 0.6767783999443054, "rewards/IngredientQuantityMatchReward/std": 0.3985203385353088, "rewards/TotalKcalExactMatchReward/mean": 0.8203125, "rewards/TotalKcalExactMatchReward/std": 0.37351303100585936, "step": 2075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 507.4, "completions/mean_length": 393.246875, "completions/min_length": 253.8, "epoch": 0.5785813630041725, "frac_reward_zero_std": 0.0875, "grad_norm": 0.603931188583374, "kl": 0.04737849328666925, "learning_rate": 4.1191791360153283e-07, "loss": 0.0018953494727611542, "reward": 3.06357889175415, "reward_std": 0.4036316812038422, "rewards/IngredientFormatReward/mean": 0.9819791555404663, "rewards/IngredientFormatReward/std": 0.09684197474271058, "rewards/IngredientMatchReward/mean": 0.6268632292747498, "rewards/IngredientMatchReward/std": 0.2809248685836792, "rewards/IngredientQuantityMatchReward/mean": 0.6234864950180053, "rewards/IngredientQuantityMatchReward/std": 0.4336868166923523, "rewards/TotalKcalExactMatchReward/mean": 0.83125, "rewards/TotalKcalExactMatchReward/std": 0.3656547605991364, "step": 2080 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 504.0, "completions/mean_length": 386.89375, "completions/min_length": 262.4, "epoch": 0.5799721835883171, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6973137855529785, "kl": 0.048510912759229544, "learning_rate": 4.0965497294570736e-07, "loss": 0.0019405171275138854, "reward": 3.129045104980469, "reward_std": 0.4232663154602051, "rewards/IngredientFormatReward/mean": 0.9909635543823242, "rewards/IngredientFormatReward/std": 0.07800297439098358, "rewards/IngredientMatchReward/mean": 0.65011967420578, "rewards/IngredientMatchReward/std": 0.2852735579013824, "rewards/IngredientQuantityMatchReward/mean": 0.6926494479179383, "rewards/IngredientQuantityMatchReward/std": 0.39057775139808654, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.4028699815273285, "step": 2085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 509.0, "completions/mean_length": 389.546875, "completions/min_length": 274.8, "epoch": 0.5813630041724618, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6442022919654846, "kl": 0.04921837141737342, "learning_rate": 4.073939437349636e-07, "loss": 0.001969057321548462, "reward": 2.978153944015503, "reward_std": 0.43001670241355894, "rewards/IngredientFormatReward/mean": 0.9893489599227905, "rewards/IngredientFormatReward/std": 0.08917459473013878, "rewards/IngredientMatchReward/mean": 0.5978354513645172, "rewards/IngredientMatchReward/std": 0.27598640620708464, "rewards/IngredientQuantityMatchReward/mean": 0.6284695625305176, "rewards/IngredientQuantityMatchReward/std": 0.4286845028400421, "rewards/TotalKcalExactMatchReward/mean": 0.7625, "rewards/TotalKcalExactMatchReward/std": 0.42297797799110415, "step": 2090 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 505.2, "completions/mean_length": 390.3421875, "completions/min_length": 277.2, "epoch": 0.5827538247566064, "frac_reward_zero_std": 0.075, "grad_norm": 0.627018392086029, "kl": 0.05088339238427579, "learning_rate": 4.0513487380627983e-07, "loss": 0.0020353712141513826, "reward": 3.052702856063843, "reward_std": 0.4378420948982239, "rewards/IngredientFormatReward/mean": 0.9756064057350159, "rewards/IngredientFormatReward/std": 0.13277253992855548, "rewards/IngredientMatchReward/mean": 0.6264713406562805, "rewards/IngredientMatchReward/std": 0.2919108152389526, "rewards/IngredientQuantityMatchReward/mean": 0.6475001513957978, "rewards/IngredientQuantityMatchReward/std": 0.40529059767723086, "rewards/TotalKcalExactMatchReward/mean": 0.803125, "rewards/TotalKcalExactMatchReward/std": 0.38602237701416015, "step": 2095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 502.0, "completions/mean_length": 385.890625, "completions/min_length": 241.6, "epoch": 0.5841446453407511, "frac_reward_zero_std": 0.1, "grad_norm": 0.6189238429069519, "kl": 0.050543641904369, "learning_rate": 4.028778109551826e-07, "loss": 0.002021963521838188, "reward": 3.0809757709503174, "reward_std": 0.34558130502700807, "rewards/IngredientFormatReward/mean": 0.993359375, "rewards/IngredientFormatReward/std": 0.06467613540589809, "rewards/IngredientMatchReward/mean": 0.6460584044456482, "rewards/IngredientMatchReward/std": 0.29698742628097535, "rewards/IngredientQuantityMatchReward/mean": 0.6228079795837402, "rewards/IngredientQuantityMatchReward/std": 0.423473596572876, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.3815928041934967, "step": 2100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 499.0, "completions/mean_length": 382.5234375, "completions/min_length": 252.6, "epoch": 0.5855354659248957, "frac_reward_zero_std": 0.1, "grad_norm": 0.6478217244148254, "kl": 0.049496779195033015, "learning_rate": 4.006228029347339e-07, "loss": 0.0019800461828708648, "reward": 3.1182824611663817, "reward_std": 0.32764893770217896, "rewards/IngredientFormatReward/mean": 0.9918229103088378, "rewards/IngredientFormatReward/std": 0.06296356171369552, "rewards/IngredientMatchReward/mean": 0.6453193426132202, "rewards/IngredientMatchReward/std": 0.3042366325855255, "rewards/IngredientQuantityMatchReward/mean": 0.6686402320861816, "rewards/IngredientQuantityMatchReward/std": 0.4119422197341919, "rewards/TotalKcalExactMatchReward/mean": 0.8125, "rewards/TotalKcalExactMatchReward/std": 0.3838110685348511, "step": 2105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 513.0, "completions/mean_length": 390.55625, "completions/min_length": 255.6, "epoch": 0.5869262865090403, "frac_reward_zero_std": 0.1, "grad_norm": 0.6478376984596252, "kl": 0.048133849655278026, "learning_rate": 3.983698974545215e-07, "loss": 0.0019253019243478775, "reward": 3.0814141750335695, "reward_std": 0.41355874538421633, "rewards/IngredientFormatReward/mean": 0.9811160683631897, "rewards/IngredientFormatReward/std": 0.12442523241043091, "rewards/IngredientMatchReward/mean": 0.675525176525116, "rewards/IngredientMatchReward/std": 0.2912832021713257, "rewards/IngredientQuantityMatchReward/mean": 0.6716479063034058, "rewards/IngredientQuantityMatchReward/std": 0.39944403171539306, "rewards/TotalKcalExactMatchReward/mean": 0.753125, "rewards/TotalKcalExactMatchReward/std": 0.41150047183036803, "step": 2110 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 501.2, "completions/mean_length": 390.93125, "completions/min_length": 275.2, "epoch": 0.588317107093185, "frac_reward_zero_std": 0.05, "grad_norm": 0.6301591396331787, "kl": 0.05105856913141906, "learning_rate": 3.9611914217964916e-07, "loss": 0.0020423345267772675, "reward": 3.1168880462646484, "reward_std": 0.3775189995765686, "rewards/IngredientFormatReward/mean": 0.9966145753860474, "rewards/IngredientFormatReward/std": 0.035576280951499936, "rewards/IngredientMatchReward/mean": 0.6276953220367432, "rewards/IngredientMatchReward/std": 0.2857614278793335, "rewards/IngredientQuantityMatchReward/mean": 0.691015648841858, "rewards/IngredientQuantityMatchReward/std": 0.4060320556163788, "rewards/TotalKcalExactMatchReward/mean": 0.8015625, "rewards/TotalKcalExactMatchReward/std": 0.3695449620485306, "step": 2115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 511.4, "completions/mean_length": 393.4765625, "completions/min_length": 262.2, "epoch": 0.5897079276773296, "frac_reward_zero_std": 0.075, "grad_norm": 0.6236531734466553, "kl": 0.0492264065425843, "learning_rate": 3.9387058472972843e-07, "loss": 0.001969209872186184, "reward": 3.0744152545928953, "reward_std": 0.39720383286476135, "rewards/IngredientFormatReward/mean": 0.9910416722297668, "rewards/IngredientFormatReward/std": 0.06891115196049213, "rewards/IngredientMatchReward/mean": 0.6055537104606629, "rewards/IngredientMatchReward/std": 0.30302745699882505, "rewards/IngredientQuantityMatchReward/mean": 0.6168824195861816, "rewards/IngredientQuantityMatchReward/std": 0.4277641594409943, "rewards/TotalKcalExactMatchReward/mean": 0.8609375, "rewards/TotalKcalExactMatchReward/std": 0.3140183940529823, "step": 2120 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 501.6, "completions/mean_length": 385.7359375, "completions/min_length": 263.2, "epoch": 0.5910987482614742, "frac_reward_zero_std": 0.05, "grad_norm": 0.6744566559791565, "kl": 0.04763327166438103, "learning_rate": 3.916242726778712e-07, "loss": 0.0019053336232900619, "reward": 3.2508696556091308, "reward_std": 0.37193950414657595, "rewards/IngredientFormatReward/mean": 0.9950520753860473, "rewards/IngredientFormatReward/std": 0.0532539501786232, "rewards/IngredientMatchReward/mean": 0.7194382548332214, "rewards/IngredientMatchReward/std": 0.2809447020292282, "rewards/IngredientQuantityMatchReward/mean": 0.7191918611526489, "rewards/IngredientQuantityMatchReward/std": 0.39092591404914856, "rewards/TotalKcalExactMatchReward/mean": 0.8171875, "rewards/TotalKcalExactMatchReward/std": 0.3763358473777771, "step": 2125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 506.8, "completions/mean_length": 389.2796875, "completions/min_length": 261.0, "epoch": 0.5924895688456189, "frac_reward_zero_std": 0.075, "grad_norm": 0.6765151619911194, "kl": 0.04687405494041741, "learning_rate": 3.893802535496834e-07, "loss": 0.0018747389316558839, "reward": 3.061327362060547, "reward_std": 0.3514033019542694, "rewards/IngredientFormatReward/mean": 0.9900520801544189, "rewards/IngredientFormatReward/std": 0.08974247500300407, "rewards/IngredientMatchReward/mean": 0.5992509841918945, "rewards/IngredientMatchReward/std": 0.31761006712913514, "rewards/IngredientQuantityMatchReward/mean": 0.6188993215560913, "rewards/IngredientQuantityMatchReward/std": 0.41553970575332644, "rewards/TotalKcalExactMatchReward/mean": 0.853125, "rewards/TotalKcalExactMatchReward/std": 0.34410153329372406, "step": 2130 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 510.2, "completions/mean_length": 390.790625, "completions/min_length": 295.4, "epoch": 0.5938803894297635, "frac_reward_zero_std": 0.1375, "grad_norm": 0.5965428352355957, "kl": 0.05046862023882568, "learning_rate": 3.871385748222583e-07, "loss": 0.0020183827728033066, "reward": 3.207280683517456, "reward_std": 0.32688103020191195, "rewards/IngredientFormatReward/mean": 0.9804427146911621, "rewards/IngredientFormatReward/std": 0.11501296181231738, "rewards/IngredientMatchReward/mean": 0.6385406732559205, "rewards/IngredientMatchReward/std": 0.2917583525180817, "rewards/IngredientQuantityMatchReward/mean": 0.7101723670959472, "rewards/IngredientQuantityMatchReward/std": 0.3938800930976868, "rewards/TotalKcalExactMatchReward/mean": 0.878125, "rewards/TotalKcalExactMatchReward/std": 0.3060554936528206, "step": 2135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 508.6, "completions/mean_length": 390.0828125, "completions/min_length": 251.4, "epoch": 0.5952712100139083, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6337137222290039, "kl": 0.05112361330538988, "learning_rate": 3.848992839231737e-07, "loss": 0.0020450735464692116, "reward": 3.1612887382507324, "reward_std": 0.41824568510055543, "rewards/IngredientFormatReward/mean": 0.9850520730018616, "rewards/IngredientFormatReward/std": 0.09320681095123291, "rewards/IngredientMatchReward/mean": 0.6647279262542725, "rewards/IngredientMatchReward/std": 0.27302170395851133, "rewards/IngredientQuantityMatchReward/mean": 0.6865087270736694, "rewards/IngredientQuantityMatchReward/std": 0.3848438024520874, "rewards/TotalKcalExactMatchReward/mean": 0.825, "rewards/TotalKcalExactMatchReward/std": 0.35974394381046293, "step": 2140 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 507.4, "completions/mean_length": 388.5140625, "completions/min_length": 274.8, "epoch": 0.5966620305980529, "frac_reward_zero_std": 0.125, "grad_norm": 0.556304395198822, "kl": 0.050310481898486616, "learning_rate": 3.826624282294875e-07, "loss": 0.002012613415718079, "reward": 3.023041009902954, "reward_std": 0.3705463230609894, "rewards/IngredientFormatReward/mean": 0.9921168088912964, "rewards/IngredientFormatReward/std": 0.06639720574021339, "rewards/IngredientMatchReward/mean": 0.6476860284805298, "rewards/IngredientMatchReward/std": 0.27975512146949766, "rewards/IngredientQuantityMatchReward/mean": 0.6629256725311279, "rewards/IngredientQuantityMatchReward/std": 0.3901920020580292, "rewards/TotalKcalExactMatchReward/mean": 0.7203125, "rewards/TotalKcalExactMatchReward/std": 0.4402440369129181, "step": 2145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 505.0, "completions/mean_length": 385.0921875, "completions/min_length": 243.8, "epoch": 0.5980528511821975, "frac_reward_zero_std": 0.0625, "grad_norm": 0.597395122051239, "kl": 0.05190782949794084, "learning_rate": 3.804280550667356e-07, "loss": 0.0020766373723745345, "reward": 3.165365695953369, "reward_std": 0.35970969796180724, "rewards/IngredientFormatReward/mean": 0.9876413464546203, "rewards/IngredientFormatReward/std": 0.09138427898287774, "rewards/IngredientMatchReward/mean": 0.6722779870033264, "rewards/IngredientMatchReward/std": 0.266836553812027, "rewards/IngredientQuantityMatchReward/mean": 0.6866963863372803, "rewards/IngredientQuantityMatchReward/std": 0.3877449631690979, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.37181003093719484, "step": 2150 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 499.4, "completions/mean_length": 389.1734375, "completions/min_length": 274.8, "epoch": 0.5994436717663422, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6142936944961548, "kl": 0.049102706369012594, "learning_rate": 3.7819621170793034e-07, "loss": 0.0019642310217022896, "reward": 3.1915061473846436, "reward_std": 0.3775101125240326, "rewards/IngredientFormatReward/mean": 0.9940624952316284, "rewards/IngredientFormatReward/std": 0.04380592703819275, "rewards/IngredientMatchReward/mean": 0.6725663423538208, "rewards/IngredientMatchReward/std": 0.2846332907676697, "rewards/IngredientQuantityMatchReward/mean": 0.6967523574829102, "rewards/IngredientQuantityMatchReward/std": 0.4011437356472015, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.3704190969467163, "step": 2155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 507.8, "completions/mean_length": 387.859375, "completions/min_length": 271.0, "epoch": 0.6008344923504868, "frac_reward_zero_std": 0.05, "grad_norm": 0.601314902305603, "kl": 0.050125482678413394, "learning_rate": 3.759669453725607e-07, "loss": 0.0020050119608640673, "reward": 3.091590166091919, "reward_std": 0.42018883228302, "rewards/IngredientFormatReward/mean": 0.9830989599227905, "rewards/IngredientFormatReward/std": 0.10198934972286225, "rewards/IngredientMatchReward/mean": 0.6450570344924926, "rewards/IngredientMatchReward/std": 0.2986615300178528, "rewards/IngredientQuantityMatchReward/mean": 0.6493716359138488, "rewards/IngredientQuantityMatchReward/std": 0.4103902757167816, "rewards/TotalKcalExactMatchReward/mean": 0.8140625, "rewards/TotalKcalExactMatchReward/std": 0.37561099529266356, "step": 2160 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 497.6, "completions/mean_length": 387.8203125, "completions/min_length": 259.8, "epoch": 0.6022253129346314, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6496283411979675, "kl": 0.04889493566006422, "learning_rate": 3.7374030322559314e-07, "loss": 0.001956149935722351, "reward": 3.112612819671631, "reward_std": 0.3562726080417633, "rewards/IngredientFormatReward/mean": 0.9957849740982055, "rewards/IngredientFormatReward/std": 0.03390633724629879, "rewards/IngredientMatchReward/mean": 0.6423406958580017, "rewards/IngredientMatchReward/std": 0.2840432107448578, "rewards/IngredientQuantityMatchReward/mean": 0.6744871377944947, "rewards/IngredientQuantityMatchReward/std": 0.39336987137794494, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.3893772900104523, "step": 2165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 509.4, "completions/mean_length": 392.259375, "completions/min_length": 272.0, "epoch": 0.6036161335187761, "frac_reward_zero_std": 0.025, "grad_norm": 0.6597304344177246, "kl": 1.401045136200264, "learning_rate": 3.71516332376474e-07, "loss": 0.055953198671340944, "reward": 3.019261026382446, "reward_std": 0.4189970850944519, "rewards/IngredientFormatReward/mean": 0.977146577835083, "rewards/IngredientFormatReward/std": 0.11853427737951279, "rewards/IngredientMatchReward/mean": 0.6084200739860535, "rewards/IngredientMatchReward/std": 0.29643736481666566, "rewards/IngredientQuantityMatchReward/mean": 0.6790068507194519, "rewards/IngredientQuantityMatchReward/std": 0.40717952251434325, "rewards/TotalKcalExactMatchReward/mean": 0.7546875, "rewards/TotalKcalExactMatchReward/std": 0.4276392340660095, "step": 2170 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 513.0, "completions/mean_length": 391.103125, "completions/min_length": 248.4, "epoch": 0.6050069541029207, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6310843229293823, "kl": 0.04990187431685626, "learning_rate": 3.692950798781318e-07, "loss": 0.001995875313878059, "reward": 3.0463210582733153, "reward_std": 0.39750961065292356, "rewards/IngredientFormatReward/mean": 0.9845368385314941, "rewards/IngredientFormatReward/std": 0.11547477096319199, "rewards/IngredientMatchReward/mean": 0.6253199338912964, "rewards/IngredientMatchReward/std": 0.2806160271167755, "rewards/IngredientQuantityMatchReward/mean": 0.6724017977714538, "rewards/IngredientQuantityMatchReward/std": 0.40632015466690063, "rewards/TotalKcalExactMatchReward/mean": 0.7640625, "rewards/TotalKcalExactMatchReward/std": 0.4197130024433136, "step": 2175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 510.2, "completions/mean_length": 388.1296875, "completions/min_length": 266.0, "epoch": 0.6063977746870653, "frac_reward_zero_std": 0.05, "grad_norm": 0.706554651260376, "kl": 0.052620567847043276, "learning_rate": 3.670765927259831e-07, "loss": 0.0021052807569503784, "reward": 3.153690004348755, "reward_std": 0.41479485630989077, "rewards/IngredientFormatReward/mean": 0.9915085434913635, "rewards/IngredientFormatReward/std": 0.06617667004466057, "rewards/IngredientMatchReward/mean": 0.6520244598388671, "rewards/IngredientMatchReward/std": 0.2850645899772644, "rewards/IngredientQuantityMatchReward/mean": 0.7070319652557373, "rewards/IngredientQuantityMatchReward/std": 0.4136449575424194, "rewards/TotalKcalExactMatchReward/mean": 0.803125, "rewards/TotalKcalExactMatchReward/std": 0.3873530626296997, "step": 2180 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 510.6, "completions/mean_length": 397.1109375, "completions/min_length": 283.8, "epoch": 0.60778859527121, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6182182431221008, "kl": 0.05071941772475839, "learning_rate": 3.648609178569374e-07, "loss": 0.0020290974527597426, "reward": 3.0801242351531983, "reward_std": 0.4037054359912872, "rewards/IngredientFormatReward/mean": 0.9916145920753479, "rewards/IngredientFormatReward/std": 0.06768649965524673, "rewards/IngredientMatchReward/mean": 0.6478013277053833, "rewards/IngredientMatchReward/std": 0.29819480180740354, "rewards/IngredientQuantityMatchReward/mean": 0.684458315372467, "rewards/IngredientQuantityMatchReward/std": 0.38002619743347166, "rewards/TotalKcalExactMatchReward/mean": 0.75625, "rewards/TotalKcalExactMatchReward/std": 0.41520002484321594, "step": 2185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 498.8, "completions/mean_length": 386.0421875, "completions/min_length": 272.4, "epoch": 0.6091794158553546, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6383892893791199, "kl": 0.04978294498287141, "learning_rate": 3.626481021484045e-07, "loss": 0.0019914690405130387, "reward": 3.1687289237976075, "reward_std": 0.34659733176231383, "rewards/IngredientFormatReward/mean": 0.9972395896911621, "rewards/IngredientFormatReward/std": 0.02697129435837269, "rewards/IngredientMatchReward/mean": 0.6780971050262451, "rewards/IngredientMatchReward/std": 0.2776097983121872, "rewards/IngredientQuantityMatchReward/mean": 0.710579776763916, "rewards/IngredientQuantityMatchReward/std": 0.3856403946876526, "rewards/TotalKcalExactMatchReward/mean": 0.7828125, "rewards/TotalKcalExactMatchReward/std": 0.4049029767513275, "step": 2190 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 502.8, "completions/mean_length": 390.121875, "completions/min_length": 266.0, "epoch": 0.6105702364394993, "frac_reward_zero_std": 0.0875, "grad_norm": 0.650026798248291, "kl": 0.05147871123626828, "learning_rate": 3.6043819241730145e-07, "loss": 0.0020590342581272127, "reward": 3.158277082443237, "reward_std": 0.3794547736644745, "rewards/IngredientFormatReward/mean": 0.9897916555404663, "rewards/IngredientFormatReward/std": 0.0673672953620553, "rewards/IngredientMatchReward/mean": 0.6765773892402649, "rewards/IngredientMatchReward/std": 0.280401399731636, "rewards/IngredientQuantityMatchReward/mean": 0.660658085346222, "rewards/IngredientQuantityMatchReward/std": 0.406155651807785, "rewards/TotalKcalExactMatchReward/mean": 0.83125, "rewards/TotalKcalExactMatchReward/std": 0.3482891798019409, "step": 2195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 508.4, "completions/mean_length": 392.6640625, "completions/min_length": 270.2, "epoch": 0.6119610570236439, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6922538876533508, "kl": 0.051177499582991004, "learning_rate": 3.5823123541906424e-07, "loss": 0.00204671174287796, "reward": 3.038310670852661, "reward_std": 0.4538100779056549, "rewards/IngredientFormatReward/mean": 0.9873809576034546, "rewards/IngredientFormatReward/std": 0.10423120856285095, "rewards/IngredientMatchReward/mean": 0.6075683832168579, "rewards/IngredientMatchReward/std": 0.2966679573059082, "rewards/IngredientQuantityMatchReward/mean": 0.6402364134788513, "rewards/IngredientQuantityMatchReward/std": 0.4361319363117218, "rewards/TotalKcalExactMatchReward/mean": 0.803125, "rewards/TotalKcalExactMatchReward/std": 0.39142870903015137, "step": 2200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 507.6, "completions/mean_length": 394.4796875, "completions/min_length": 276.0, "epoch": 0.6133518776077886, "frac_reward_zero_std": 0.0, "grad_norm": 0.6373947262763977, "kl": 0.04903403371572494, "learning_rate": 3.560272778466569e-07, "loss": 0.001961347088217735, "reward": 3.1263299942016602, "reward_std": 0.38947783708572387, "rewards/IngredientFormatReward/mean": 0.9933854103088379, "rewards/IngredientFormatReward/std": 0.04924058765172958, "rewards/IngredientMatchReward/mean": 0.6449187755584717, "rewards/IngredientMatchReward/std": 0.27405828833580015, "rewards/IngredientQuantityMatchReward/mean": 0.6349007368087769, "rewards/IngredientQuantityMatchReward/std": 0.41207821369171144, "rewards/TotalKcalExactMatchReward/mean": 0.853125, "rewards/TotalKcalExactMatchReward/std": 0.35279104113578796, "step": 2205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 508.2, "completions/mean_length": 389.8140625, "completions/min_length": 267.2, "epoch": 0.6147426981919333, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6941333413124084, "kl": 0.20643644286319612, "learning_rate": 3.5382636632958405e-07, "loss": 0.008257143199443817, "reward": 3.016005277633667, "reward_std": 0.4019562005996704, "rewards/IngredientFormatReward/mean": 0.9874702215194702, "rewards/IngredientFormatReward/std": 0.08696230705827475, "rewards/IngredientMatchReward/mean": 0.6278738975524902, "rewards/IngredientMatchReward/std": 0.293187814950943, "rewards/IngredientQuantityMatchReward/mean": 0.6319111466407776, "rewards/IngredientQuantityMatchReward/std": 0.43076640367507935, "rewards/TotalKcalExactMatchReward/mean": 0.76875, "rewards/TotalKcalExactMatchReward/std": 0.41989484429359436, "step": 2210 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 508.6, "completions/mean_length": 393.309375, "completions/min_length": 265.8, "epoch": 0.6161335187760779, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6270684003829956, "kl": 0.04865258978679776, "learning_rate": 3.516285474329044e-07, "loss": 0.0019460489973425865, "reward": 3.1262417793273927, "reward_std": 0.3906321346759796, "rewards/IngredientFormatReward/mean": 0.99453125, "rewards/IngredientFormatReward/std": 0.05367666333913803, "rewards/IngredientMatchReward/mean": 0.6526476263999939, "rewards/IngredientMatchReward/std": 0.27952222228050233, "rewards/IngredientQuantityMatchReward/mean": 0.6915629744529724, "rewards/IngredientQuantityMatchReward/std": 0.400864040851593, "rewards/TotalKcalExactMatchReward/mean": 0.7875, "rewards/TotalKcalExactMatchReward/std": 0.3981053829193115, "step": 2215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 509.8, "completions/mean_length": 395.0640625, "completions/min_length": 248.2, "epoch": 0.6175243393602226, "frac_reward_zero_std": 0.125, "grad_norm": 0.5712026953697205, "kl": 0.05006823530420661, "learning_rate": 3.4943386765624563e-07, "loss": 0.002002709172666073, "reward": 3.077920436859131, "reward_std": 0.392623233795166, "rewards/IngredientFormatReward/mean": 0.9903906106948852, "rewards/IngredientFormatReward/std": 0.07571376450359821, "rewards/IngredientMatchReward/mean": 0.647243320941925, "rewards/IngredientMatchReward/std": 0.284138685464859, "rewards/IngredientQuantityMatchReward/mean": 0.6262240290641785, "rewards/IngredientQuantityMatchReward/std": 0.4356621503829956, "rewards/TotalKcalExactMatchReward/mean": 0.8140625, "rewards/TotalKcalExactMatchReward/std": 0.3902096688747406, "step": 2220 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 497.8, "completions/mean_length": 387.2, "completions/min_length": 234.8, "epoch": 0.6189151599443672, "frac_reward_zero_std": 0.1125, "grad_norm": 0.6324188113212585, "kl": 0.046689180727116766, "learning_rate": 3.472423734328204e-07, "loss": 0.001867634803056717, "reward": 3.09603590965271, "reward_std": 0.35402976274490355, "rewards/IngredientFormatReward/mean": 0.9944270849227905, "rewards/IngredientFormatReward/std": 0.05454032719135284, "rewards/IngredientMatchReward/mean": 0.6432204723358155, "rewards/IngredientMatchReward/std": 0.28954875469207764, "rewards/IngredientQuantityMatchReward/mean": 0.6849508166313172, "rewards/IngredientQuantityMatchReward/std": 0.3964232563972473, "rewards/TotalKcalExactMatchReward/mean": 0.7734375, "rewards/TotalKcalExactMatchReward/std": 0.4028464019298553, "step": 2225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 506.8, "completions/mean_length": 384.578125, "completions/min_length": 252.4, "epoch": 0.6203059805285118, "frac_reward_zero_std": 0.075, "grad_norm": 0.6558968424797058, "kl": 0.053798887552693485, "learning_rate": 3.450541111284447e-07, "loss": 0.0021519482135772703, "reward": 3.114434099197388, "reward_std": 0.43681026697158815, "rewards/IngredientFormatReward/mean": 0.9728868842124939, "rewards/IngredientFormatReward/std": 0.12590492386370897, "rewards/IngredientMatchReward/mean": 0.6601122260093689, "rewards/IngredientMatchReward/std": 0.28873300552368164, "rewards/IngredientQuantityMatchReward/mean": 0.7111225247383117, "rewards/IngredientQuantityMatchReward/std": 0.41141672134399415, "rewards/TotalKcalExactMatchReward/mean": 0.7703125, "rewards/TotalKcalExactMatchReward/std": 0.38224449157714846, "step": 2230 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 513.0, "completions/mean_length": 391.4203125, "completions/min_length": 262.8, "epoch": 0.6216968011126565, "frac_reward_zero_std": 0.05, "grad_norm": 0.6426628828048706, "kl": 0.0496185167459771, "learning_rate": 3.4286912704055505e-07, "loss": 0.00198509581387043, "reward": 3.095783233642578, "reward_std": 0.39932904839515687, "rewards/IngredientFormatReward/mean": 0.9864955544471741, "rewards/IngredientFormatReward/std": 0.10071748532354832, "rewards/IngredientMatchReward/mean": 0.6558928847312927, "rewards/IngredientMatchReward/std": 0.2895587205886841, "rewards/IngredientQuantityMatchReward/mean": 0.6299573540687561, "rewards/IngredientQuantityMatchReward/std": 0.4130510151386261, "rewards/TotalKcalExactMatchReward/mean": 0.8234375, "rewards/TotalKcalExactMatchReward/std": 0.3680810809135437, "step": 2235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 499.0, "completions/mean_length": 393.0328125, "completions/min_length": 260.4, "epoch": 0.6230876216968011, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6363694071769714, "kl": 0.04835720381233841, "learning_rate": 3.406874673972312e-07, "loss": 0.0019343402236700058, "reward": 3.0554323196411133, "reward_std": 0.39848679304122925, "rewards/IngredientFormatReward/mean": 0.9841145753860474, "rewards/IngredientFormatReward/std": 0.08837253898382187, "rewards/IngredientMatchReward/mean": 0.6605754017829895, "rewards/IngredientMatchReward/std": 0.2916788816452026, "rewards/IngredientQuantityMatchReward/mean": 0.6326173186302185, "rewards/IngredientQuantityMatchReward/std": 0.40582574605941774, "rewards/TotalKcalExactMatchReward/mean": 0.778125, "rewards/TotalKcalExactMatchReward/std": 0.3950036108493805, "step": 2240 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 506.2, "completions/mean_length": 394.8921875, "completions/min_length": 286.6, "epoch": 0.6244784422809457, "frac_reward_zero_std": 0.075, "grad_norm": 0.5934588313102722, "kl": 0.05474424082785845, "learning_rate": 3.385091783562168e-07, "loss": 0.002190130949020386, "reward": 3.085518455505371, "reward_std": 0.3905796468257904, "rewards/IngredientFormatReward/mean": 0.9823958396911621, "rewards/IngredientFormatReward/std": 0.11776968874037266, "rewards/IngredientMatchReward/mean": 0.6468719124794007, "rewards/IngredientMatchReward/std": 0.29382801055908203, "rewards/IngredientQuantityMatchReward/mean": 0.6828132152557373, "rewards/IngredientQuantityMatchReward/std": 0.4147923529148102, "rewards/TotalKcalExactMatchReward/mean": 0.7734375, "rewards/TotalKcalExactMatchReward/std": 0.4130185544490814, "step": 2245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 510.2, "completions/mean_length": 391.8046875, "completions/min_length": 259.8, "epoch": 0.6258692628650904, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7126057744026184, "kl": 0.051664074324071406, "learning_rate": 3.363343060039431e-07, "loss": 0.0020667586475610735, "reward": 2.926431131362915, "reward_std": 0.4142996668815613, "rewards/IngredientFormatReward/mean": 0.9811979055404663, "rewards/IngredientFormatReward/std": 0.12898623794317246, "rewards/IngredientMatchReward/mean": 0.621659231185913, "rewards/IngredientMatchReward/std": 0.29404624104499816, "rewards/IngredientQuantityMatchReward/mean": 0.5407614946365357, "rewards/IngredientQuantityMatchReward/std": 0.43060929179191587, "rewards/TotalKcalExactMatchReward/mean": 0.7828125, "rewards/TotalKcalExactMatchReward/std": 0.3951663553714752, "step": 2250 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 509.2, "completions/mean_length": 386.009375, "completions/min_length": 257.4, "epoch": 0.627260083449235, "frac_reward_zero_std": 0.1125, "grad_norm": 0.6304970383644104, "kl": 0.049507897021248935, "learning_rate": 3.3416289635455364e-07, "loss": 0.001980157569050789, "reward": 3.180800724029541, "reward_std": 0.40125504732131956, "rewards/IngredientFormatReward/mean": 0.9903125047683716, "rewards/IngredientFormatReward/std": 0.08373235017061234, "rewards/IngredientMatchReward/mean": 0.6888733983039856, "rewards/IngredientMatchReward/std": 0.2980964004993439, "rewards/IngredientQuantityMatchReward/mean": 0.6859898090362548, "rewards/IngredientQuantityMatchReward/std": 0.4196448683738708, "rewards/TotalKcalExactMatchReward/mean": 0.815625, "rewards/TotalKcalExactMatchReward/std": 0.3790489435195923, "step": 2255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 493.0, "completions/mean_length": 389.01875, "completions/min_length": 274.0, "epoch": 0.6286509040333796, "frac_reward_zero_std": 0.1, "grad_norm": 0.5745730996131897, "kl": 0.052322480361908674, "learning_rate": 3.3199499534893126e-07, "loss": 0.0020928554236888885, "reward": 3.0668933391571045, "reward_std": 0.37459617853164673, "rewards/IngredientFormatReward/mean": 0.9922395944595337, "rewards/IngredientFormatReward/std": 0.054567881673574445, "rewards/IngredientMatchReward/mean": 0.672517991065979, "rewards/IngredientMatchReward/std": 0.2672812879085541, "rewards/IngredientQuantityMatchReward/mean": 0.6583857655525207, "rewards/IngredientQuantityMatchReward/std": 0.4205239474773407, "rewards/TotalKcalExactMatchReward/mean": 0.74375, "rewards/TotalKcalExactMatchReward/std": 0.43603759407997134, "step": 2260 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 500.0, "completions/mean_length": 394.8109375, "completions/min_length": 271.0, "epoch": 0.6300417246175244, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6598750352859497, "kl": 0.06744713457301259, "learning_rate": 3.298306488537257e-07, "loss": 0.0026970580220222473, "reward": 3.1390488147735596, "reward_std": 0.36806185245513917, "rewards/IngredientFormatReward/mean": 0.9928125143051147, "rewards/IngredientFormatReward/std": 0.06443778797984123, "rewards/IngredientMatchReward/mean": 0.6801209092140198, "rewards/IngredientMatchReward/std": 0.26344968378543854, "rewards/IngredientQuantityMatchReward/mean": 0.659865391254425, "rewards/IngredientQuantityMatchReward/std": 0.40859387516975404, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.3911212682723999, "step": 2265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 501.2, "completions/mean_length": 391.24375, "completions/min_length": 270.8, "epoch": 0.631432545201669, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7604984641075134, "kl": 0.06415583626367152, "learning_rate": 3.276699026603835e-07, "loss": 0.00256730355322361, "reward": 3.187400960922241, "reward_std": 0.38949325680732727, "rewards/IngredientFormatReward/mean": 0.9868749976158142, "rewards/IngredientFormatReward/std": 0.09232946932315826, "rewards/IngredientMatchReward/mean": 0.6554042816162109, "rewards/IngredientMatchReward/std": 0.2981587827205658, "rewards/IngredientQuantityMatchReward/mean": 0.6841842651367187, "rewards/IngredientQuantityMatchReward/std": 0.3940977334976196, "rewards/TotalKcalExactMatchReward/mean": 0.8609375, "rewards/TotalKcalExactMatchReward/std": 0.3312438249588013, "step": 2270 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 506.4, "completions/mean_length": 397.3140625, "completions/min_length": 277.4, "epoch": 0.6328233657858137, "frac_reward_zero_std": 0.05, "grad_norm": 0.6403272747993469, "kl": 0.0502887899056077, "learning_rate": 3.2551280248417855e-07, "loss": 0.0020116424188017846, "reward": 3.098349761962891, "reward_std": 0.4067712306976318, "rewards/IngredientFormatReward/mean": 0.9885267853736878, "rewards/IngredientFormatReward/std": 0.09110912680625916, "rewards/IngredientMatchReward/mean": 0.6192803025245667, "rewards/IngredientMatchReward/std": 0.28211758434772494, "rewards/IngredientQuantityMatchReward/mean": 0.6592926383018494, "rewards/IngredientQuantityMatchReward/std": 0.4010209202766418, "rewards/TotalKcalExactMatchReward/mean": 0.83125, "rewards/TotalKcalExactMatchReward/std": 0.3651015400886536, "step": 2275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 508.2, "completions/mean_length": 395.259375, "completions/min_length": 273.6, "epoch": 0.6342141863699583, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6547976732254028, "kl": 0.04664964359253645, "learning_rate": 3.2335939396324574e-07, "loss": 0.001865985430777073, "reward": 3.156012487411499, "reward_std": 0.36845679879188536, "rewards/IngredientFormatReward/mean": 0.9910937547683716, "rewards/IngredientFormatReward/std": 0.07213107720017434, "rewards/IngredientMatchReward/mean": 0.673369300365448, "rewards/IngredientMatchReward/std": 0.29961636662483215, "rewards/IngredientQuantityMatchReward/mean": 0.6462370038032532, "rewards/IngredientQuantityMatchReward/std": 0.4035289347171783, "rewards/TotalKcalExactMatchReward/mean": 0.8453125, "rewards/TotalKcalExactMatchReward/std": 0.33718748241662977, "step": 2280 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 510.4, "completions/mean_length": 402.1625, "completions/min_length": 282.4, "epoch": 0.6356050069541029, "frac_reward_zero_std": 0.1, "grad_norm": 0.6613708734512329, "kl": 0.04664005022495985, "learning_rate": 3.21209722657615e-07, "loss": 0.0018660373985767364, "reward": 3.0435181617736817, "reward_std": 0.3617978811264038, "rewards/IngredientFormatReward/mean": 0.9830654621124267, "rewards/IngredientFormatReward/std": 0.09950861930847169, "rewards/IngredientMatchReward/mean": 0.6392280459403992, "rewards/IngredientMatchReward/std": 0.2842461735010147, "rewards/IngredientQuantityMatchReward/mean": 0.6149746000766754, "rewards/IngredientQuantityMatchReward/std": 0.4098484456539154, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.384172785282135, "step": 2285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 504.2, "completions/mean_length": 388.5296875, "completions/min_length": 260.8, "epoch": 0.6369958275382476, "frac_reward_zero_std": 0.0625, "grad_norm": 0.5699143409729004, "kl": 0.04985083430074155, "learning_rate": 3.1906383404824735e-07, "loss": 0.001994183287024498, "reward": 3.0484458923339846, "reward_std": 0.372654390335083, "rewards/IngredientFormatReward/mean": 0.9962500095367431, "rewards/IngredientFormatReward/std": 0.03921364024281502, "rewards/IngredientMatchReward/mean": 0.6028466105461121, "rewards/IngredientMatchReward/std": 0.28491021394729615, "rewards/IngredientQuantityMatchReward/mean": 0.6618492722511291, "rewards/IngredientQuantityMatchReward/std": 0.41888644695281985, "rewards/TotalKcalExactMatchReward/mean": 0.7875, "rewards/TotalKcalExactMatchReward/std": 0.3779851108789444, "step": 2290 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 507.6, "completions/mean_length": 394.1171875, "completions/min_length": 288.0, "epoch": 0.6383866481223922, "frac_reward_zero_std": 0.075, "grad_norm": 0.5812567472457886, "kl": 0.048090280825272204, "learning_rate": 3.169217735360721e-07, "loss": 0.001923968270421028, "reward": 3.0279534339904783, "reward_std": 0.3708513677120209, "rewards/IngredientFormatReward/mean": 0.9885416507720948, "rewards/IngredientFormatReward/std": 0.09087646007537842, "rewards/IngredientMatchReward/mean": 0.6266480684280396, "rewards/IngredientMatchReward/std": 0.2966554284095764, "rewards/IngredientQuantityMatchReward/mean": 0.6346386492252349, "rewards/IngredientQuantityMatchReward/std": 0.41230267882347105, "rewards/TotalKcalExactMatchReward/mean": 0.778125, "rewards/TotalKcalExactMatchReward/std": 0.4041856527328491, "step": 2295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 499.6, "completions/mean_length": 385.1796875, "completions/min_length": 244.2, "epoch": 0.6397774687065368, "frac_reward_zero_std": 0.1, "grad_norm": 0.6119884252548218, "kl": 0.05169716072268784, "learning_rate": 3.147835864410276e-07, "loss": 0.002067873626947403, "reward": 3.1669593334197996, "reward_std": 0.4164513349533081, "rewards/IngredientFormatReward/mean": 0.985141372680664, "rewards/IngredientFormatReward/std": 0.10293094590306281, "rewards/IngredientMatchReward/mean": 0.6206455707550049, "rewards/IngredientMatchReward/std": 0.29413315653800964, "rewards/IngredientQuantityMatchReward/mean": 0.7252350091934204, "rewards/IngredientQuantityMatchReward/std": 0.39733303189277647, "rewards/TotalKcalExactMatchReward/mean": 0.8359375, "rewards/TotalKcalExactMatchReward/std": 0.35827775597572326, "step": 2300 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 510.4, "completions/mean_length": 392.31875, "completions/min_length": 252.2, "epoch": 0.6411682892906815, "frac_reward_zero_std": 0.075, "grad_norm": 0.6846728920936584, "kl": 0.04950613779947162, "learning_rate": 3.126493180011014e-07, "loss": 0.001980242505669594, "reward": 3.1633121967315674, "reward_std": 0.35973186790943146, "rewards/IngredientFormatReward/mean": 0.9850520730018616, "rewards/IngredientFormatReward/std": 0.10152259021997452, "rewards/IngredientMatchReward/mean": 0.6418793201446533, "rewards/IngredientMatchReward/std": 0.27253933548927306, "rewards/IngredientQuantityMatchReward/mean": 0.7113808035850525, "rewards/IngredientQuantityMatchReward/std": 0.40159483551979064, "rewards/TotalKcalExactMatchReward/mean": 0.825, "rewards/TotalKcalExactMatchReward/std": 0.36822456419467925, "step": 2305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 508.2, "completions/mean_length": 397.2546875, "completions/min_length": 294.4, "epoch": 0.6425591098748261, "frac_reward_zero_std": 0.1, "grad_norm": 0.6922708749771118, "kl": 0.053410388017073276, "learning_rate": 3.105190133713733e-07, "loss": 0.002136606350541115, "reward": 3.185721254348755, "reward_std": 0.43321176767349245, "rewards/IngredientFormatReward/mean": 0.9797656297683716, "rewards/IngredientFormatReward/std": 0.11258613169193268, "rewards/IngredientMatchReward/mean": 0.6650626420974731, "rewards/IngredientMatchReward/std": 0.28691638112068174, "rewards/IngredientQuantityMatchReward/mean": 0.7252679705619812, "rewards/IngredientQuantityMatchReward/std": 0.37057950496673586, "rewards/TotalKcalExactMatchReward/mean": 0.815625, "rewards/TotalKcalExactMatchReward/std": 0.38518795371055603, "step": 2310 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 510.6, "completions/mean_length": 394.528125, "completions/min_length": 274.8, "epoch": 0.6439499304589708, "frac_reward_zero_std": 0.05, "grad_norm": 0.6853731274604797, "kl": 0.04699210901744664, "learning_rate": 3.0839271762306005e-07, "loss": 0.00188013669103384, "reward": 3.0963015079498293, "reward_std": 0.37388708591461184, "rewards/IngredientFormatReward/mean": 0.985156238079071, "rewards/IngredientFormatReward/std": 0.09949494898319244, "rewards/IngredientMatchReward/mean": 0.6302870869636535, "rewards/IngredientMatchReward/std": 0.27550233602523805, "rewards/IngredientQuantityMatchReward/mean": 0.6964831590652466, "rewards/IngredientQuantityMatchReward/std": 0.4114985764026642, "rewards/TotalKcalExactMatchReward/mean": 0.784375, "rewards/TotalKcalExactMatchReward/std": 0.41028077006340025, "step": 2315 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 497.4, "completions/mean_length": 389.7984375, "completions/min_length": 265.0, "epoch": 0.6453407510431154, "frac_reward_zero_std": 0.025, "grad_norm": 0.6921623349189758, "kl": 0.046600739704445004, "learning_rate": 3.0627047574256216e-07, "loss": 0.0018642637878656387, "reward": 3.076311540603638, "reward_std": 0.3626805365085602, "rewards/IngredientFormatReward/mean": 0.996666669845581, "rewards/IngredientFormatReward/std": 0.02143365629017353, "rewards/IngredientMatchReward/mean": 0.6558581352233886, "rewards/IngredientMatchReward/std": 0.28329201638698576, "rewards/IngredientQuantityMatchReward/mean": 0.6050367593765259, "rewards/IngredientQuantityMatchReward/std": 0.4378706097602844, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.3791636288166046, "step": 2320 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 508.6, "completions/mean_length": 396.7640625, "completions/min_length": 269.4, "epoch": 0.6467315716272601, "frac_reward_zero_std": 0.0625, "grad_norm": 0.5751472115516663, "kl": 0.04943703303579241, "learning_rate": 3.041523326305112e-07, "loss": 0.001977463066577911, "reward": 3.194409990310669, "reward_std": 0.4305049300193787, "rewards/IngredientFormatReward/mean": 0.9797916531562805, "rewards/IngredientFormatReward/std": 0.12461420521140099, "rewards/IngredientMatchReward/mean": 0.6630220651626587, "rewards/IngredientMatchReward/std": 0.2750160127878189, "rewards/IngredientQuantityMatchReward/mean": 0.6890962600708008, "rewards/IngredientQuantityMatchReward/std": 0.385354483127594, "rewards/TotalKcalExactMatchReward/mean": 0.8625, "rewards/TotalKcalExactMatchReward/std": 0.3144735336303711, "step": 2325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 513.0, "completions/mean_length": 393.4203125, "completions/min_length": 266.4, "epoch": 0.6481223922114048, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6151979565620422, "kl": 0.047250632150098684, "learning_rate": 3.020383331008212e-07, "loss": 0.0018899183720350265, "reward": 2.952303075790405, "reward_std": 0.4416388809680939, "rewards/IngredientFormatReward/mean": 0.9795163750648499, "rewards/IngredientFormatReward/std": 0.13366409242153168, "rewards/IngredientMatchReward/mean": 0.6384114503860474, "rewards/IngredientMatchReward/std": 0.29739701747894287, "rewards/IngredientQuantityMatchReward/mean": 0.5906252264976501, "rewards/IngredientQuantityMatchReward/std": 0.4250226616859436, "rewards/TotalKcalExactMatchReward/mean": 0.74375, "rewards/TotalKcalExactMatchReward/std": 0.40911481976509095, "step": 2330 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 507.4, "completions/mean_length": 392.5203125, "completions/min_length": 262.2, "epoch": 0.6495132127955494, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6049847602844238, "kl": 0.04755361638963222, "learning_rate": 2.9992852187973874e-07, "loss": 0.001902458816766739, "reward": 3.0733123302459715, "reward_std": 0.3850603997707367, "rewards/IngredientFormatReward/mean": 0.9869791746139527, "rewards/IngredientFormatReward/std": 0.08390166461467743, "rewards/IngredientMatchReward/mean": 0.610430920124054, "rewards/IngredientMatchReward/std": 0.29726721048355104, "rewards/IngredientQuantityMatchReward/mean": 0.641527247428894, "rewards/IngredientQuantityMatchReward/std": 0.4174162566661835, "rewards/TotalKcalExactMatchReward/mean": 0.834375, "rewards/TotalKcalExactMatchReward/std": 0.3625717520713806, "step": 2335 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 509.8, "completions/mean_length": 391.6453125, "completions/min_length": 270.2, "epoch": 0.650904033379694, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7412999272346497, "kl": 0.05683932460378856, "learning_rate": 2.9782294360489824e-07, "loss": 0.0022735811769962313, "reward": 3.0667440414428713, "reward_std": 0.4293572545051575, "rewards/IngredientFormatReward/mean": 0.9919270753860474, "rewards/IngredientFormatReward/std": 0.07807534784078599, "rewards/IngredientMatchReward/mean": 0.6268408894538879, "rewards/IngredientMatchReward/std": 0.29498333036899566, "rewards/IngredientQuantityMatchReward/mean": 0.6745386362075806, "rewards/IngredientQuantityMatchReward/std": 0.4114294946193695, "rewards/TotalKcalExactMatchReward/mean": 0.7734375, "rewards/TotalKcalExactMatchReward/std": 0.4150561511516571, "step": 2340 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 505.8, "completions/mean_length": 394.3515625, "completions/min_length": 272.6, "epoch": 0.6522948539638387, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6499749422073364, "kl": 0.04816399768460542, "learning_rate": 2.957216428243772e-07, "loss": 0.001926695369184017, "reward": 2.9874334812164305, "reward_std": 0.44137861132621764, "rewards/IngredientFormatReward/mean": 0.9909077405929565, "rewards/IngredientFormatReward/std": 0.07433488909155131, "rewards/IngredientMatchReward/mean": 0.6615043878555298, "rewards/IngredientMatchReward/std": 0.2838159501552582, "rewards/IngredientQuantityMatchReward/mean": 0.6256462693214416, "rewards/IngredientQuantityMatchReward/std": 0.43513540625572206, "rewards/TotalKcalExactMatchReward/mean": 0.709375, "rewards/TotalKcalExactMatchReward/std": 0.4434183180332184, "step": 2345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 505.0, "completions/mean_length": 392.26875, "completions/min_length": 270.8, "epoch": 0.6536856745479833, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6249247193336487, "kl": 0.04946722569875419, "learning_rate": 2.936246639957529e-07, "loss": 0.001978399232029915, "reward": 3.093481254577637, "reward_std": 0.4033039391040802, "rewards/IngredientFormatReward/mean": 0.9912500023841858, "rewards/IngredientFormatReward/std": 0.06858442425727844, "rewards/IngredientMatchReward/mean": 0.6428137421607971, "rewards/IngredientMatchReward/std": 0.26454198360443115, "rewards/IngredientQuantityMatchReward/mean": 0.6312925815582275, "rewards/IngredientQuantityMatchReward/std": 0.4156554341316223, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.37494794726371766, "step": 2350 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 503.4, "completions/mean_length": 389.2296875, "completions/min_length": 238.6, "epoch": 0.655076495132128, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6602175235748291, "kl": 0.050524069415405395, "learning_rate": 2.915320514851627e-07, "loss": 0.002020845375955105, "reward": 3.121209001541138, "reward_std": 0.4390512526035309, "rewards/IngredientFormatReward/mean": 0.9868749976158142, "rewards/IngredientFormatReward/std": 0.08971829488873481, "rewards/IngredientMatchReward/mean": 0.6932979822158813, "rewards/IngredientMatchReward/std": 0.27488317489624026, "rewards/IngredientQuantityMatchReward/mean": 0.6410360217094422, "rewards/IngredientQuantityMatchReward/std": 0.41615588665008546, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.3990713179111481, "step": 2355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 495.8, "completions/mean_length": 393.64375, "completions/min_length": 273.0, "epoch": 0.6564673157162726, "frac_reward_zero_std": 0.075, "grad_norm": 0.5934226512908936, "kl": 0.04695138649549335, "learning_rate": 2.8944384956636436e-07, "loss": 0.0018780270591378212, "reward": 3.1596553325653076, "reward_std": 0.3678014278411865, "rewards/IngredientFormatReward/mean": 0.9976041674613952, "rewards/IngredientFormatReward/std": 0.02309950739145279, "rewards/IngredientMatchReward/mean": 0.6859988927841186, "rewards/IngredientMatchReward/std": 0.2769916921854019, "rewards/IngredientQuantityMatchReward/mean": 0.6854272723197937, "rewards/IngredientQuantityMatchReward/std": 0.40260618925094604, "rewards/TotalKcalExactMatchReward/mean": 0.790625, "rewards/TotalKcalExactMatchReward/std": 0.38472620248794553, "step": 2360 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 505.8, "completions/mean_length": 386.9125, "completions/min_length": 261.0, "epoch": 0.6578581363004172, "frac_reward_zero_std": 0.075, "grad_norm": 0.6448837518692017, "kl": 0.04659861430991441, "learning_rate": 2.873601024198011e-07, "loss": 0.001864098198711872, "reward": 3.1337858200073243, "reward_std": 0.336954790353775, "rewards/IngredientFormatReward/mean": 0.99375, "rewards/IngredientFormatReward/std": 0.06025671809911728, "rewards/IngredientMatchReward/mean": 0.6501909971237183, "rewards/IngredientMatchReward/std": 0.29020614326000216, "rewards/IngredientQuantityMatchReward/mean": 0.6664073586463928, "rewards/IngredientQuantityMatchReward/std": 0.41391525864601136, "rewards/TotalKcalExactMatchReward/mean": 0.8234375, "rewards/TotalKcalExactMatchReward/std": 0.38152927756309507, "step": 2365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 505.6, "completions/mean_length": 389.090625, "completions/min_length": 271.4, "epoch": 0.6592489568845619, "frac_reward_zero_std": 0.05, "grad_norm": 0.6337307691574097, "kl": 0.10345027199946344, "learning_rate": 2.8528085413166524e-07, "loss": 0.004148460924625397, "reward": 3.137276220321655, "reward_std": 0.42400845885276794, "rewards/IngredientFormatReward/mean": 0.9884374976158142, "rewards/IngredientFormatReward/std": 0.09104880094528198, "rewards/IngredientMatchReward/mean": 0.6613380670547485, "rewards/IngredientMatchReward/std": 0.2888974130153656, "rewards/IngredientQuantityMatchReward/mean": 0.6265631735324859, "rewards/IngredientQuantityMatchReward/std": 0.41725427508354185, "rewards/TotalKcalExactMatchReward/mean": 0.8609375, "rewards/TotalKcalExactMatchReward/std": 0.3248093917965889, "step": 2370 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 507.2, "completions/mean_length": 393.634375, "completions/min_length": 273.2, "epoch": 0.6606397774687065, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6260727047920227, "kl": 0.04925237852148712, "learning_rate": 2.832061486929656e-07, "loss": 0.0019701194018125535, "reward": 3.1139048099517823, "reward_std": 0.41956409215927126, "rewards/IngredientFormatReward/mean": 0.9828125, "rewards/IngredientFormatReward/std": 0.09825026765465736, "rewards/IngredientMatchReward/mean": 0.6576175332069397, "rewards/IngredientMatchReward/std": 0.3032703876495361, "rewards/IngredientQuantityMatchReward/mean": 0.6437873005867004, "rewards/IngredientQuantityMatchReward/std": 0.43343395590782163, "rewards/TotalKcalExactMatchReward/mean": 0.8296875, "rewards/TotalKcalExactMatchReward/std": 0.3699494957923889, "step": 2375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 510.0, "completions/mean_length": 393.4859375, "completions/min_length": 253.0, "epoch": 0.6620305980528511, "frac_reward_zero_std": 0.1125, "grad_norm": 0.6546136736869812, "kl": 0.04857088509015739, "learning_rate": 2.8113602999859756e-07, "loss": 0.0019431136548519134, "reward": 3.0654694080352782, "reward_std": 0.38434099555015566, "rewards/IngredientFormatReward/mean": 0.984375, "rewards/IngredientFormatReward/std": 0.10262673646211624, "rewards/IngredientMatchReward/mean": 0.6572916746139527, "rewards/IngredientMatchReward/std": 0.2996719777584076, "rewards/IngredientQuantityMatchReward/mean": 0.6441153109073638, "rewards/IngredientQuantityMatchReward/std": 0.4026394784450531, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.4011474013328552, "step": 2380 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 508.2, "completions/mean_length": 397.2328125, "completions/min_length": 286.4, "epoch": 0.6634214186369958, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7171292304992676, "kl": 0.04705277043394744, "learning_rate": 2.790705418464141e-07, "loss": 0.0018818611279129982, "reward": 3.167479467391968, "reward_std": 0.40963455438613894, "rewards/IngredientFormatReward/mean": 0.9888727784156799, "rewards/IngredientFormatReward/std": 0.07853043712675571, "rewards/IngredientMatchReward/mean": 0.7033141136169434, "rewards/IngredientMatchReward/std": 0.2901200741529465, "rewards/IngredientQuantityMatchReward/mean": 0.7034175992012024, "rewards/IngredientQuantityMatchReward/std": 0.39401066303253174, "rewards/TotalKcalExactMatchReward/mean": 0.771875, "rewards/TotalKcalExactMatchReward/std": 0.4138317406177521, "step": 2385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 508.0, "completions/mean_length": 391.1328125, "completions/min_length": 260.8, "epoch": 0.6648122392211405, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7062774896621704, "kl": 0.0725515817059204, "learning_rate": 2.770097279362986e-07, "loss": 0.0029019616544246674, "reward": 3.0785242557525634, "reward_std": 0.4356281340122223, "rewards/IngredientFormatReward/mean": 0.9871577501296998, "rewards/IngredientFormatReward/std": 0.08785807080566883, "rewards/IngredientMatchReward/mean": 0.6512655138969421, "rewards/IngredientMatchReward/std": 0.2800576239824295, "rewards/IngredientQuantityMatchReward/mean": 0.6526010870933533, "rewards/IngredientQuantityMatchReward/std": 0.40168362855911255, "rewards/TotalKcalExactMatchReward/mean": 0.7875, "rewards/TotalKcalExactMatchReward/std": 0.39667311906814573, "step": 2390 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 513.0, "completions/mean_length": 395.40625, "completions/min_length": 275.8, "epoch": 0.6662030598052852, "frac_reward_zero_std": 0.05, "grad_norm": 0.6249176263809204, "kl": 0.047266633994877336, "learning_rate": 2.749536318692412e-07, "loss": 0.0018904266878962516, "reward": 3.128342056274414, "reward_std": 0.40333980321884155, "rewards/IngredientFormatReward/mean": 0.9869791626930237, "rewards/IngredientFormatReward/std": 0.10851280093193054, "rewards/IngredientMatchReward/mean": 0.6634164214134216, "rewards/IngredientMatchReward/std": 0.2770405501127243, "rewards/IngredientQuantityMatchReward/mean": 0.6513839364051819, "rewards/IngredientQuantityMatchReward/std": 0.40092127323150634, "rewards/TotalKcalExactMatchReward/mean": 0.8265625, "rewards/TotalKcalExactMatchReward/std": 0.37440555095672606, "step": 2395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 511.6, "completions/mean_length": 398.01875, "completions/min_length": 271.0, "epoch": 0.6675938803894298, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6117978692054749, "kl": 0.04874067837372422, "learning_rate": 2.7290229714641545e-07, "loss": 0.0019499026238918304, "reward": 3.1176537036895753, "reward_std": 0.39257633686065674, "rewards/IngredientFormatReward/mean": 0.9817708253860473, "rewards/IngredientFormatReward/std": 0.1140602245926857, "rewards/IngredientMatchReward/mean": 0.6361191749572754, "rewards/IngredientMatchReward/std": 0.293190997838974, "rewards/IngredientQuantityMatchReward/mean": 0.6950762033462524, "rewards/IngredientQuantityMatchReward/std": 0.3991281509399414, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.3917708516120911, "step": 2400 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 506.8, "completions/mean_length": 395.8390625, "completions/min_length": 275.2, "epoch": 0.6689847009735744, "frac_reward_zero_std": 0.05, "grad_norm": 0.6851879358291626, "kl": 0.04812838714569807, "learning_rate": 2.708557671682585e-07, "loss": 0.0019251177087426185, "reward": 3.0077756881713866, "reward_std": 0.3903347373008728, "rewards/IngredientFormatReward/mean": 0.9932142972946167, "rewards/IngredientFormatReward/std": 0.060679107904434204, "rewards/IngredientMatchReward/mean": 0.6542746782302856, "rewards/IngredientMatchReward/std": 0.2864963531494141, "rewards/IngredientQuantityMatchReward/mean": 0.6087242603302002, "rewards/IngredientQuantityMatchReward/std": 0.43384186625480653, "rewards/TotalKcalExactMatchReward/mean": 0.7515625, "rewards/TotalKcalExactMatchReward/std": 0.4268688678741455, "step": 2405 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 504.8, "completions/mean_length": 394.96875, "completions/min_length": 269.2, "epoch": 0.6703755215577191, "frac_reward_zero_std": 0.075, "grad_norm": 0.651664674282074, "kl": 0.05091009079478681, "learning_rate": 2.6881408523355296e-07, "loss": 0.0020366348326206207, "reward": 3.191921615600586, "reward_std": 0.40628759264945985, "rewards/IngredientFormatReward/mean": 0.9890625, "rewards/IngredientFormatReward/std": 0.07843081802129745, "rewards/IngredientMatchReward/mean": 0.7092541575431823, "rewards/IngredientMatchReward/std": 0.2892185389995575, "rewards/IngredientQuantityMatchReward/mean": 0.6748549461364746, "rewards/IngredientQuantityMatchReward/std": 0.4222606658935547, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.374184775352478, "step": 2410 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 502.4, "completions/mean_length": 387.9296875, "completions/min_length": 265.2, "epoch": 0.6717663421418637, "frac_reward_zero_std": 0.05, "grad_norm": 0.6594770550727844, "kl": 0.04690086429473013, "learning_rate": 2.667772945385096e-07, "loss": 0.0018760446459054948, "reward": 3.105556106567383, "reward_std": 0.34251810908317565, "rewards/IngredientFormatReward/mean": 0.9966145753860474, "rewards/IngredientFormatReward/std": 0.035576280951499936, "rewards/IngredientMatchReward/mean": 0.6220870614051819, "rewards/IngredientMatchReward/std": 0.2938617765903473, "rewards/IngredientQuantityMatchReward/mean": 0.6837294578552247, "rewards/IngredientQuantityMatchReward/std": 0.4037123382091522, "rewards/TotalKcalExactMatchReward/mean": 0.803125, "rewards/TotalKcalExactMatchReward/std": 0.3925239682197571, "step": 2415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.6, "completions/mean_length": 392.546875, "completions/min_length": 273.0, "epoch": 0.6731571627260083, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6033135056495667, "kl": 0.04802413797006011, "learning_rate": 2.647454381758557e-07, "loss": 0.0019207324832677842, "reward": 3.080314111709595, "reward_std": 0.4404604434967041, "rewards/IngredientFormatReward/mean": 0.9889583230018616, "rewards/IngredientFormatReward/std": 0.0812400221824646, "rewards/IngredientMatchReward/mean": 0.6385317444801331, "rewards/IngredientMatchReward/std": 0.288109216094017, "rewards/IngredientQuantityMatchReward/mean": 0.6762615799903869, "rewards/IngredientQuantityMatchReward/std": 0.40315200090408326, "rewards/TotalKcalExactMatchReward/mean": 0.7765625, "rewards/TotalKcalExactMatchReward/std": 0.4032036066055298, "step": 2420 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 513.0, "completions/mean_length": 394.6265625, "completions/min_length": 265.2, "epoch": 0.674547983310153, "frac_reward_zero_std": 0.075, "grad_norm": 0.6370003819465637, "kl": 0.04624895497690886, "learning_rate": 2.627185591339212e-07, "loss": 0.0018500026315450668, "reward": 3.1716074466705324, "reward_std": 0.39525836110115053, "rewards/IngredientFormatReward/mean": 0.98515625, "rewards/IngredientFormatReward/std": 0.11617832779884338, "rewards/IngredientMatchReward/mean": 0.6576748609542846, "rewards/IngredientMatchReward/std": 0.2919008910655975, "rewards/IngredientQuantityMatchReward/mean": 0.6819014191627503, "rewards/IngredientQuantityMatchReward/std": 0.40078869462013245, "rewards/TotalKcalExactMatchReward/mean": 0.846875, "rewards/TotalKcalExactMatchReward/std": 0.35469257831573486, "step": 2425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 505.6, "completions/mean_length": 386.7359375, "completions/min_length": 245.2, "epoch": 0.6759388038942976, "frac_reward_zero_std": 0.075, "grad_norm": 0.6779886484146118, "kl": 0.04996294579468667, "learning_rate": 2.606967002957303e-07, "loss": 0.0019983302801847456, "reward": 3.0284729957580567, "reward_std": 0.43177472352981566, "rewards/IngredientFormatReward/mean": 0.9650520801544189, "rewards/IngredientFormatReward/std": 0.15353514421731235, "rewards/IngredientMatchReward/mean": 0.6522922873497009, "rewards/IngredientMatchReward/std": 0.31233987808227537, "rewards/IngredientQuantityMatchReward/mean": 0.6408162117004395, "rewards/IngredientQuantityMatchReward/std": 0.410346245765686, "rewards/TotalKcalExactMatchReward/mean": 0.7703125, "rewards/TotalKcalExactMatchReward/std": 0.4039487600326538, "step": 2430 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 509.0, "completions/mean_length": 392.6703125, "completions/min_length": 281.4, "epoch": 0.6773296244784422, "frac_reward_zero_std": 0.05, "grad_norm": 0.6478075981140137, "kl": 0.04866862255148589, "learning_rate": 2.5867990443809365e-07, "loss": 0.0019466172903776168, "reward": 3.102987194061279, "reward_std": 0.44889168739318847, "rewards/IngredientFormatReward/mean": 0.9877343773841858, "rewards/IngredientFormatReward/std": 0.08591293394565583, "rewards/IngredientMatchReward/mean": 0.6462679743766785, "rewards/IngredientMatchReward/std": 0.2983993053436279, "rewards/IngredientQuantityMatchReward/mean": 0.6330472826957703, "rewards/IngredientQuantityMatchReward/std": 0.42972511053085327, "rewards/TotalKcalExactMatchReward/mean": 0.8359375, "rewards/TotalKcalExactMatchReward/std": 0.3645253419876099, "step": 2435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 510.2, "completions/mean_length": 389.6546875, "completions/min_length": 233.2, "epoch": 0.6787204450625869, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6969110369682312, "kl": 0.0487577939638868, "learning_rate": 2.5666821423070384e-07, "loss": 0.001950256898999214, "reward": 3.03529806137085, "reward_std": 0.39315775632858274, "rewards/IngredientFormatReward/mean": 0.9934375047683716, "rewards/IngredientFormatReward/std": 0.0710334450006485, "rewards/IngredientMatchReward/mean": 0.643232274055481, "rewards/IngredientMatchReward/std": 0.27578197717666625, "rewards/IngredientQuantityMatchReward/mean": 0.600190806388855, "rewards/IngredientQuantityMatchReward/std": 0.4328667461872101, "rewards/TotalKcalExactMatchReward/mean": 0.7984375, "rewards/TotalKcalExactMatchReward/std": 0.39933934807777405, "step": 2440 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 496.2, "completions/mean_length": 395.3625, "completions/min_length": 277.2, "epoch": 0.6801112656467315, "frac_reward_zero_std": 0.0375, "grad_norm": 0.629336953163147, "kl": 0.049779637204483154, "learning_rate": 2.5466167223523207e-07, "loss": 0.0019913079217076302, "reward": 3.1274917125701904, "reward_std": 0.35138294100761414, "rewards/IngredientFormatReward/mean": 0.9953348159790039, "rewards/IngredientFormatReward/std": 0.04012070763856172, "rewards/IngredientMatchReward/mean": 0.6890367865562439, "rewards/IngredientMatchReward/std": 0.2609582722187042, "rewards/IngredientQuantityMatchReward/mean": 0.6618700504302979, "rewards/IngredientQuantityMatchReward/std": 0.40753733515739443, "rewards/TotalKcalExactMatchReward/mean": 0.78125, "rewards/TotalKcalExactMatchReward/std": 0.4101200461387634, "step": 2445 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 509.4, "completions/mean_length": 396.7015625, "completions/min_length": 273.4, "epoch": 0.6815020862308763, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6589193344116211, "kl": 0.0490143105853349, "learning_rate": 2.5266032090442857e-07, "loss": 0.001960846781730652, "reward": 3.1061669826507567, "reward_std": 0.4119180262088776, "rewards/IngredientFormatReward/mean": 0.9869642972946167, "rewards/IngredientFormatReward/std": 0.0859108954668045, "rewards/IngredientMatchReward/mean": 0.6312419414520264, "rewards/IngredientMatchReward/std": 0.2935998737812042, "rewards/IngredientQuantityMatchReward/mean": 0.7145232081413269, "rewards/IngredientQuantityMatchReward/std": 0.38715381026268003, "rewards/TotalKcalExactMatchReward/mean": 0.7734375, "rewards/TotalKcalExactMatchReward/std": 0.41800680160522463, "step": 2450 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 509.8, "completions/mean_length": 401.896875, "completions/min_length": 264.4, "epoch": 0.6828929068150209, "frac_reward_zero_std": 0.05, "grad_norm": 5.397259712219238, "kl": 0.0734444867586717, "learning_rate": 2.50664202581223e-07, "loss": 0.0029396936297416687, "reward": 2.9933862686157227, "reward_std": 0.4221198558807373, "rewards/IngredientFormatReward/mean": 0.9723698019981384, "rewards/IngredientFormatReward/std": 0.1247881144285202, "rewards/IngredientMatchReward/mean": 0.6287630200386047, "rewards/IngredientMatchReward/std": 0.29730098247528075, "rewards/IngredientQuantityMatchReward/mean": 0.5828785061836242, "rewards/IngredientQuantityMatchReward/std": 0.4397486746311188, "rewards/TotalKcalExactMatchReward/mean": 0.809375, "rewards/TotalKcalExactMatchReward/std": 0.37806967496871946, "step": 2455 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 510.0, "completions/mean_length": 394.15625, "completions/min_length": 247.4, "epoch": 0.6842837273991655, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6566894054412842, "kl": 0.07639441229403018, "learning_rate": 2.4867335949782977e-07, "loss": 0.003058093413710594, "reward": 3.0628830909729006, "reward_std": 0.42465184926986693, "rewards/IngredientFormatReward/mean": 0.9839192509651185, "rewards/IngredientFormatReward/std": 0.11864397823810577, "rewards/IngredientMatchReward/mean": 0.6475834965705871, "rewards/IngredientMatchReward/std": 0.2913681983947754, "rewards/IngredientQuantityMatchReward/mean": 0.6204427242279053, "rewards/IngredientQuantityMatchReward/std": 0.4370134472846985, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.38879077434539794, "step": 2460 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.8, "completions/mean_length": 396.884375, "completions/min_length": 264.0, "epoch": 0.6856745479833102, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6536556482315063, "kl": 0.045558814378455284, "learning_rate": 2.4668783377485406e-07, "loss": 0.0018225433304905891, "reward": 3.0459858417510985, "reward_std": 0.43674235939979555, "rewards/IngredientFormatReward/mean": 0.9867038726806641, "rewards/IngredientFormatReward/std": 0.09818108770996332, "rewards/IngredientMatchReward/mean": 0.6260255217552185, "rewards/IngredientMatchReward/std": 0.2841141104698181, "rewards/IngredientQuantityMatchReward/mean": 0.6441938519477844, "rewards/IngredientQuantityMatchReward/std": 0.4117276132106781, "rewards/TotalKcalExactMatchReward/mean": 0.7890625, "rewards/TotalKcalExactMatchReward/std": 0.399785590171814, "step": 2465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 510.6, "completions/mean_length": 398.1921875, "completions/min_length": 265.0, "epoch": 0.6870653685674548, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6002658605575562, "kl": 0.047888010111637416, "learning_rate": 2.4470766742040107e-07, "loss": 0.0019158570095896721, "reward": 3.166787576675415, "reward_std": 0.38818521797657013, "rewards/IngredientFormatReward/mean": 0.9827827453613281, "rewards/IngredientFormatReward/std": 0.11082592085003853, "rewards/IngredientMatchReward/mean": 0.64622642993927, "rewards/IngredientMatchReward/std": 0.28593012094497683, "rewards/IngredientQuantityMatchReward/mean": 0.6784033298492431, "rewards/IngredientQuantityMatchReward/std": 0.41570827960968015, "rewards/TotalKcalExactMatchReward/mean": 0.859375, "rewards/TotalKcalExactMatchReward/std": 0.3468464195728302, "step": 2470 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 508.6, "completions/mean_length": 394.9578125, "completions/min_length": 263.6, "epoch": 0.6884561891515995, "frac_reward_zero_std": 0.075, "grad_norm": 0.6655333638191223, "kl": 0.05003447597846389, "learning_rate": 2.427329023291864e-07, "loss": 0.002001406252384186, "reward": 3.0961260318756105, "reward_std": 0.4364780128002167, "rewards/IngredientFormatReward/mean": 0.9879687428474426, "rewards/IngredientFormatReward/std": 0.0807617649435997, "rewards/IngredientMatchReward/mean": 0.6726500511169433, "rewards/IngredientMatchReward/std": 0.28994928300380707, "rewards/IngredientQuantityMatchReward/mean": 0.6292572021484375, "rewards/IngredientQuantityMatchReward/std": 0.43481956124305726, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.39507427215576174, "step": 2475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 512.6, "completions/mean_length": 394.7015625, "completions/min_length": 247.6, "epoch": 0.6898470097357441, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6104139685630798, "kl": 0.05500267161987722, "learning_rate": 2.4076358028165053e-07, "loss": 0.002200280874967575, "reward": 3.086128854751587, "reward_std": 0.42833173274993896, "rewards/IngredientFormatReward/mean": 0.9812053442001343, "rewards/IngredientFormatReward/std": 0.11338990479707718, "rewards/IngredientMatchReward/mean": 0.6116282224655152, "rewards/IngredientMatchReward/std": 0.2804492115974426, "rewards/IngredientQuantityMatchReward/mean": 0.6339202046394348, "rewards/IngredientQuantityMatchReward/std": 0.4178707242012024, "rewards/TotalKcalExactMatchReward/mean": 0.859375, "rewards/TotalKcalExactMatchReward/std": 0.32549999952316283, "step": 2480 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 511.4, "completions/mean_length": 389.6890625, "completions/min_length": 294.0, "epoch": 0.6912378303198887, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6676151752471924, "kl": 0.048110749153420326, "learning_rate": 2.387997429430746e-07, "loss": 0.001924731768667698, "reward": 3.2237563610076903, "reward_std": 0.36057083010673524, "rewards/IngredientFormatReward/mean": 0.9906845331192017, "rewards/IngredientFormatReward/std": 0.08420297093689441, "rewards/IngredientMatchReward/mean": 0.7073366641998291, "rewards/IngredientMatchReward/std": 0.2901332199573517, "rewards/IngredientQuantityMatchReward/mean": 0.7022976398468017, "rewards/IngredientQuantityMatchReward/std": 0.3529376655817032, "rewards/TotalKcalExactMatchReward/mean": 0.8234375, "rewards/TotalKcalExactMatchReward/std": 0.3717910945415497, "step": 2485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 497.2, "completions/mean_length": 389.11875, "completions/min_length": 252.8, "epoch": 0.6926286509040334, "frac_reward_zero_std": 0.05, "grad_norm": 0.6633597612380981, "kl": 0.048306941101327536, "learning_rate": 2.3684143186269885e-07, "loss": 0.0019321149215102197, "reward": 3.178741788864136, "reward_std": 0.377359014749527, "rewards/IngredientFormatReward/mean": 0.9950000047683716, "rewards/IngredientFormatReward/std": 0.04611458256840706, "rewards/IngredientMatchReward/mean": 0.7043855547904968, "rewards/IngredientMatchReward/std": 0.2860237658023834, "rewards/IngredientQuantityMatchReward/mean": 0.6965437650680542, "rewards/IngredientQuantityMatchReward/std": 0.39232956767082217, "rewards/TotalKcalExactMatchReward/mean": 0.7828125, "rewards/TotalKcalExactMatchReward/std": 0.40132089853286745, "step": 2490 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 498.2, "completions/mean_length": 391.478125, "completions/min_length": 264.6, "epoch": 0.694019471488178, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7364270687103271, "kl": 0.04957497245632112, "learning_rate": 2.3488868847284292e-07, "loss": 0.0019831184297800066, "reward": 3.1426888942718505, "reward_std": 0.3652894914150238, "rewards/IngredientFormatReward/mean": 0.996932053565979, "rewards/IngredientFormatReward/std": 0.02726769857108593, "rewards/IngredientMatchReward/mean": 0.6408079981803894, "rewards/IngredientMatchReward/std": 0.2871991991996765, "rewards/IngredientQuantityMatchReward/mean": 0.6596363306045532, "rewards/IngredientQuantityMatchReward/std": 0.41957661509513855, "rewards/TotalKcalExactMatchReward/mean": 0.8453125, "rewards/TotalKcalExactMatchReward/std": 0.3346115857362747, "step": 2495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 504.2, "completions/mean_length": 394.440625, "completions/min_length": 263.6, "epoch": 0.6954102920723226, "frac_reward_zero_std": 0.075, "grad_norm": 0.664070725440979, "kl": 0.05223251197021454, "learning_rate": 2.3294155408803067e-07, "loss": 0.0020896628499031066, "reward": 3.0636887550354004, "reward_std": 0.3960974931716919, "rewards/IngredientFormatReward/mean": 0.9897916555404663, "rewards/IngredientFormatReward/std": 0.08864808697253465, "rewards/IngredientMatchReward/mean": 0.6588827013969422, "rewards/IngredientMatchReward/std": 0.3019312024116516, "rewards/IngredientQuantityMatchReward/mean": 0.585326862335205, "rewards/IngredientQuantityMatchReward/std": 0.4312814950942993, "rewards/TotalKcalExactMatchReward/mean": 0.8296875, "rewards/TotalKcalExactMatchReward/std": 0.3555891036987305, "step": 2500 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.028125, "completions/max_length": 511.6, "completions/mean_length": 399.46875, "completions/min_length": 279.0, "epoch": 0.6968011126564673, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6669741868972778, "kl": 0.050500425044447184, "learning_rate": 2.3100006990411474e-07, "loss": 0.0020199045538902283, "reward": 3.1506638526916504, "reward_std": 0.468947970867157, "rewards/IngredientFormatReward/mean": 0.9685267925262451, "rewards/IngredientFormatReward/std": 0.1518846545368433, "rewards/IngredientMatchReward/mean": 0.6512828588485717, "rewards/IngredientMatchReward/std": 0.2848780870437622, "rewards/IngredientQuantityMatchReward/mean": 0.6996042728424072, "rewards/IngredientQuantityMatchReward/std": 0.4034592926502228, "rewards/TotalKcalExactMatchReward/mean": 0.83125, "rewards/TotalKcalExactMatchReward/std": 0.34879711270332336, "step": 2505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 508.2, "completions/mean_length": 392.584375, "completions/min_length": 253.6, "epoch": 0.6981919332406119, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7342477440834045, "kl": 0.05032761418260634, "learning_rate": 2.2906427699740627e-07, "loss": 0.0020130135118961332, "reward": 3.1510085582733156, "reward_std": 0.439526242017746, "rewards/IngredientFormatReward/mean": 0.9832812547683716, "rewards/IngredientFormatReward/std": 0.11577541343867778, "rewards/IngredientMatchReward/mean": 0.668173360824585, "rewards/IngredientMatchReward/std": 0.2981277287006378, "rewards/IngredientQuantityMatchReward/mean": 0.6729914307594299, "rewards/IngredientQuantityMatchReward/std": 0.42339866757392886, "rewards/TotalKcalExactMatchReward/mean": 0.8265625, "rewards/TotalKcalExactMatchReward/std": 0.37619284987449647, "step": 2510 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 513.0, "completions/mean_length": 398.271875, "completions/min_length": 269.4, "epoch": 0.6995827538247567, "frac_reward_zero_std": 0.075, "grad_norm": 0.6291016340255737, "kl": 0.045997858187183736, "learning_rate": 2.271342163238041e-07, "loss": 0.0018398519605398178, "reward": 3.1013444900512694, "reward_std": 0.43752923607826233, "rewards/IngredientFormatReward/mean": 0.96875, "rewards/IngredientFormatReward/std": 0.16352393478155136, "rewards/IngredientMatchReward/mean": 0.6443898916244507, "rewards/IngredientMatchReward/std": 0.3039585888385773, "rewards/IngredientQuantityMatchReward/mean": 0.6710170984268189, "rewards/IngredientQuantityMatchReward/std": 0.42277750968933103, "rewards/TotalKcalExactMatchReward/mean": 0.8171875, "rewards/TotalKcalExactMatchReward/std": 0.3669287174940109, "step": 2515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 513.0, "completions/mean_length": 392.303125, "completions/min_length": 234.8, "epoch": 0.7009735744089013, "frac_reward_zero_std": 0.05, "grad_norm": 4.0983734130859375, "kl": 0.06823289862368256, "learning_rate": 2.2520992871792977e-07, "loss": 0.0027291785925626756, "reward": 2.946266031265259, "reward_std": 0.4240442216396332, "rewards/IngredientFormatReward/mean": 0.9885416507720948, "rewards/IngredientFormatReward/std": 0.10327765196561814, "rewards/IngredientMatchReward/mean": 0.6329321742057801, "rewards/IngredientMatchReward/std": 0.2933188438415527, "rewards/IngredientQuantityMatchReward/mean": 0.6232297539710998, "rewards/IngredientQuantityMatchReward/std": 0.44109280705451964, "rewards/TotalKcalExactMatchReward/mean": 0.7015625, "rewards/TotalKcalExactMatchReward/std": 0.4390376150608063, "step": 2520 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 503.8, "completions/mean_length": 391.29375, "completions/min_length": 262.6, "epoch": 0.7023643949930459, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6922910213470459, "kl": 0.05018776603974402, "learning_rate": 2.2329145489226304e-07, "loss": 0.0020075805485248567, "reward": 3.0498045921325683, "reward_std": 0.4153340578079224, "rewards/IngredientFormatReward/mean": 0.9825595259666443, "rewards/IngredientFormatReward/std": 0.0890352837741375, "rewards/IngredientMatchReward/mean": 0.6578689217567444, "rewards/IngredientMatchReward/std": 0.2839093506336212, "rewards/IngredientQuantityMatchReward/mean": 0.6328136444091796, "rewards/IngredientQuantityMatchReward/std": 0.40785926580429077, "rewards/TotalKcalExactMatchReward/mean": 0.7765625, "rewards/TotalKcalExactMatchReward/std": 0.4117879748344421, "step": 2525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 506.8, "completions/mean_length": 394.7578125, "completions/min_length": 260.6, "epoch": 0.7037552155771906, "frac_reward_zero_std": 0.075, "grad_norm": 0.6483815908432007, "kl": 0.04718795008957386, "learning_rate": 2.2137883543628044e-07, "loss": 0.0018875142559409142, "reward": 3.110764265060425, "reward_std": 0.40262900590896605, "rewards/IngredientFormatReward/mean": 0.9931249976158142, "rewards/IngredientFormatReward/std": 0.06067222654819489, "rewards/IngredientMatchReward/mean": 0.671670389175415, "rewards/IngredientMatchReward/std": 0.2906846523284912, "rewards/IngredientQuantityMatchReward/mean": 0.6490937829017639, "rewards/IngredientQuantityMatchReward/std": 0.4294052541255951, "rewards/TotalKcalExactMatchReward/mean": 0.796875, "rewards/TotalKcalExactMatchReward/std": 0.3973664820194244, "step": 2530 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 511.6, "completions/mean_length": 395.5609375, "completions/min_length": 270.4, "epoch": 0.7051460361613352, "frac_reward_zero_std": 0.1, "grad_norm": 0.6322222948074341, "kl": 0.047202359139919284, "learning_rate": 2.1947211081559664e-07, "loss": 0.0018881641328334809, "reward": 3.0978555202484133, "reward_std": 0.4240407347679138, "rewards/IngredientFormatReward/mean": 0.979281997680664, "rewards/IngredientFormatReward/std": 0.1245550911873579, "rewards/IngredientMatchReward/mean": 0.6488405346870423, "rewards/IngredientMatchReward/std": 0.29211672842502595, "rewards/IngredientQuantityMatchReward/mean": 0.660357940196991, "rewards/IngredientQuantityMatchReward/std": 0.41334924697875974, "rewards/TotalKcalExactMatchReward/mean": 0.809375, "rewards/TotalKcalExactMatchReward/std": 0.392572158575058, "step": 2535 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 510.6, "completions/mean_length": 391.8828125, "completions/min_length": 262.8, "epoch": 0.7065368567454798, "frac_reward_zero_std": 0.075, "grad_norm": 0.6519886255264282, "kl": 0.060429237270727756, "learning_rate": 2.1757132137110823e-07, "loss": 0.002402695082128048, "reward": 3.05753607749939, "reward_std": 0.4117233157157898, "rewards/IngredientFormatReward/mean": 0.9845312595367431, "rewards/IngredientFormatReward/std": 0.10385205522179604, "rewards/IngredientMatchReward/mean": 0.6303422689437866, "rewards/IngredientMatchReward/std": 0.2817304372787476, "rewards/IngredientQuantityMatchReward/mean": 0.6317250967025757, "rewards/IngredientQuantityMatchReward/std": 0.4206810534000397, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.3870365977287292, "step": 2540 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 508.6, "completions/mean_length": 393.0890625, "completions/min_length": 270.6, "epoch": 0.7079276773296245, "frac_reward_zero_std": 0.0625, "grad_norm": 0.5975901484489441, "kl": 0.05394169162027538, "learning_rate": 2.156765073181404e-07, "loss": 0.0021578073501586914, "reward": 2.9798144340515136, "reward_std": 0.37361904978752136, "rewards/IngredientFormatReward/mean": 0.9895572900772095, "rewards/IngredientFormatReward/std": 0.07822814472019672, "rewards/IngredientMatchReward/mean": 0.5983500719070435, "rewards/IngredientMatchReward/std": 0.30022152662277224, "rewards/IngredientQuantityMatchReward/mean": 0.6169070601463318, "rewards/IngredientQuantityMatchReward/std": 0.44377334117889405, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.4066710531711578, "step": 2545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 513.0, "completions/mean_length": 396.8390625, "completions/min_length": 242.4, "epoch": 0.7093184979137691, "frac_reward_zero_std": 0.075, "grad_norm": 0.6239141225814819, "kl": 0.046976763568818566, "learning_rate": 2.1378770874559605e-07, "loss": 0.001879183202981949, "reward": 3.0559041023254396, "reward_std": 0.4519073784351349, "rewards/IngredientFormatReward/mean": 0.9795312404632568, "rewards/IngredientFormatReward/std": 0.13446595668792724, "rewards/IngredientMatchReward/mean": 0.6565352320671082, "rewards/IngredientMatchReward/std": 0.2887867510318756, "rewards/IngredientQuantityMatchReward/mean": 0.6714001893997192, "rewards/IngredientQuantityMatchReward/std": 0.40756980776786805, "rewards/TotalKcalExactMatchReward/mean": 0.7484375, "rewards/TotalKcalExactMatchReward/std": 0.4154438376426697, "step": 2550 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 513.0, "completions/mean_length": 395.01875, "completions/min_length": 279.0, "epoch": 0.7107093184979137, "frac_reward_zero_std": 0.075, "grad_norm": 0.6433897614479065, "kl": 0.04894828526303172, "learning_rate": 2.119049656151069e-07, "loss": 0.001957985386252403, "reward": 3.025065851211548, "reward_std": 0.4066974937915802, "rewards/IngredientFormatReward/mean": 0.980859375, "rewards/IngredientFormatReward/std": 0.13396839499473573, "rewards/IngredientMatchReward/mean": 0.6333761096000672, "rewards/IngredientMatchReward/std": 0.2985145926475525, "rewards/IngredientQuantityMatchReward/mean": 0.6202053546905517, "rewards/IngredientQuantityMatchReward/std": 0.42285839915275575, "rewards/TotalKcalExactMatchReward/mean": 0.790625, "rewards/TotalKcalExactMatchReward/std": 0.4044800400733948, "step": 2555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 494.8, "completions/mean_length": 383.159375, "completions/min_length": 244.6, "epoch": 0.7121001390820584, "frac_reward_zero_std": 0.025, "grad_norm": 0.6727476119995117, "kl": 0.04658746891655028, "learning_rate": 2.100283177601892e-07, "loss": 0.0018635163083672523, "reward": 3.1245415687561033, "reward_std": 0.40727477669715884, "rewards/IngredientFormatReward/mean": 0.99375, "rewards/IngredientFormatReward/std": 0.06025671809911728, "rewards/IngredientMatchReward/mean": 0.6800031065940857, "rewards/IngredientMatchReward/std": 0.27216280102729795, "rewards/IngredientQuantityMatchReward/mean": 0.64922593832016, "rewards/IngredientQuantityMatchReward/std": 0.4209359347820282, "rewards/TotalKcalExactMatchReward/mean": 0.8015625, "rewards/TotalKcalExactMatchReward/std": 0.3881435811519623, "step": 2560 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 511.8, "completions/mean_length": 394.890625, "completions/min_length": 268.8, "epoch": 0.713490959666203, "frac_reward_zero_std": 0.075, "grad_norm": 0.6577991843223572, "kl": 0.04966968102380633, "learning_rate": 2.081578048854007e-07, "loss": 0.001986708864569664, "reward": 3.146220636367798, "reward_std": 0.397566819190979, "rewards/IngredientFormatReward/mean": 0.9839787960052491, "rewards/IngredientFormatReward/std": 0.10710346139967442, "rewards/IngredientMatchReward/mean": 0.6619359254837036, "rewards/IngredientMatchReward/std": 0.27924045324325564, "rewards/IngredientQuantityMatchReward/mean": 0.6721809267997741, "rewards/IngredientQuantityMatchReward/std": 0.3993909597396851, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.36588002145290377, "step": 2565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 510.0, "completions/mean_length": 395.153125, "completions/min_length": 275.8, "epoch": 0.7148817802503477, "frac_reward_zero_std": 0.1, "grad_norm": 0.6442493200302124, "kl": 0.045088665070943536, "learning_rate": 2.0629346656549995e-07, "loss": 0.001803700253367424, "reward": 3.17347731590271, "reward_std": 0.41533090472221373, "rewards/IngredientFormatReward/mean": 0.9840625047683715, "rewards/IngredientFormatReward/std": 0.10534819960594177, "rewards/IngredientMatchReward/mean": 0.6939707398414612, "rewards/IngredientMatchReward/std": 0.2825161337852478, "rewards/IngredientQuantityMatchReward/mean": 0.7095066428184509, "rewards/IngredientQuantityMatchReward/std": 0.38486247062683104, "rewards/TotalKcalExactMatchReward/mean": 0.7859375, "rewards/TotalKcalExactMatchReward/std": 0.40840469002723695, "step": 2570 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 513.0, "completions/mean_length": 397.44375, "completions/min_length": 277.8, "epoch": 0.7162726008344924, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7509127259254456, "kl": 0.046041283709928396, "learning_rate": 2.0443534224460907e-07, "loss": 0.001841634139418602, "reward": 3.1562008380889894, "reward_std": 0.42731393575668336, "rewards/IngredientFormatReward/mean": 0.9847916603088379, "rewards/IngredientFormatReward/std": 0.11797711700201034, "rewards/IngredientMatchReward/mean": 0.650058913230896, "rewards/IngredientMatchReward/std": 0.28965330123901367, "rewards/IngredientQuantityMatchReward/mean": 0.7229127407073974, "rewards/IngredientQuantityMatchReward/std": 0.3845226109027863, "rewards/TotalKcalExactMatchReward/mean": 0.7984375, "rewards/TotalKcalExactMatchReward/std": 0.3828358560800552, "step": 2575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 513.0, "completions/mean_length": 395.8515625, "completions/min_length": 267.8, "epoch": 0.717663421418637, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6616153120994568, "kl": 0.049780993908643725, "learning_rate": 2.025834712353801e-07, "loss": 0.0019911348819732664, "reward": 3.0544140338897705, "reward_std": 0.39741013646125795, "rewards/IngredientFormatReward/mean": 0.9822098255157471, "rewards/IngredientFormatReward/std": 0.12083162069320678, "rewards/IngredientMatchReward/mean": 0.6724361538887024, "rewards/IngredientMatchReward/std": 0.28971803188323975, "rewards/IngredientQuantityMatchReward/mean": 0.6857055306434632, "rewards/IngredientQuantityMatchReward/std": 0.3946858704090118, "rewards/TotalKcalExactMatchReward/mean": 0.7140625, "rewards/TotalKcalExactMatchReward/std": 0.4354426920413971, "step": 2580 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 505.2, "completions/mean_length": 393.409375, "completions/min_length": 230.8, "epoch": 0.7190542420027817, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6178598403930664, "kl": 0.048539007431827486, "learning_rate": 2.0073789271816243e-07, "loss": 0.0019415821880102158, "reward": 3.0660136222839354, "reward_std": 0.3811270177364349, "rewards/IngredientFormatReward/mean": 0.991796875, "rewards/IngredientFormatReward/std": 0.05703234635293484, "rewards/IngredientMatchReward/mean": 0.6717540979385376, "rewards/IngredientMatchReward/std": 0.28819016814231874, "rewards/IngredientQuantityMatchReward/mean": 0.6602752327919006, "rewards/IngredientQuantityMatchReward/std": 0.40727475881576536, "rewards/TotalKcalExactMatchReward/mean": 0.7421875, "rewards/TotalKcalExactMatchReward/std": 0.4277864396572113, "step": 2585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 504.2, "completions/mean_length": 392.73125, "completions/min_length": 267.8, "epoch": 0.7204450625869263, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6725686192512512, "kl": 0.04864217657595873, "learning_rate": 1.9889864574017428e-07, "loss": 0.001945541426539421, "reward": 3.1187971591949464, "reward_std": 0.3799791693687439, "rewards/IngredientFormatReward/mean": 0.9953645825386047, "rewards/IngredientFormatReward/std": 0.04649723395705223, "rewards/IngredientMatchReward/mean": 0.6216610789299011, "rewards/IngredientMatchReward/std": 0.2948246955871582, "rewards/IngredientQuantityMatchReward/mean": 0.6548964500427246, "rewards/IngredientQuantityMatchReward/std": 0.41355618834495544, "rewards/TotalKcalExactMatchReward/mean": 0.846875, "rewards/TotalKcalExactMatchReward/std": 0.34510646760463715, "step": 2590 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 512.8, "completions/mean_length": 394.4046875, "completions/min_length": 270.8, "epoch": 0.721835883171071, "frac_reward_zero_std": 0.05, "grad_norm": 0.6282512545585632, "kl": 0.046746038901619616, "learning_rate": 1.9706576921467627e-07, "loss": 0.0018698884174227714, "reward": 3.0880001544952393, "reward_std": 0.4526852428913116, "rewards/IngredientFormatReward/mean": 0.9884895801544189, "rewards/IngredientFormatReward/std": 0.09109702110290527, "rewards/IngredientMatchReward/mean": 0.6641840219497681, "rewards/IngredientMatchReward/std": 0.289777010679245, "rewards/IngredientQuantityMatchReward/mean": 0.6540765881538391, "rewards/IngredientQuantityMatchReward/std": 0.4062558710575104, "rewards/TotalKcalExactMatchReward/mean": 0.78125, "rewards/TotalKcalExactMatchReward/std": 0.4044061481952667, "step": 2595 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 508.0, "completions/mean_length": 390.415625, "completions/min_length": 253.6, "epoch": 0.7232267037552156, "frac_reward_zero_std": 0.05, "grad_norm": 0.6331611275672913, "kl": 0.04611538765020669, "learning_rate": 1.9523930192014838e-07, "loss": 0.0018447006121277809, "reward": 3.1182607650756835, "reward_std": 0.4262834906578064, "rewards/IngredientFormatReward/mean": 0.9869791507720947, "rewards/IngredientFormatReward/std": 0.10090548861771823, "rewards/IngredientMatchReward/mean": 0.6147550702095032, "rewards/IngredientMatchReward/std": 0.285012686252594, "rewards/IngredientQuantityMatchReward/mean": 0.7415265083312989, "rewards/IngredientQuantityMatchReward/std": 0.3865606963634491, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.40770527720451355, "step": 2600 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 505.8, "completions/mean_length": 387.396875, "completions/min_length": 236.4, "epoch": 0.7246175243393602, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6415396332740784, "kl": 0.05292570195160806, "learning_rate": 1.9341928249946937e-07, "loss": 0.0021176621317863463, "reward": 3.027360534667969, "reward_std": 0.39936086535453796, "rewards/IngredientFormatReward/mean": 0.9881770730018615, "rewards/IngredientFormatReward/std": 0.09301825910806656, "rewards/IngredientMatchReward/mean": 0.6336619615554809, "rewards/IngredientMatchReward/std": 0.2789487838745117, "rewards/IngredientQuantityMatchReward/mean": 0.6789589166641236, "rewards/IngredientQuantityMatchReward/std": 0.4066279888153076, "rewards/TotalKcalExactMatchReward/mean": 0.7265625, "rewards/TotalKcalExactMatchReward/std": 0.43816269040107725, "step": 2605 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 507.6, "completions/mean_length": 389.815625, "completions/min_length": 249.6, "epoch": 0.7260083449235049, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6317312717437744, "kl": 0.05009318320080638, "learning_rate": 1.9160574945909934e-07, "loss": 0.0020037142559885977, "reward": 3.151788091659546, "reward_std": 0.3983535706996918, "rewards/IngredientFormatReward/mean": 0.98302081823349, "rewards/IngredientFormatReward/std": 0.09761920589953661, "rewards/IngredientMatchReward/mean": 0.6275359749794006, "rewards/IngredientMatchReward/std": 0.28289095759391786, "rewards/IngredientQuantityMatchReward/mean": 0.6756062507629395, "rewards/IngredientQuantityMatchReward/std": 0.4136093854904175, "rewards/TotalKcalExactMatchReward/mean": 0.865625, "rewards/TotalKcalExactMatchReward/std": 0.3261595815420151, "step": 2610 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 502.6, "completions/mean_length": 387.7125, "completions/min_length": 240.0, "epoch": 0.7273991655076495, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6524665355682373, "kl": 0.049897806439548734, "learning_rate": 1.8979874116826434e-07, "loss": 0.0019955608993768694, "reward": 3.136799764633179, "reward_std": 0.4250612795352936, "rewards/IngredientFormatReward/mean": 0.9895833253860473, "rewards/IngredientFormatReward/std": 0.08923212587833404, "rewards/IngredientMatchReward/mean": 0.6881510496139527, "rewards/IngredientMatchReward/std": 0.2866966396570206, "rewards/IngredientQuantityMatchReward/mean": 0.6512529373168945, "rewards/IngredientQuantityMatchReward/std": 0.3973442554473877, "rewards/TotalKcalExactMatchReward/mean": 0.8078125, "rewards/TotalKcalExactMatchReward/std": 0.3602495089173317, "step": 2615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 502.2, "completions/mean_length": 387.3828125, "completions/min_length": 239.4, "epoch": 0.7287899860917941, "frac_reward_zero_std": 0.1125, "grad_norm": 0.611609160900116, "kl": 0.04605715712532401, "learning_rate": 1.8799829585814626e-07, "loss": 0.0018422754481434822, "reward": 3.1545867919921875, "reward_std": 0.34553852677345276, "rewards/IngredientFormatReward/mean": 0.9854166507720947, "rewards/IngredientFormatReward/std": 0.08996033240109683, "rewards/IngredientMatchReward/mean": 0.6572426676750183, "rewards/IngredientMatchReward/std": 0.2847525715827942, "rewards/IngredientQuantityMatchReward/mean": 0.7134899377822876, "rewards/IngredientQuantityMatchReward/std": 0.3990268886089325, "rewards/TotalKcalExactMatchReward/mean": 0.7984375, "rewards/TotalKcalExactMatchReward/std": 0.39305478930473325, "step": 2620 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 498.8, "completions/mean_length": 389.04375, "completions/min_length": 270.8, "epoch": 0.7301808066759388, "frac_reward_zero_std": 0.025, "grad_norm": 0.6962397694587708, "kl": 0.04859261708334088, "learning_rate": 1.8620445162107202e-07, "loss": 0.0019435137510299683, "reward": 3.1344559669494627, "reward_std": 0.43296399116516116, "rewards/IngredientFormatReward/mean": 0.9953125, "rewards/IngredientFormatReward/std": 0.036588290333747865, "rewards/IngredientMatchReward/mean": 0.657600736618042, "rewards/IngredientMatchReward/std": 0.2820374250411987, "rewards/IngredientQuantityMatchReward/mean": 0.6909177303314209, "rewards/IngredientQuantityMatchReward/std": 0.4004930078983307, "rewards/TotalKcalExactMatchReward/mean": 0.790625, "rewards/TotalKcalExactMatchReward/std": 0.39993174076080323, "step": 2625 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 505.4, "completions/mean_length": 393.1640625, "completions/min_length": 269.6, "epoch": 0.7315716272600834, "frac_reward_zero_std": 0.05, "grad_norm": 0.7534053325653076, "kl": 0.05059192152693868, "learning_rate": 1.84417246409709e-07, "loss": 0.0020237032324075697, "reward": 3.184044027328491, "reward_std": 0.4079024374485016, "rewards/IngredientFormatReward/mean": 0.9871875047683716, "rewards/IngredientFormatReward/std": 0.09551474601030349, "rewards/IngredientMatchReward/mean": 0.6866220116615296, "rewards/IngredientMatchReward/std": 0.28260781764984133, "rewards/IngredientQuantityMatchReward/mean": 0.7149220108985901, "rewards/IngredientQuantityMatchReward/std": 0.39453916549682616, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.3818435907363892, "step": 2630 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 513.0, "completions/mean_length": 394.4578125, "completions/min_length": 267.6, "epoch": 0.7329624478442281, "frac_reward_zero_std": 0.0875, "grad_norm": 0.5793359279632568, "kl": 0.0508285547606647, "learning_rate": 1.826367180362612e-07, "loss": 0.0020331652835011483, "reward": 3.0561100482940673, "reward_std": 0.4060769438743591, "rewards/IngredientFormatReward/mean": 0.9804538607597351, "rewards/IngredientFormatReward/std": 0.1272687464952469, "rewards/IngredientMatchReward/mean": 0.6044321060180664, "rewards/IngredientMatchReward/std": 0.2949958324432373, "rewards/IngredientQuantityMatchReward/mean": 0.67903653383255, "rewards/IngredientQuantityMatchReward/std": 0.42194249033927916, "rewards/TotalKcalExactMatchReward/mean": 0.7921875, "rewards/TotalKcalExactMatchReward/std": 0.40656169056892394, "step": 2635 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 500.6, "completions/mean_length": 390.5515625, "completions/min_length": 281.2, "epoch": 0.7343532684283728, "frac_reward_zero_std": 0.075, "grad_norm": 0.6770186424255371, "kl": 0.047754300339147446, "learning_rate": 1.8086290417166977e-07, "loss": 0.0018999837338924408, "reward": 3.0642573833465576, "reward_std": 0.3867913544178009, "rewards/IngredientFormatReward/mean": 0.9903497099876404, "rewards/IngredientFormatReward/std": 0.07810738943517208, "rewards/IngredientMatchReward/mean": 0.638651430606842, "rewards/IngredientMatchReward/std": 0.29393835067749025, "rewards/IngredientQuantityMatchReward/mean": 0.6524439454078674, "rewards/IngredientQuantityMatchReward/std": 0.4144913673400879, "rewards/TotalKcalExactMatchReward/mean": 0.7828125, "rewards/TotalKcalExactMatchReward/std": 0.4063707053661346, "step": 2640 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 496.0, "completions/mean_length": 393.4171875, "completions/min_length": 285.4, "epoch": 0.7357440890125174, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7029209136962891, "kl": 0.04800816969946027, "learning_rate": 1.790958423448159e-07, "loss": 0.0019205402582883834, "reward": 3.187349796295166, "reward_std": 0.3639358103275299, "rewards/IngredientFormatReward/mean": 0.9950000047683716, "rewards/IngredientFormatReward/std": 0.04290181696414948, "rewards/IngredientMatchReward/mean": 0.60570809841156, "rewards/IngredientMatchReward/std": 0.29033520817756653, "rewards/IngredientQuantityMatchReward/mean": 0.6882042646408081, "rewards/IngredientQuantityMatchReward/std": 0.4012244701385498, "rewards/TotalKcalExactMatchReward/mean": 0.8984375, "rewards/TotalKcalExactMatchReward/std": 0.29061916172504426, "step": 2645 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.6, "completions/mean_length": 388.65, "completions/min_length": 230.4, "epoch": 0.7371349095966621, "frac_reward_zero_std": 0.075, "grad_norm": 0.6581112742424011, "kl": 0.047632145672105254, "learning_rate": 1.7733556994172715e-07, "loss": 0.0019055701792240142, "reward": 2.9977715492248533, "reward_std": 0.3989086389541626, "rewards/IngredientFormatReward/mean": 0.9749330282211304, "rewards/IngredientFormatReward/std": 0.12201923131942749, "rewards/IngredientMatchReward/mean": 0.6350185871124268, "rewards/IngredientMatchReward/std": 0.2913230359554291, "rewards/IngredientQuantityMatchReward/mean": 0.6721949458122254, "rewards/IngredientQuantityMatchReward/std": 0.4181333541870117, "rewards/TotalKcalExactMatchReward/mean": 0.715625, "rewards/TotalKcalExactMatchReward/std": 0.4446301102638245, "step": 2650 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 512.4, "completions/mean_length": 387.428125, "completions/min_length": 267.8, "epoch": 0.7385257301808067, "frac_reward_zero_std": 0.125, "grad_norm": 0.7104362845420837, "kl": 0.046882928302511576, "learning_rate": 1.755821242047853e-07, "loss": 0.0018756043165922165, "reward": 3.117317819595337, "reward_std": 0.3763087272644043, "rewards/IngredientFormatReward/mean": 0.9822916746139526, "rewards/IngredientFormatReward/std": 0.12619206607341765, "rewards/IngredientMatchReward/mean": 0.67864009141922, "rewards/IngredientMatchReward/std": 0.2986286699771881, "rewards/IngredientQuantityMatchReward/mean": 0.6751359701156616, "rewards/IngredientQuantityMatchReward/std": 0.4090684413909912, "rewards/TotalKcalExactMatchReward/mean": 0.78125, "rewards/TotalKcalExactMatchReward/std": 0.41046456098556516, "step": 2655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 509.6, "completions/mean_length": 390.596875, "completions/min_length": 269.0, "epoch": 0.7399165507649513, "frac_reward_zero_std": 0.075, "grad_norm": 0.7171072363853455, "kl": 0.048218688229098916, "learning_rate": 1.7383554223193975e-07, "loss": 0.001929207518696785, "reward": 3.160565423965454, "reward_std": 0.3804220199584961, "rewards/IngredientFormatReward/mean": 0.9854557156562805, "rewards/IngredientFormatReward/std": 0.09512746483087539, "rewards/IngredientMatchReward/mean": 0.6397488951683045, "rewards/IngredientMatchReward/std": 0.3007284939289093, "rewards/IngredientQuantityMatchReward/mean": 0.7212984204292298, "rewards/IngredientQuantityMatchReward/std": 0.37817172408103944, "rewards/TotalKcalExactMatchReward/mean": 0.8140625, "rewards/TotalKcalExactMatchReward/std": 0.3665972471237183, "step": 2660 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 509.8, "completions/mean_length": 392.85, "completions/min_length": 271.8, "epoch": 0.741307371349096, "frac_reward_zero_std": 0.05, "grad_norm": 0.6789398789405823, "kl": 0.053108946816064415, "learning_rate": 1.7209586097592188e-07, "loss": 0.0021249458193778993, "reward": 3.0680684566497805, "reward_std": 0.3909214198589325, "rewards/IngredientFormatReward/mean": 0.9868749618530274, "rewards/IngredientFormatReward/std": 0.09705453515052795, "rewards/IngredientMatchReward/mean": 0.6739999413490295, "rewards/IngredientMatchReward/std": 0.27780568301677705, "rewards/IngredientQuantityMatchReward/mean": 0.6884435415267944, "rewards/IngredientQuantityMatchReward/std": 0.40112354755401614, "rewards/TotalKcalExactMatchReward/mean": 0.71875, "rewards/TotalKcalExactMatchReward/std": 0.44095770716667176, "step": 2665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 504.6, "completions/mean_length": 394.878125, "completions/min_length": 276.6, "epoch": 0.7426981919332406, "frac_reward_zero_std": 0.05, "grad_norm": 0.6982694268226624, "kl": 0.04645406629424542, "learning_rate": 1.7036311724346358e-07, "loss": 0.0018581092357635499, "reward": 3.106524705886841, "reward_std": 0.3474747657775879, "rewards/IngredientFormatReward/mean": 0.9910416722297668, "rewards/IngredientFormatReward/std": 0.04977382682263851, "rewards/IngredientMatchReward/mean": 0.6606696367263794, "rewards/IngredientMatchReward/std": 0.25854469537734986, "rewards/IngredientQuantityMatchReward/mean": 0.7063759326934814, "rewards/IngredientQuantityMatchReward/std": 0.39123660922050474, "rewards/TotalKcalExactMatchReward/mean": 0.7484375, "rewards/TotalKcalExactMatchReward/std": 0.42715450525283816, "step": 2670 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 505.0, "completions/mean_length": 386.7671875, "completions/min_length": 250.2, "epoch": 0.7440890125173852, "frac_reward_zero_std": 0.05, "grad_norm": 0.6369605660438538, "kl": 0.048309245309792456, "learning_rate": 1.686373476945182e-07, "loss": 0.00193245280534029, "reward": 3.176955461502075, "reward_std": 0.39188035726547243, "rewards/IngredientFormatReward/mean": 0.9866666555404663, "rewards/IngredientFormatReward/std": 0.08516232818365096, "rewards/IngredientMatchReward/mean": 0.6687934160232544, "rewards/IngredientMatchReward/std": 0.28387183248996734, "rewards/IngredientQuantityMatchReward/mean": 0.6886829137802124, "rewards/IngredientQuantityMatchReward/std": 0.3933670938014984, "rewards/TotalKcalExactMatchReward/mean": 0.8328125, "rewards/TotalKcalExactMatchReward/std": 0.36141787767410277, "step": 2675 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 509.8, "completions/mean_length": 393.225, "completions/min_length": 255.0, "epoch": 0.7454798331015299, "frac_reward_zero_std": 0.05, "grad_norm": 0.6349705457687378, "kl": 0.051582214632071556, "learning_rate": 1.6691858884148525e-07, "loss": 0.0020633328706026076, "reward": 3.123720407485962, "reward_std": 0.38062790036201477, "rewards/IngredientFormatReward/mean": 0.9845535635948182, "rewards/IngredientFormatReward/std": 0.0967326819896698, "rewards/IngredientMatchReward/mean": 0.6521354436874389, "rewards/IngredientMatchReward/std": 0.28357810974121095, "rewards/IngredientQuantityMatchReward/mean": 0.7073439955711365, "rewards/IngredientQuantityMatchReward/std": 0.3868045091629028, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.40714595913887025, "step": 2680 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 499.2, "completions/mean_length": 392.975, "completions/min_length": 281.0, "epoch": 0.7468706536856745, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6418431401252747, "kl": 0.04696134240366519, "learning_rate": 1.6520687704843761e-07, "loss": 0.001878587156534195, "reward": 3.185035562515259, "reward_std": 0.4062525689601898, "rewards/IngredientFormatReward/mean": 0.9868750095367431, "rewards/IngredientFormatReward/std": 0.07441277205944061, "rewards/IngredientMatchReward/mean": 0.6845523357391358, "rewards/IngredientMatchReward/std": 0.2672650098800659, "rewards/IngredientQuantityMatchReward/mean": 0.7026707291603088, "rewards/IngredientQuantityMatchReward/std": 0.41382738947868347, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.38031128644943235, "step": 2685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 507.0, "completions/mean_length": 392.071875, "completions/min_length": 270.2, "epoch": 0.7482614742698191, "frac_reward_zero_std": 0.075, "grad_norm": 0.6632663607597351, "kl": 0.04900345194619149, "learning_rate": 1.6350224853035266e-07, "loss": 0.0019601194187998773, "reward": 3.096248960494995, "reward_std": 0.3981503129005432, "rewards/IngredientFormatReward/mean": 0.9906473159790039, "rewards/IngredientFormatReward/std": 0.07666670083999634, "rewards/IngredientMatchReward/mean": 0.6439918160438538, "rewards/IngredientMatchReward/std": 0.2905239939689636, "rewards/IngredientQuantityMatchReward/mean": 0.6991097688674927, "rewards/IngredientQuantityMatchReward/std": 0.41301283836364744, "rewards/TotalKcalExactMatchReward/mean": 0.7625, "rewards/TotalKcalExactMatchReward/std": 0.3906921327114105, "step": 2690 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 487.2, "completions/mean_length": 384.54375, "completions/min_length": 256.6, "epoch": 0.7496522948539638, "frac_reward_zero_std": 0.025, "grad_norm": 0.6603626608848572, "kl": 0.046728350967168805, "learning_rate": 1.6180473935234508e-07, "loss": 0.001869351789355278, "reward": 3.1564847946166994, "reward_std": 0.4320804297924042, "rewards/IngredientFormatReward/mean": 0.9943750143051148, "rewards/IngredientFormatReward/std": 0.057214077562093735, "rewards/IngredientMatchReward/mean": 0.663671863079071, "rewards/IngredientMatchReward/std": 0.287596070766449, "rewards/IngredientQuantityMatchReward/mean": 0.6671879291534424, "rewards/IngredientQuantityMatchReward/std": 0.4132254719734192, "rewards/TotalKcalExactMatchReward/mean": 0.83125, "rewards/TotalKcalExactMatchReward/std": 0.367131644487381, "step": 2695 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 512.6, "completions/mean_length": 393.4859375, "completions/min_length": 254.0, "epoch": 0.7510431154381085, "frac_reward_zero_std": 0.1, "grad_norm": 0.6157970428466797, "kl": 0.04571370454505086, "learning_rate": 1.6011438542890481e-07, "loss": 0.0018287433311343193, "reward": 3.0726105690002443, "reward_std": 0.3633168041706085, "rewards/IngredientFormatReward/mean": 0.9866666674613953, "rewards/IngredientFormatReward/std": 0.10891215056180954, "rewards/IngredientMatchReward/mean": 0.6230196475982666, "rewards/IngredientMatchReward/std": 0.29381694197654723, "rewards/IngredientQuantityMatchReward/mean": 0.6301117181777954, "rewards/IngredientQuantityMatchReward/std": 0.4268217861652374, "rewards/TotalKcalExactMatchReward/mean": 0.8328125, "rewards/TotalKcalExactMatchReward/std": 0.3544869065284729, "step": 2700 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 502.0, "completions/mean_length": 390.8546875, "completions/min_length": 276.2, "epoch": 0.7524339360222532, "frac_reward_zero_std": 0.1125, "grad_norm": 0.6334781646728516, "kl": 0.04829642854165286, "learning_rate": 1.584312225231373e-07, "loss": 0.0019318748265504837, "reward": 3.1972563743591307, "reward_std": 0.3812137722969055, "rewards/IngredientFormatReward/mean": 0.9905580282211304, "rewards/IngredientFormatReward/std": 0.06447032019495964, "rewards/IngredientMatchReward/mean": 0.6358643293380737, "rewards/IngredientMatchReward/std": 0.29064218103885653, "rewards/IngredientQuantityMatchReward/mean": 0.6880215525627136, "rewards/IngredientQuantityMatchReward/std": 0.41757785677909853, "rewards/TotalKcalExactMatchReward/mean": 0.8828125, "rewards/TotalKcalExactMatchReward/std": 0.2939798399806023, "step": 2705 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 508.6, "completions/mean_length": 387.80625, "completions/min_length": 260.2, "epoch": 0.7538247566063978, "frac_reward_zero_std": 0.0125, "grad_norm": 0.662911057472229, "kl": 0.04823250975459814, "learning_rate": 1.5675528624600598e-07, "loss": 0.0019293729215860366, "reward": 3.137434196472168, "reward_std": 0.4250971019268036, "rewards/IngredientFormatReward/mean": 0.9915624976158142, "rewards/IngredientFormatReward/std": 0.06401846744120121, "rewards/IngredientMatchReward/mean": 0.6299597263336182, "rewards/IngredientMatchReward/std": 0.2888274848461151, "rewards/IngredientQuantityMatchReward/mean": 0.6893495202064515, "rewards/IngredientQuantityMatchReward/std": 0.39977465867996215, "rewards/TotalKcalExactMatchReward/mean": 0.8265625, "rewards/TotalKcalExactMatchReward/std": 0.3759679079055786, "step": 2710 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 505.2, "completions/mean_length": 386.7421875, "completions/min_length": 259.0, "epoch": 0.7552155771905424, "frac_reward_zero_std": 0.1, "grad_norm": 0.6722151637077332, "kl": 0.046527455444447696, "learning_rate": 1.5508661205557898e-07, "loss": 0.0018611792474985123, "reward": 3.195002794265747, "reward_std": 0.3524541914463043, "rewards/IngredientFormatReward/mean": 0.9923214316368103, "rewards/IngredientFormatReward/std": 0.06965172737836837, "rewards/IngredientMatchReward/mean": 0.6588944554328918, "rewards/IngredientMatchReward/std": 0.2890048623085022, "rewards/IngredientQuantityMatchReward/mean": 0.7062869071960449, "rewards/IngredientQuantityMatchReward/std": 0.41538079380989074, "rewards/TotalKcalExactMatchReward/mean": 0.8375, "rewards/TotalKcalExactMatchReward/std": 0.3644015729427338, "step": 2715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 503.6, "completions/mean_length": 388.00625, "completions/min_length": 251.0, "epoch": 0.7566063977746871, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7072897553443909, "kl": 0.043907552887685594, "learning_rate": 1.534252352562797e-07, "loss": 0.0017563527449965476, "reward": 3.0336654663085936, "reward_std": 0.37396565079689026, "rewards/IngredientFormatReward/mean": 0.996875, "rewards/IngredientFormatReward/std": 0.03535533845424652, "rewards/IngredientMatchReward/mean": 0.6149045109748841, "rewards/IngredientMatchReward/std": 0.30236271023750305, "rewards/IngredientQuantityMatchReward/mean": 0.6406359076499939, "rewards/IngredientQuantityMatchReward/std": 0.42598314881324767, "rewards/TotalKcalExactMatchReward/mean": 0.78125, "rewards/TotalKcalExactMatchReward/std": 0.38697641491889956, "step": 2720 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 511.8, "completions/mean_length": 391.503125, "completions/min_length": 258.2, "epoch": 0.7579972183588317, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6333484053611755, "kl": 0.04943839656189084, "learning_rate": 1.5177119099813922e-07, "loss": 0.0019777391105890276, "reward": 3.254866600036621, "reward_std": 0.36433430314064025, "rewards/IngredientFormatReward/mean": 0.9916666626930237, "rewards/IngredientFormatReward/std": 0.06988214328885078, "rewards/IngredientMatchReward/mean": 0.716341781616211, "rewards/IngredientMatchReward/std": 0.2796147793531418, "rewards/IngredientQuantityMatchReward/mean": 0.7327957391738892, "rewards/IngredientQuantityMatchReward/std": 0.37077197432518005, "rewards/TotalKcalExactMatchReward/mean": 0.8140625, "rewards/TotalKcalExactMatchReward/std": 0.386597740650177, "step": 2725 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 504.8, "completions/mean_length": 385.6921875, "completions/min_length": 246.0, "epoch": 0.7593880389429764, "frac_reward_zero_std": 0.0625, "grad_norm": 0.669300377368927, "kl": 0.04992357234004885, "learning_rate": 1.5012451427605295e-07, "loss": 0.0019968684762716295, "reward": 3.1358933448791504, "reward_std": 0.3479516267776489, "rewards/IngredientFormatReward/mean": 0.9936979055404663, "rewards/IngredientFormatReward/std": 0.057158236391842365, "rewards/IngredientMatchReward/mean": 0.6550303936004639, "rewards/IngredientMatchReward/std": 0.2722333997488022, "rewards/IngredientQuantityMatchReward/mean": 0.659040093421936, "rewards/IngredientQuantityMatchReward/std": 0.4142662465572357, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.3717784404754639, "step": 2730 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 508.6, "completions/mean_length": 389.209375, "completions/min_length": 242.4, "epoch": 0.760778859527121, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6446650624275208, "kl": 0.04634713255800307, "learning_rate": 1.4848523992903978e-07, "loss": 0.0018548401072621345, "reward": 3.045489501953125, "reward_std": 0.43291773796081545, "rewards/IngredientFormatReward/mean": 0.9876562476158142, "rewards/IngredientFormatReward/std": 0.09445627927780151, "rewards/IngredientMatchReward/mean": 0.6676263451576233, "rewards/IngredientMatchReward/std": 0.2896029084920883, "rewards/IngredientQuantityMatchReward/mean": 0.6136444568634033, "rewards/IngredientQuantityMatchReward/std": 0.4410985291004181, "rewards/TotalKcalExactMatchReward/mean": 0.7765625, "rewards/TotalKcalExactMatchReward/std": 0.4121951341629028, "step": 2735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 507.2, "completions/mean_length": 395.109375, "completions/min_length": 274.0, "epoch": 0.7621696801112656, "frac_reward_zero_std": 0.075, "grad_norm": 0.6219557523727417, "kl": 0.045795883238315585, "learning_rate": 1.468534026395056e-07, "loss": 0.0018320389091968537, "reward": 3.003160810470581, "reward_std": 0.3470380902290344, "rewards/IngredientFormatReward/mean": 0.9934375047683716, "rewards/IngredientFormatReward/std": 0.0515897773206234, "rewards/IngredientMatchReward/mean": 0.6431274652481079, "rewards/IngredientMatchReward/std": 0.2938917726278305, "rewards/IngredientQuantityMatchReward/mean": 0.5962833166122437, "rewards/IngredientQuantityMatchReward/std": 0.43143598437309266, "rewards/TotalKcalExactMatchReward/mean": 0.7703125, "rewards/TotalKcalExactMatchReward/std": 0.4162842929363251, "step": 2740 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 511.2, "completions/mean_length": 387.2671875, "completions/min_length": 268.6, "epoch": 0.7635605006954103, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6196995377540588, "kl": 0.046195710496976974, "learning_rate": 1.4522903693250904e-07, "loss": 0.0018479309976100922, "reward": 3.1460134029388427, "reward_std": 0.4049987256526947, "rewards/IngredientFormatReward/mean": 0.9915364503860473, "rewards/IngredientFormatReward/std": 0.07865389734506607, "rewards/IngredientMatchReward/mean": 0.6382266998291015, "rewards/IngredientMatchReward/std": 0.29503755569458007, "rewards/IngredientQuantityMatchReward/mean": 0.7006252765655517, "rewards/IngredientQuantityMatchReward/std": 0.3964582920074463, "rewards/TotalKcalExactMatchReward/mean": 0.815625, "rewards/TotalKcalExactMatchReward/std": 0.3763843536376953, "step": 2745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 502.8, "completions/mean_length": 386.3515625, "completions/min_length": 249.8, "epoch": 0.7649513212795549, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6867331862449646, "kl": 0.04726094771176577, "learning_rate": 1.4361217717503143e-07, "loss": 0.0018905647099018096, "reward": 3.0409469604492188, "reward_std": 0.36254254579544065, "rewards/IngredientFormatReward/mean": 0.996875, "rewards/IngredientFormatReward/std": 0.02490137964487076, "rewards/IngredientMatchReward/mean": 0.5823530673980712, "rewards/IngredientMatchReward/std": 0.313857102394104, "rewards/IngredientQuantityMatchReward/mean": 0.6867189168930053, "rewards/IngredientQuantityMatchReward/std": 0.4137806832790375, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.41760230660438535, "step": 2750 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 504.2, "completions/mean_length": 400.6734375, "completions/min_length": 285.4, "epoch": 0.7663421418636995, "frac_reward_zero_std": 0.05, "grad_norm": 0.6223328113555908, "kl": 0.048341109650209546, "learning_rate": 1.4200285757524894e-07, "loss": 0.0019336594268679619, "reward": 3.2103977680206297, "reward_std": 0.3802697777748108, "rewards/IngredientFormatReward/mean": 0.9953125, "rewards/IngredientFormatReward/std": 0.04257904887199402, "rewards/IngredientMatchReward/mean": 0.6562047481536866, "rewards/IngredientMatchReward/std": 0.2683935105800629, "rewards/IngredientQuantityMatchReward/mean": 0.7104430794715881, "rewards/IngredientQuantityMatchReward/std": 0.39671446681022643, "rewards/TotalKcalExactMatchReward/mean": 0.8484375, "rewards/TotalKcalExactMatchReward/std": 0.35886216163635254, "step": 2755 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 500.0, "completions/mean_length": 388.2171875, "completions/min_length": 250.4, "epoch": 0.7677329624478443, "frac_reward_zero_std": 0.075, "grad_norm": 0.6476879715919495, "kl": 0.04689363501966, "learning_rate": 1.4040111218180966e-07, "loss": 0.0018757812678813935, "reward": 3.1677744388580322, "reward_std": 0.34500511884689333, "rewards/IngredientFormatReward/mean": 0.9977306485176086, "rewards/IngredientFormatReward/std": 0.021293053217232228, "rewards/IngredientMatchReward/mean": 0.6576048016548157, "rewards/IngredientMatchReward/std": 0.27767792642116546, "rewards/IngredientQuantityMatchReward/mean": 0.6718139767646789, "rewards/IngredientQuantityMatchReward/std": 0.40898163318634034, "rewards/TotalKcalExactMatchReward/mean": 0.840625, "rewards/TotalKcalExactMatchReward/std": 0.3573881030082703, "step": 2760 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 500.2, "completions/mean_length": 386.4, "completions/min_length": 268.8, "epoch": 0.7691237830319889, "frac_reward_zero_std": 0.15, "grad_norm": 0.6938906908035278, "kl": 0.046820966992527246, "learning_rate": 1.3880697488311326e-07, "loss": 0.0018731381744146347, "reward": 3.158130073547363, "reward_std": 0.3146629989147186, "rewards/IngredientFormatReward/mean": 0.9966257452964783, "rewards/IngredientFormatReward/std": 0.03063418995589018, "rewards/IngredientMatchReward/mean": 0.658064866065979, "rewards/IngredientMatchReward/std": 0.2844700157642365, "rewards/IngredientQuantityMatchReward/mean": 0.6815645098686218, "rewards/IngredientQuantityMatchReward/std": 0.40854601860046386, "rewards/TotalKcalExactMatchReward/mean": 0.821875, "rewards/TotalKcalExactMatchReward/std": 0.33308517932891846, "step": 2765 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 507.6, "completions/mean_length": 389.0953125, "completions/min_length": 279.4, "epoch": 0.7705146036161336, "frac_reward_zero_std": 0.0125, "grad_norm": 0.6388007402420044, "kl": 0.04836219367571175, "learning_rate": 1.3722047940659326e-07, "loss": 0.0019346587359905243, "reward": 3.077744960784912, "reward_std": 0.3748651802539825, "rewards/IngredientFormatReward/mean": 0.9966145753860474, "rewards/IngredientFormatReward/std": 0.03830161709338427, "rewards/IngredientMatchReward/mean": 0.6705307602882385, "rewards/IngredientMatchReward/std": 0.25971758365631104, "rewards/IngredientQuantityMatchReward/mean": 0.6543496251106262, "rewards/IngredientQuantityMatchReward/std": 0.4099151134490967, "rewards/TotalKcalExactMatchReward/mean": 0.75625, "rewards/TotalKcalExactMatchReward/std": 0.4248959481716156, "step": 2770 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 496.8, "completions/mean_length": 384.6, "completions/min_length": 256.0, "epoch": 0.7719054242002782, "frac_reward_zero_std": 0.05, "grad_norm": 0.7773398160934448, "kl": 0.05296712294220925, "learning_rate": 1.356416593180036e-07, "loss": 0.0021185621619224547, "reward": 3.0785492420196534, "reward_std": 0.3721375286579132, "rewards/IngredientFormatReward/mean": 0.9884579658508301, "rewards/IngredientFormatReward/std": 0.06780673321336508, "rewards/IngredientMatchReward/mean": 0.6230674982070923, "rewards/IngredientMatchReward/std": 0.2801555782556534, "rewards/IngredientQuantityMatchReward/mean": 0.6560862421989441, "rewards/IngredientQuantityMatchReward/std": 0.39724615812301634, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.38468171954154967, "step": 2775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 506.8, "completions/mean_length": 390.98125, "completions/min_length": 266.4, "epoch": 0.7732962447844228, "frac_reward_zero_std": 0.1, "grad_norm": 0.6303114891052246, "kl": 0.048468559980392456, "learning_rate": 1.3407054802070923e-07, "loss": 0.0019390463829040527, "reward": 3.252069044113159, "reward_std": 0.3501974791288376, "rewards/IngredientFormatReward/mean": 0.983392858505249, "rewards/IngredientFormatReward/std": 0.09273147732019424, "rewards/IngredientMatchReward/mean": 0.7026979684829712, "rewards/IngredientMatchReward/std": 0.2675772845745087, "rewards/IngredientQuantityMatchReward/mean": 0.7081657171249389, "rewards/IngredientQuantityMatchReward/std": 0.37891228795051574, "rewards/TotalKcalExactMatchReward/mean": 0.8578125, "rewards/TotalKcalExactMatchReward/std": 0.34481988549232484, "step": 2780 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 513.0, "completions/mean_length": 393.4015625, "completions/min_length": 269.6, "epoch": 0.7746870653685675, "frac_reward_zero_std": 0.05, "grad_norm": 0.7151004076004028, "kl": 0.04872276636306196, "learning_rate": 1.3250717875497864e-07, "loss": 0.0019489120692014693, "reward": 3.0865904331207275, "reward_std": 0.42741536498069765, "rewards/IngredientFormatReward/mean": 0.9932663679122925, "rewards/IngredientFormatReward/std": 0.0710913971066475, "rewards/IngredientMatchReward/mean": 0.6273127555847168, "rewards/IngredientMatchReward/std": 0.29151912927627566, "rewards/IngredientQuantityMatchReward/mean": 0.6535112500190735, "rewards/IngredientQuantityMatchReward/std": 0.4204251229763031, "rewards/TotalKcalExactMatchReward/mean": 0.8125, "rewards/TotalKcalExactMatchReward/std": 0.38339470624923705, "step": 2785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 510.8, "completions/mean_length": 385.0453125, "completions/min_length": 271.8, "epoch": 0.7760778859527121, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6237897276878357, "kl": 0.04654221590608358, "learning_rate": 1.3095158459728088e-07, "loss": 0.0018617745488882064, "reward": 3.0792617797851562, "reward_std": 0.37338990569114683, "rewards/IngredientFormatReward/mean": 0.993359375, "rewards/IngredientFormatReward/std": 0.07122094184160233, "rewards/IngredientMatchReward/mean": 0.680286455154419, "rewards/IngredientMatchReward/std": 0.2871041715145111, "rewards/IngredientQuantityMatchReward/mean": 0.6446785092353821, "rewards/IngredientQuantityMatchReward/std": 0.39799618124961855, "rewards/TotalKcalExactMatchReward/mean": 0.7609375, "rewards/TotalKcalExactMatchReward/std": 0.4222764551639557, "step": 2790 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 504.4, "completions/mean_length": 385.3328125, "completions/min_length": 267.0, "epoch": 0.7774687065368567, "frac_reward_zero_std": 0.075, "grad_norm": 0.6749888062477112, "kl": 0.06970281926915049, "learning_rate": 1.2940379845958588e-07, "loss": 0.0027874993160367013, "reward": 3.0723299026489257, "reward_std": 0.42997482419013977, "rewards/IngredientFormatReward/mean": 0.9851785659790039, "rewards/IngredientFormatReward/std": 0.10947459004819393, "rewards/IngredientMatchReward/mean": 0.621602201461792, "rewards/IngredientMatchReward/std": 0.2980756342411041, "rewards/IngredientQuantityMatchReward/mean": 0.6827366828918457, "rewards/IngredientQuantityMatchReward/std": 0.42669169306755067, "rewards/TotalKcalExactMatchReward/mean": 0.7828125, "rewards/TotalKcalExactMatchReward/std": 0.39410466253757476, "step": 2795 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 509.6, "completions/mean_length": 390.4203125, "completions/min_length": 274.2, "epoch": 0.7788595271210014, "frac_reward_zero_std": 0.075, "grad_norm": 0.6576166152954102, "kl": 0.04848268665373325, "learning_rate": 1.278638530886677e-07, "loss": 0.00193945299834013, "reward": 3.1979696273803713, "reward_std": 0.3592884033918381, "rewards/IngredientFormatReward/mean": 0.9953125, "rewards/IngredientFormatReward/std": 0.04257904887199402, "rewards/IngredientMatchReward/mean": 0.6500452756881714, "rewards/IngredientMatchReward/std": 0.2857871323823929, "rewards/IngredientQuantityMatchReward/mean": 0.6838618040084838, "rewards/IngredientQuantityMatchReward/std": 0.4050181329250336, "rewards/TotalKcalExactMatchReward/mean": 0.86875, "rewards/TotalKcalExactMatchReward/std": 0.31754682660102845, "step": 2800 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 506.2, "completions/mean_length": 387.875, "completions/min_length": 247.4, "epoch": 0.780250347705146, "frac_reward_zero_std": 0.075, "grad_norm": 0.7045865654945374, "kl": 0.045610047527588904, "learning_rate": 1.2633178106541214e-07, "loss": 0.0018244288861751557, "reward": 3.0865010738372805, "reward_std": 0.40377230048179624, "rewards/IngredientFormatReward/mean": 0.9901041626930237, "rewards/IngredientFormatReward/std": 0.05828761123120785, "rewards/IngredientMatchReward/mean": 0.6850093126296997, "rewards/IngredientMatchReward/std": 0.2759521782398224, "rewards/IngredientQuantityMatchReward/mean": 0.623887574672699, "rewards/IngredientQuantityMatchReward/std": 0.410383015871048, "rewards/TotalKcalExactMatchReward/mean": 0.7875, "rewards/TotalKcalExactMatchReward/std": 0.406889545917511, "step": 2805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 488.6, "completions/mean_length": 382.325, "completions/min_length": 264.8, "epoch": 0.7816411682892906, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7042024731636047, "kl": 0.04793581950943917, "learning_rate": 1.2480761480412738e-07, "loss": 0.0019172405824065208, "reward": 3.1829829692840574, "reward_std": 0.35909828543663025, "rewards/IngredientFormatReward/mean": 0.9947916746139527, "rewards/IngredientFormatReward/std": 0.05394516885280609, "rewards/IngredientMatchReward/mean": 0.6396100044250488, "rewards/IngredientMatchReward/std": 0.29437714219093325, "rewards/IngredientQuantityMatchReward/mean": 0.7001438021659852, "rewards/IngredientQuantityMatchReward/std": 0.4026173710823059, "rewards/TotalKcalExactMatchReward/mean": 0.8484375, "rewards/TotalKcalExactMatchReward/std": 0.3493222385644913, "step": 2810 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 508.2, "completions/mean_length": 388.528125, "completions/min_length": 253.4, "epoch": 0.7830319888734353, "frac_reward_zero_std": 0.05, "grad_norm": 0.6657385230064392, "kl": 0.049304631818085906, "learning_rate": 1.2329138655185733e-07, "loss": 0.0019720766693353655, "reward": 3.040740966796875, "reward_std": 0.4243003189563751, "rewards/IngredientFormatReward/mean": 0.9796316981315613, "rewards/IngredientFormatReward/std": 0.10241263806819915, "rewards/IngredientMatchReward/mean": 0.6618905305862427, "rewards/IngredientMatchReward/std": 0.31010615825653076, "rewards/IngredientQuantityMatchReward/mean": 0.6554687976837158, "rewards/IngredientQuantityMatchReward/std": 0.43676078915596006, "rewards/TotalKcalExactMatchReward/mean": 0.74375, "rewards/TotalKcalExactMatchReward/std": 0.4280146896839142, "step": 2815 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 508.4, "completions/mean_length": 390.76875, "completions/min_length": 253.0, "epoch": 0.7844228094575799, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6594431400299072, "kl": 0.054166854242794216, "learning_rate": 1.2178312838770115e-07, "loss": 0.002167240343987942, "reward": 3.222614049911499, "reward_std": 0.3778720319271088, "rewards/IngredientFormatReward/mean": 0.9932291746139527, "rewards/IngredientFormatReward/std": 0.07162283807992935, "rewards/IngredientMatchReward/mean": 0.6805189847946167, "rewards/IngredientMatchReward/std": 0.280741947889328, "rewards/IngredientQuantityMatchReward/mean": 0.6848033666610718, "rewards/IngredientQuantityMatchReward/std": 0.38373754024505613, "rewards/TotalKcalExactMatchReward/mean": 0.8640625, "rewards/TotalKcalExactMatchReward/std": 0.2921728253364563, "step": 2820 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 502.8, "completions/mean_length": 395.3796875, "completions/min_length": 272.4, "epoch": 0.7858136300417247, "frac_reward_zero_std": 0.075, "grad_norm": 0.6570203304290771, "kl": 0.04562732768245041, "learning_rate": 1.2028287222213286e-07, "loss": 0.0018250590190291404, "reward": 3.0641520023345947, "reward_std": 0.3975339472293854, "rewards/IngredientFormatReward/mean": 0.989661455154419, "rewards/IngredientFormatReward/std": 0.06459243036806583, "rewards/IngredientMatchReward/mean": 0.6596819043159485, "rewards/IngredientMatchReward/std": 0.2889078378677368, "rewards/IngredientQuantityMatchReward/mean": 0.6054336071014405, "rewards/IngredientQuantityMatchReward/std": 0.4233911633491516, "rewards/TotalKcalExactMatchReward/mean": 0.809375, "rewards/TotalKcalExactMatchReward/std": 0.3768185257911682, "step": 2825 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 497.8, "completions/mean_length": 388.7671875, "completions/min_length": 262.2, "epoch": 0.7872044506258693, "frac_reward_zero_std": 0.0375, "grad_norm": 2.9295973777770996, "kl": 0.06034769716206938, "learning_rate": 1.1879064979632702e-07, "loss": 0.0024134255945682526, "reward": 2.941163158416748, "reward_std": 0.3750101804733276, "rewards/IngredientFormatReward/mean": 0.9932291746139527, "rewards/IngredientFormatReward/std": 0.04726078063249588, "rewards/IngredientMatchReward/mean": 0.5866058945655823, "rewards/IngredientMatchReward/std": 0.3109011113643646, "rewards/IngredientQuantityMatchReward/mean": 0.6300781726837158, "rewards/IngredientQuantityMatchReward/std": 0.42629345059394835, "rewards/TotalKcalExactMatchReward/mean": 0.73125, "rewards/TotalKcalExactMatchReward/std": 0.43264933228492736, "step": 2830 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 504.4, "completions/mean_length": 386.409375, "completions/min_length": 260.8, "epoch": 0.7885952712100139, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6922607421875, "kl": 0.04646100748796016, "learning_rate": 1.1730649268148662e-07, "loss": 0.0018586434423923492, "reward": 2.9923593521118166, "reward_std": 0.43100237250328066, "rewards/IngredientFormatReward/mean": 0.9923437356948852, "rewards/IngredientFormatReward/std": 0.05373183339834213, "rewards/IngredientMatchReward/mean": 0.6438275218009949, "rewards/IngredientMatchReward/std": 0.27490496039390566, "rewards/IngredientQuantityMatchReward/mean": 0.6374380946159363, "rewards/IngredientQuantityMatchReward/std": 0.41457298398017883, "rewards/TotalKcalExactMatchReward/mean": 0.71875, "rewards/TotalKcalExactMatchReward/std": 0.4473869621753693, "step": 2835 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 505.8, "completions/mean_length": 385.346875, "completions/min_length": 264.2, "epoch": 0.7899860917941586, "frac_reward_zero_std": 0.075, "grad_norm": 0.5764864683151245, "kl": 0.06799533735029399, "learning_rate": 1.1583043227817607e-07, "loss": 0.0027182653546333315, "reward": 3.036129665374756, "reward_std": 0.4375645399093628, "rewards/IngredientFormatReward/mean": 0.985531997680664, "rewards/IngredientFormatReward/std": 0.0772362755611539, "rewards/IngredientMatchReward/mean": 0.6238765120506287, "rewards/IngredientMatchReward/std": 0.3162325620651245, "rewards/IngredientQuantityMatchReward/mean": 0.6470335960388184, "rewards/IngredientQuantityMatchReward/std": 0.4036581993103027, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.41052433252334597, "step": 2840 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 501.6, "completions/mean_length": 382.0109375, "completions/min_length": 247.8, "epoch": 0.7913769123783032, "frac_reward_zero_std": 0.075, "grad_norm": 0.6719501614570618, "kl": 0.046000178135000167, "learning_rate": 1.1436249981565576e-07, "loss": 0.0018522294238209724, "reward": 3.1666172981262206, "reward_std": 0.3686310827732086, "rewards/IngredientFormatReward/mean": 0.9954166531562805, "rewards/IngredientFormatReward/std": 0.047038369067013266, "rewards/IngredientMatchReward/mean": 0.6741952061653137, "rewards/IngredientMatchReward/std": 0.2847984731197357, "rewards/IngredientQuantityMatchReward/mean": 0.7157554388046264, "rewards/IngredientQuantityMatchReward/std": 0.4073028028011322, "rewards/TotalKcalExactMatchReward/mean": 0.78125, "rewards/TotalKcalExactMatchReward/std": 0.40991033911705016, "step": 2845 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 503.0, "completions/mean_length": 388.4515625, "completions/min_length": 243.4, "epoch": 0.7927677329624478, "frac_reward_zero_std": 0.0125, "grad_norm": 0.672032356262207, "kl": 0.04769486370496452, "learning_rate": 1.1290272635122256e-07, "loss": 0.0019078360870480537, "reward": 2.9494760036468506, "reward_std": 0.3965733826160431, "rewards/IngredientFormatReward/mean": 0.9865625023841857, "rewards/IngredientFormatReward/std": 0.08309870734810829, "rewards/IngredientMatchReward/mean": 0.5798406481742859, "rewards/IngredientMatchReward/std": 0.2907747864723206, "rewards/IngredientQuantityMatchReward/mean": 0.6393229246139527, "rewards/IngredientQuantityMatchReward/std": 0.42374877333641053, "rewards/TotalKcalExactMatchReward/mean": 0.74375, "rewards/TotalKcalExactMatchReward/std": 0.41585314869880674, "step": 2850 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 503.6, "completions/mean_length": 388.7546875, "completions/min_length": 252.8, "epoch": 0.7941585535465925, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6961467266082764, "kl": 0.048885095841251315, "learning_rate": 1.114511427695512e-07, "loss": 0.0019551247358322145, "reward": 3.0520843505859374, "reward_std": 0.45254541039466856, "rewards/IngredientFormatReward/mean": 0.9900892972946167, "rewards/IngredientFormatReward/std": 0.07334856688976288, "rewards/IngredientMatchReward/mean": 0.6228714346885681, "rewards/IngredientMatchReward/std": 0.2974251747131348, "rewards/IngredientQuantityMatchReward/mean": 0.6313111305236816, "rewards/IngredientQuantityMatchReward/std": 0.42033830285072327, "rewards/TotalKcalExactMatchReward/mean": 0.8078125, "rewards/TotalKcalExactMatchReward/std": 0.39325725436210635, "step": 2855 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 504.8, "completions/mean_length": 387.7484375, "completions/min_length": 264.2, "epoch": 0.7955493741307371, "frac_reward_zero_std": 0.0875, "grad_norm": 0.7616549134254456, "kl": 0.04823335437104106, "learning_rate": 1.1000777978204211e-07, "loss": 0.0019295023754239081, "reward": 3.0298017501831054, "reward_std": 0.38049039244651794, "rewards/IngredientFormatReward/mean": 0.9914434432983399, "rewards/IngredientFormatReward/std": 0.0785360500216484, "rewards/IngredientMatchReward/mean": 0.6197124242782592, "rewards/IngredientMatchReward/std": 0.2868492126464844, "rewards/IngredientQuantityMatchReward/mean": 0.6389583468437194, "rewards/IngredientQuantityMatchReward/std": 0.42402172088623047, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.40851550102233886, "step": 2860 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 505.8, "completions/mean_length": 393.36875, "completions/min_length": 271.6, "epoch": 0.7969401947148818, "frac_reward_zero_std": 0.0625, "grad_norm": 0.647112250328064, "kl": 0.045586750074289736, "learning_rate": 1.0857266792617121e-07, "loss": 0.0018235519528388977, "reward": 3.1301424980163572, "reward_std": 0.36510835886001586, "rewards/IngredientFormatReward/mean": 0.9830729126930237, "rewards/IngredientFormatReward/std": 0.09126228243112564, "rewards/IngredientMatchReward/mean": 0.6374433040618896, "rewards/IngredientMatchReward/std": 0.30091888308525083, "rewards/IngredientQuantityMatchReward/mean": 0.6236887693405151, "rewards/IngredientQuantityMatchReward/std": 0.42500794529914854, "rewards/TotalKcalExactMatchReward/mean": 0.8859375, "rewards/TotalKcalExactMatchReward/std": 0.30419874787330625, "step": 2865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 494.6, "completions/mean_length": 386.471875, "completions/min_length": 238.8, "epoch": 0.7983310152990264, "frac_reward_zero_std": 0.0375, "grad_norm": 0.665838897228241, "kl": 0.0487892878241837, "learning_rate": 1.071458375648438e-07, "loss": 0.001963186264038086, "reward": 3.08637056350708, "reward_std": 0.36597791910171507, "rewards/IngredientFormatReward/mean": 0.9940885305404663, "rewards/IngredientFormatReward/std": 0.06024602744728327, "rewards/IngredientMatchReward/mean": 0.6441040515899659, "rewards/IngredientMatchReward/std": 0.2958334803581238, "rewards/IngredientQuantityMatchReward/mean": 0.6200530052185058, "rewards/IngredientQuantityMatchReward/std": 0.4220093309879303, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.33209551125764847, "step": 2870 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 511.0, "completions/mean_length": 393.8234375, "completions/min_length": 266.0, "epoch": 0.799721835883171, "frac_reward_zero_std": 0.075, "grad_norm": 0.6649406552314758, "kl": 0.04649213703814894, "learning_rate": 1.0572731888575209e-07, "loss": 0.0018598034977912903, "reward": 3.0285863399505617, "reward_std": 0.40862372517585754, "rewards/IngredientFormatReward/mean": 0.9753236532211303, "rewards/IngredientFormatReward/std": 0.1264018550515175, "rewards/IngredientMatchReward/mean": 0.6302784562110901, "rewards/IngredientMatchReward/std": 0.30015496611595155, "rewards/IngredientQuantityMatchReward/mean": 0.65423424243927, "rewards/IngredientQuantityMatchReward/std": 0.4166111648082733, "rewards/TotalKcalExactMatchReward/mean": 0.76875, "rewards/TotalKcalExactMatchReward/std": 0.4195781469345093, "step": 2875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 506.0, "completions/mean_length": 389.603125, "completions/min_length": 238.4, "epoch": 0.8011126564673157, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6201867461204529, "kl": 0.046519194776192305, "learning_rate": 1.043171419007367e-07, "loss": 0.0018607662990689279, "reward": 3.052107810974121, "reward_std": 0.4012625515460968, "rewards/IngredientFormatReward/mean": 0.9785788774490356, "rewards/IngredientFormatReward/std": 0.0843801498413086, "rewards/IngredientMatchReward/mean": 0.6381122231483459, "rewards/IngredientMatchReward/std": 0.30227981209754945, "rewards/IngredientQuantityMatchReward/mean": 0.6541667699813842, "rewards/IngredientQuantityMatchReward/std": 0.43643729090690614, "rewards/TotalKcalExactMatchReward/mean": 0.78125, "rewards/TotalKcalExactMatchReward/std": 0.40987553000450133, "step": 2880 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.2, "completions/mean_length": 390.0890625, "completions/min_length": 273.0, "epoch": 0.8025034770514604, "frac_reward_zero_std": 0.075, "grad_norm": 0.5778425335884094, "kl": 0.04776061798911542, "learning_rate": 1.0291533644515166e-07, "loss": 0.0019105035811662675, "reward": 3.0733968734741213, "reward_std": 0.42165454030036925, "rewards/IngredientFormatReward/mean": 0.987287950515747, "rewards/IngredientFormatReward/std": 0.0955145888030529, "rewards/IngredientMatchReward/mean": 0.6008587718009949, "rewards/IngredientMatchReward/std": 0.28134148716926577, "rewards/IngredientQuantityMatchReward/mean": 0.6790001511573791, "rewards/IngredientQuantityMatchReward/std": 0.4184362173080444, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.3811137080192566, "step": 2885 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.4, "completions/mean_length": 389.5046875, "completions/min_length": 259.0, "epoch": 0.803894297635605, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7275820374488831, "kl": 0.048515303037129344, "learning_rate": 1.0152193217723315e-07, "loss": 0.0019406870007514954, "reward": 3.0267982006073, "reward_std": 0.399430650472641, "rewards/IngredientFormatReward/mean": 0.9880431652069092, "rewards/IngredientFormatReward/std": 0.09137556552886963, "rewards/IngredientMatchReward/mean": 0.6249008059501648, "rewards/IngredientMatchReward/std": 0.29404258728027344, "rewards/IngredientQuantityMatchReward/mean": 0.6091667413711548, "rewards/IngredientQuantityMatchReward/std": 0.4240457832813263, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.3877830386161804, "step": 2890 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 506.2, "completions/mean_length": 394.7546875, "completions/min_length": 278.8, "epoch": 0.8052851182197497, "frac_reward_zero_std": 0.05, "grad_norm": 0.6058497428894043, "kl": 0.046461883233860135, "learning_rate": 1.0013695857747172e-07, "loss": 0.0018587440252304077, "reward": 3.025787353515625, "reward_std": 0.36919748187065127, "rewards/IngredientFormatReward/mean": 0.9911681652069092, "rewards/IngredientFormatReward/std": 0.07216834891587495, "rewards/IngredientMatchReward/mean": 0.623315978050232, "rewards/IngredientMatchReward/std": 0.2807840764522552, "rewards/IngredientQuantityMatchReward/mean": 0.622240686416626, "rewards/IngredientQuantityMatchReward/std": 0.42640805840492246, "rewards/TotalKcalExactMatchReward/mean": 0.7890625, "rewards/TotalKcalExactMatchReward/std": 0.3843976229429245, "step": 2895 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 502.2, "completions/mean_length": 388.85625, "completions/min_length": 271.2, "epoch": 0.8066759388038943, "frac_reward_zero_std": 0.0625, "grad_norm": 113.02948760986328, "kl": 0.6814675867091864, "learning_rate": 9.876044494798897e-08, "loss": 0.027234816551208497, "reward": 3.025029420852661, "reward_std": 0.386387437582016, "rewards/IngredientFormatReward/mean": 0.990833330154419, "rewards/IngredientFormatReward/std": 0.0677767276763916, "rewards/IngredientMatchReward/mean": 0.6841958165168762, "rewards/IngredientMatchReward/std": 0.29024302363395693, "rewards/IngredientQuantityMatchReward/mean": 0.6140627264976501, "rewards/IngredientQuantityMatchReward/std": 0.4360509991645813, "rewards/TotalKcalExactMatchReward/mean": 0.7359375, "rewards/TotalKcalExactMatchReward/std": 0.4356089770793915, "step": 2900 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 503.8, "completions/mean_length": 393.8546875, "completions/min_length": 285.6, "epoch": 0.808066759388039, "frac_reward_zero_std": 0.125, "grad_norm": 0.6261676549911499, "kl": 0.04418616164475679, "learning_rate": 9.73924204119178e-08, "loss": 0.001767529733479023, "reward": 3.1210771560668946, "reward_std": 0.36356693506240845, "rewards/IngredientFormatReward/mean": 0.9943750023841857, "rewards/IngredientFormatReward/std": 0.04674905762076378, "rewards/IngredientMatchReward/mean": 0.6692689657211304, "rewards/IngredientMatchReward/std": 0.2812860310077667, "rewards/IngredientQuantityMatchReward/mean": 0.6855581760406494, "rewards/IngredientQuantityMatchReward/std": 0.3925300002098083, "rewards/TotalKcalExactMatchReward/mean": 0.771875, "rewards/TotalKcalExactMatchReward/std": 0.4036038815975189, "step": 2905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 491.2, "completions/mean_length": 379.5515625, "completions/min_length": 250.4, "epoch": 0.8094575799721836, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7257301807403564, "kl": 0.04759784871712327, "learning_rate": 9.603291391278568e-08, "loss": 0.0019040904939174653, "reward": 3.122410774230957, "reward_std": 0.39100813269615176, "rewards/IngredientFormatReward/mean": 0.9928645849227905, "rewards/IngredientFormatReward/std": 0.0445655882358551, "rewards/IngredientMatchReward/mean": 0.6938579082489014, "rewards/IngredientMatchReward/std": 0.28338626623153684, "rewards/IngredientQuantityMatchReward/mean": 0.6185007572174073, "rewards/IngredientQuantityMatchReward/std": 0.4316678524017334, "rewards/TotalKcalExactMatchReward/mean": 0.8171875, "rewards/TotalKcalExactMatchReward/std": 0.37480711936950684, "step": 2910 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 503.8, "completions/mean_length": 389.6546875, "completions/min_length": 239.8, "epoch": 0.8108484005563282, "frac_reward_zero_std": 0.0875, "grad_norm": 0.612226128578186, "kl": 0.0521865741815418, "learning_rate": 9.46819542139023e-08, "loss": 0.0020877527073025703, "reward": 3.0481145858764647, "reward_std": 0.3714823067188263, "rewards/IngredientFormatReward/mean": 0.9858705282211304, "rewards/IngredientFormatReward/std": 0.08456218093633652, "rewards/IngredientMatchReward/mean": 0.6272296667098999, "rewards/IngredientMatchReward/std": 0.2885073244571686, "rewards/IngredientQuantityMatchReward/mean": 0.6459518194198608, "rewards/IngredientQuantityMatchReward/std": 0.42370365262031556, "rewards/TotalKcalExactMatchReward/mean": 0.7890625, "rewards/TotalKcalExactMatchReward/std": 0.4039949357509613, "step": 2915 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 491.8, "completions/mean_length": 386.0296875, "completions/min_length": 248.4, "epoch": 0.8122392211404729, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6456689834594727, "kl": 0.048775306739844385, "learning_rate": 9.333956989775149e-08, "loss": 0.0019510995596647262, "reward": 3.202948474884033, "reward_std": 0.3436084806919098, "rewards/IngredientFormatReward/mean": 0.9966145753860474, "rewards/IngredientFormatReward/std": 0.0278476582840085, "rewards/IngredientMatchReward/mean": 0.6758649468421936, "rewards/IngredientMatchReward/std": 0.295900821685791, "rewards/IngredientQuantityMatchReward/mean": 0.7242189645767212, "rewards/IngredientQuantityMatchReward/std": 0.4036065459251404, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.37954323291778563, "step": 2920 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 513.0, "completions/mean_length": 389.5921875, "completions/min_length": 256.2, "epoch": 0.8136300417246175, "frac_reward_zero_std": 0.075, "grad_norm": 0.7192108035087585, "kl": 0.051587856956757605, "learning_rate": 9.200578936538628e-08, "loss": 0.002063538506627083, "reward": 3.1155295848846434, "reward_std": 0.39563546180725095, "rewards/IngredientFormatReward/mean": 0.9871875047683716, "rewards/IngredientFormatReward/std": 0.1086337298154831, "rewards/IngredientMatchReward/mean": 0.6720517158508301, "rewards/IngredientMatchReward/std": 0.304996919631958, "rewards/IngredientQuantityMatchReward/mean": 0.6562903761863709, "rewards/IngredientQuantityMatchReward/std": 0.40350993871688845, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.3642785012722015, "step": 2925 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 505.6, "completions/mean_length": 392.51875, "completions/min_length": 263.0, "epoch": 0.8150208623087621, "frac_reward_zero_std": 0.05, "grad_norm": 0.6309117674827576, "kl": 0.04693901997525245, "learning_rate": 9.068064083582789e-08, "loss": 0.0018776778131723403, "reward": 3.034470796585083, "reward_std": 0.4299850046634674, "rewards/IngredientFormatReward/mean": 0.9903125047683716, "rewards/IngredientFormatReward/std": 0.07649115696549416, "rewards/IngredientMatchReward/mean": 0.6331597208976746, "rewards/IngredientMatchReward/std": 0.28655193746089935, "rewards/IngredientQuantityMatchReward/mean": 0.63599853515625, "rewards/IngredientQuantityMatchReward/std": 0.4148427963256836, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.41708130240440366, "step": 2930 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 512.6, "completions/mean_length": 383.5734375, "completions/min_length": 258.6, "epoch": 0.8164116828929068, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6402202844619751, "kl": 0.04907006416469813, "learning_rate": 8.93641523454688e-08, "loss": 0.001963062025606632, "reward": 3.087191581726074, "reward_std": 0.389241361618042, "rewards/IngredientFormatReward/mean": 0.9887500047683716, "rewards/IngredientFormatReward/std": 0.07457910180091858, "rewards/IngredientMatchReward/mean": 0.6492887973785401, "rewards/IngredientMatchReward/std": 0.30371943712234495, "rewards/IngredientQuantityMatchReward/mean": 0.6538402795791626, "rewards/IngredientQuantityMatchReward/std": 0.39292051196098327, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.38929224014282227, "step": 2935 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 497.6, "completions/mean_length": 393.7984375, "completions/min_length": 254.8, "epoch": 0.8178025034770514, "frac_reward_zero_std": 0.05, "grad_norm": 0.6798999309539795, "kl": 0.04532430283725262, "learning_rate": 8.805635174747961e-08, "loss": 0.0018129955977201461, "reward": 3.1188385486602783, "reward_std": 0.40057496428489686, "rewards/IngredientFormatReward/mean": 0.995312488079071, "rewards/IngredientFormatReward/std": 0.03757027108222246, "rewards/IngredientMatchReward/mean": 0.651212191581726, "rewards/IngredientMatchReward/std": 0.27968040108680725, "rewards/IngredientQuantityMatchReward/mean": 0.6723138213157653, "rewards/IngredientQuantityMatchReward/std": 0.4165939033031464, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.3970761299133301, "step": 2940 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 495.8, "completions/mean_length": 390.9703125, "completions/min_length": 279.4, "epoch": 0.8191933240611962, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7016852498054504, "kl": 0.04716307336930185, "learning_rate": 8.675726671121968e-08, "loss": 0.0018864812329411507, "reward": 3.167627954483032, "reward_std": 0.39934841394424436, "rewards/IngredientFormatReward/mean": 0.9858463525772094, "rewards/IngredientFormatReward/std": 0.06388909518718719, "rewards/IngredientMatchReward/mean": 0.6783639073371888, "rewards/IngredientMatchReward/std": 0.2869694232940674, "rewards/IngredientQuantityMatchReward/mean": 0.6706052541732788, "rewards/IngredientQuantityMatchReward/std": 0.4015600621700287, "rewards/TotalKcalExactMatchReward/mean": 0.8328125, "rewards/TotalKcalExactMatchReward/std": 0.3679316818714142, "step": 2945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 503.2, "completions/mean_length": 387.15625, "completions/min_length": 278.6, "epoch": 0.8205841446453408, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6271607875823975, "kl": 0.04509587762877345, "learning_rate": 8.546692472165195e-08, "loss": 0.0018041588366031647, "reward": 3.275380325317383, "reward_std": 0.3708919048309326, "rewards/IngredientFormatReward/mean": 0.9959895730018615, "rewards/IngredientFormatReward/std": 0.04264734834432602, "rewards/IngredientMatchReward/mean": 0.709428322315216, "rewards/IngredientMatchReward/std": 0.2649887174367905, "rewards/IngredientQuantityMatchReward/mean": 0.7152748942375183, "rewards/IngredientQuantityMatchReward/std": 0.4001452922821045, "rewards/TotalKcalExactMatchReward/mean": 0.8546875, "rewards/TotalKcalExactMatchReward/std": 0.35101463794708254, "step": 2950 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 506.4, "completions/mean_length": 393.584375, "completions/min_length": 252.8, "epoch": 0.8219749652294854, "frac_reward_zero_std": 0.075, "grad_norm": 0.6617047190666199, "kl": 0.05037928493693471, "learning_rate": 8.418535307876057e-08, "loss": 0.0020150141790509224, "reward": 3.1745755672454834, "reward_std": 0.39280285835266116, "rewards/IngredientFormatReward/mean": 0.9833072900772095, "rewards/IngredientFormatReward/std": 0.10187680423259735, "rewards/IngredientMatchReward/mean": 0.6803211688995361, "rewards/IngredientMatchReward/std": 0.29972589910030367, "rewards/IngredientQuantityMatchReward/mean": 0.704697048664093, "rewards/IngredientQuantityMatchReward/std": 0.39612128734588625, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.39199302196502683, "step": 2955 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 503.0, "completions/mean_length": 389.19375, "completions/min_length": 266.0, "epoch": 0.8233657858136301, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6642850637435913, "kl": 0.04720083279535174, "learning_rate": 8.291257889697484e-08, "loss": 0.001888134703040123, "reward": 3.070013332366943, "reward_std": 0.41594597697257996, "rewards/IngredientFormatReward/mean": 0.9950000047683716, "rewards/IngredientFormatReward/std": 0.04611458256840706, "rewards/IngredientMatchReward/mean": 0.6542714476585388, "rewards/IngredientMatchReward/std": 0.25903120040893557, "rewards/IngredientQuantityMatchReward/mean": 0.6598044037818909, "rewards/IngredientQuantityMatchReward/std": 0.42194475531578063, "rewards/TotalKcalExactMatchReward/mean": 0.7609375, "rewards/TotalKcalExactMatchReward/std": 0.4252581536769867, "step": 2960 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 498.4, "completions/mean_length": 389.225, "completions/min_length": 275.6, "epoch": 0.8247566063977747, "frac_reward_zero_std": 0.075, "grad_norm": 0.6487119197845459, "kl": 0.045345666841603814, "learning_rate": 8.164862910459419e-08, "loss": 0.0018140103667974472, "reward": 3.168592071533203, "reward_std": 0.3895823538303375, "rewards/IngredientFormatReward/mean": 0.9959895730018615, "rewards/IngredientFormatReward/std": 0.04264734834432602, "rewards/IngredientMatchReward/mean": 0.6649200201034546, "rewards/IngredientMatchReward/std": 0.29957814812660216, "rewards/IngredientQuantityMatchReward/mean": 0.6608074545860291, "rewards/IngredientQuantityMatchReward/std": 0.42049002051353457, "rewards/TotalKcalExactMatchReward/mean": 0.846875, "rewards/TotalKcalExactMatchReward/std": 0.35720263719558715, "step": 2965 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 495.2, "completions/mean_length": 385.578125, "completions/min_length": 256.4, "epoch": 0.8261474269819193, "frac_reward_zero_std": 0.0625, "grad_norm": 0.653709888458252, "kl": 0.05087601642590016, "learning_rate": 8.039353044321918e-08, "loss": 0.002034936100244522, "reward": 3.047412872314453, "reward_std": 0.4022683262825012, "rewards/IngredientFormatReward/mean": 0.9851041674613953, "rewards/IngredientFormatReward/std": 0.08566536605358124, "rewards/IngredientMatchReward/mean": 0.6148604869842529, "rewards/IngredientMatchReward/std": 0.28956696689128875, "rewards/IngredientQuantityMatchReward/mean": 0.6208856463432312, "rewards/IngredientQuantityMatchReward/std": 0.42820629477500916, "rewards/TotalKcalExactMatchReward/mean": 0.8265625, "rewards/TotalKcalExactMatchReward/std": 0.36872145533561707, "step": 2970 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 508.4, "completions/mean_length": 391.1875, "completions/min_length": 247.8, "epoch": 0.827538247566064, "frac_reward_zero_std": 0.05, "grad_norm": 0.6367550492286682, "kl": 0.04671785989776254, "learning_rate": 7.914730946718507e-08, "loss": 0.0018687203526496887, "reward": 3.100764036178589, "reward_std": 0.4259964883327484, "rewards/IngredientFormatReward/mean": 0.9904017925262452, "rewards/IngredientFormatReward/std": 0.0754810044541955, "rewards/IngredientMatchReward/mean": 0.6349913835525512, "rewards/IngredientMatchReward/std": 0.2922417223453522, "rewards/IngredientQuantityMatchReward/mean": 0.6753708362579346, "rewards/IngredientQuantityMatchReward/std": 0.4144640028476715, "rewards/TotalKcalExactMatchReward/mean": 0.8, "rewards/TotalKcalExactMatchReward/std": 0.3949720025062561, "step": 2975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 507.6, "completions/mean_length": 391.1828125, "completions/min_length": 265.6, "epoch": 0.8289290681502086, "frac_reward_zero_std": 0.1125, "grad_norm": 0.5961768627166748, "kl": 0.05127133592031896, "learning_rate": 7.790999254300079e-08, "loss": 0.0020504213869571688, "reward": 3.1191386222839355, "reward_std": 0.3484513610601425, "rewards/IngredientFormatReward/mean": 0.9890625, "rewards/IngredientFormatReward/std": 0.08017933368682861, "rewards/IngredientMatchReward/mean": 0.6748840570449829, "rewards/IngredientMatchReward/std": 0.2862056910991669, "rewards/IngredientQuantityMatchReward/mean": 0.5989420533180236, "rewards/IngredientQuantityMatchReward/std": 0.42672608494758607, "rewards/TotalKcalExactMatchReward/mean": 0.85625, "rewards/TotalKcalExactMatchReward/std": 0.3235033631324768, "step": 2980 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 508.6, "completions/mean_length": 386.3671875, "completions/min_length": 248.2, "epoch": 0.8303198887343533, "frac_reward_zero_std": 0.0875, "grad_norm": 0.655218243598938, "kl": 0.048838711553253235, "learning_rate": 7.668160584879068e-08, "loss": 0.001953764446079731, "reward": 3.109603023529053, "reward_std": 0.3838211238384247, "rewards/IngredientFormatReward/mean": 0.9916865229606628, "rewards/IngredientFormatReward/std": 0.0682242352515459, "rewards/IngredientMatchReward/mean": 0.6327531456947326, "rewards/IngredientMatchReward/std": 0.2947903752326965, "rewards/IngredientQuantityMatchReward/mean": 0.6726634740829468, "rewards/IngredientQuantityMatchReward/std": 0.4099680960178375, "rewards/TotalKcalExactMatchReward/mean": 0.8125, "rewards/TotalKcalExactMatchReward/std": 0.38938270807266234, "step": 2985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 507.2, "completions/mean_length": 391.321875, "completions/min_length": 272.0, "epoch": 0.8317107093184979, "frac_reward_zero_std": 0.125, "grad_norm": 0.5529534816741943, "kl": 0.042588020046241584, "learning_rate": 7.546217537374072e-08, "loss": 0.0017037199810147285, "reward": 3.1710736751556396, "reward_std": 0.33150983452796934, "rewards/IngredientFormatReward/mean": 0.9946056485176087, "rewards/IngredientFormatReward/std": 0.03346139807254076, "rewards/IngredientMatchReward/mean": 0.6643967270851135, "rewards/IngredientMatchReward/std": 0.29930198192596436, "rewards/IngredientQuantityMatchReward/mean": 0.6495713591575623, "rewards/IngredientQuantityMatchReward/std": 0.4249771058559418, "rewards/TotalKcalExactMatchReward/mean": 0.8625, "rewards/TotalKcalExactMatchReward/std": 0.33672092854976654, "step": 2990 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 504.6, "completions/mean_length": 389.7640625, "completions/min_length": 259.6, "epoch": 0.8331015299026425, "frac_reward_zero_std": 0.075, "grad_norm": 0.6654627919197083, "kl": 0.0474369776668027, "learning_rate": 7.42517269175485e-08, "loss": 0.001897449791431427, "reward": 3.0309982299804688, "reward_std": 0.39750537276268005, "rewards/IngredientFormatReward/mean": 0.9892187476158142, "rewards/IngredientFormatReward/std": 0.08536323457956314, "rewards/IngredientMatchReward/mean": 0.6950334906578064, "rewards/IngredientMatchReward/std": 0.29454014301300047, "rewards/IngredientQuantityMatchReward/mean": 0.5983084917068482, "rewards/IngredientQuantityMatchReward/std": 0.42347745299339296, "rewards/TotalKcalExactMatchReward/mean": 0.7484375, "rewards/TotalKcalExactMatchReward/std": 0.418589586019516, "step": 2995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 502.4, "completions/mean_length": 382.84375, "completions/min_length": 256.4, "epoch": 0.8344923504867872, "frac_reward_zero_std": 0.075, "grad_norm": 7.202028751373291, "kl": 0.06541607140097767, "learning_rate": 7.305028608987762e-08, "loss": 0.0026157567277550696, "reward": 3.107814598083496, "reward_std": 0.3948281526565552, "rewards/IngredientFormatReward/mean": 0.9901041746139526, "rewards/IngredientFormatReward/std": 0.0734422504901886, "rewards/IngredientMatchReward/mean": 0.6643247961997986, "rewards/IngredientMatchReward/std": 0.27318963408470154, "rewards/IngredientQuantityMatchReward/mean": 0.6783856272697448, "rewards/IngredientQuantityMatchReward/std": 0.4000774621963501, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.40691637992858887, "step": 3000 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 495.2, "completions/mean_length": 390.521875, "completions/min_length": 248.8, "epoch": 0.8358831710709318, "frac_reward_zero_std": 0.025, "grad_norm": 0.6431273818016052, "kl": 0.04963696929626167, "learning_rate": 7.185787830981571e-08, "loss": 0.001985309273004532, "reward": 3.116542387008667, "reward_std": 0.38055387139320374, "rewards/IngredientFormatReward/mean": 0.9975353479385376, "rewards/IngredientFormatReward/std": 0.024885546416044235, "rewards/IngredientMatchReward/mean": 0.6366834044456482, "rewards/IngredientMatchReward/std": 0.2909588277339935, "rewards/IngredientQuantityMatchReward/mean": 0.6620110869407654, "rewards/IngredientQuantityMatchReward/std": 0.3996714651584625, "rewards/TotalKcalExactMatchReward/mean": 0.8203125, "rewards/TotalKcalExactMatchReward/std": 0.368584543466568, "step": 3005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 508.2, "completions/mean_length": 392.6125, "completions/min_length": 277.6, "epoch": 0.8372739916550765, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6810948252677917, "kl": 0.0459325288888067, "learning_rate": 7.067452880533664e-08, "loss": 0.0018374029546976089, "reward": 3.079831838607788, "reward_std": 0.40001912117004396, "rewards/IngredientFormatReward/mean": 0.9881584882736206, "rewards/IngredientFormatReward/std": 0.08064654730260372, "rewards/IngredientMatchReward/mean": 0.6321608543395996, "rewards/IngredientMatchReward/std": 0.30199761390686036, "rewards/IngredientQuantityMatchReward/mean": 0.6610750675201416, "rewards/IngredientQuantityMatchReward/std": 0.4067903161048889, "rewards/TotalKcalExactMatchReward/mean": 0.7984375, "rewards/TotalKcalExactMatchReward/std": 0.38732614517211916, "step": 3010 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 502.4, "completions/mean_length": 389.1453125, "completions/min_length": 269.0, "epoch": 0.8386648122392212, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6376404762268066, "kl": 0.05073229791596532, "learning_rate": 6.950026261276698e-08, "loss": 0.002029072307050228, "reward": 3.1437541007995606, "reward_std": 0.35086989402770996, "rewards/IngredientFormatReward/mean": 0.99469496011734, "rewards/IngredientFormatReward/std": 0.04523397646844387, "rewards/IngredientMatchReward/mean": 0.6596317172050477, "rewards/IngredientMatchReward/std": 0.28066009283065796, "rewards/IngredientQuantityMatchReward/mean": 0.6753650188446045, "rewards/IngredientQuantityMatchReward/std": 0.40645899176597594, "rewards/TotalKcalExactMatchReward/mean": 0.8140625, "rewards/TotalKcalExactMatchReward/std": 0.3786509335041046, "step": 3015 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 505.4, "completions/mean_length": 397.4359375, "completions/min_length": 298.6, "epoch": 0.8400556328233658, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6552076935768127, "kl": 0.04711475428193808, "learning_rate": 6.833510457625585e-08, "loss": 0.0018848728388547898, "reward": 3.2228328227996825, "reward_std": 0.37228912115097046, "rewards/IngredientFormatReward/mean": 0.9916666626930237, "rewards/IngredientFormatReward/std": 0.06616733074188233, "rewards/IngredientMatchReward/mean": 0.6734991908073426, "rewards/IngredientMatchReward/std": 0.29104630947113036, "rewards/IngredientQuantityMatchReward/mean": 0.7248544454574585, "rewards/IngredientQuantityMatchReward/std": 0.3884549975395203, "rewards/TotalKcalExactMatchReward/mean": 0.8328125, "rewards/TotalKcalExactMatchReward/std": 0.35898853838443756, "step": 3020 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 507.2, "completions/mean_length": 391.75625, "completions/min_length": 274.8, "epoch": 0.8414464534075105, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6299717426300049, "kl": 0.04465638347901404, "learning_rate": 6.717907934724982e-08, "loss": 0.0017864655703306199, "reward": 3.200212860107422, "reward_std": 0.3218538582324982, "rewards/IngredientFormatReward/mean": 0.9934895753860473, "rewards/IngredientFormatReward/std": 0.052749037928879264, "rewards/IngredientMatchReward/mean": 0.675394332408905, "rewards/IngredientMatchReward/std": 0.2600559026002884, "rewards/IngredientQuantityMatchReward/mean": 0.7000789642333984, "rewards/IngredientQuantityMatchReward/std": 0.3908191204071045, "rewards/TotalKcalExactMatchReward/mean": 0.83125, "rewards/TotalKcalExactMatchReward/std": 0.37399315237998965, "step": 3025 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 508.0, "completions/mean_length": 395.0703125, "completions/min_length": 292.4, "epoch": 0.8428372739916551, "frac_reward_zero_std": 0.05, "grad_norm": 0.6462646126747131, "kl": 0.04630208441521973, "learning_rate": 6.603221138397102e-08, "loss": 0.0018521670252084732, "reward": 3.1141008377075194, "reward_std": 0.43531657457351686, "rewards/IngredientFormatReward/mean": 0.9883407711982727, "rewards/IngredientFormatReward/std": 0.08107312843203544, "rewards/IngredientMatchReward/mean": 0.6730586767196656, "rewards/IngredientMatchReward/std": 0.2721665918827057, "rewards/IngredientQuantityMatchReward/mean": 0.6464512586593628, "rewards/IngredientQuantityMatchReward/std": 0.40384441614151, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.3568021148443222, "step": 3030 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 499.6, "completions/mean_length": 393.7890625, "completions/min_length": 278.2, "epoch": 0.8442280945757997, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6232410073280334, "kl": 0.046137299528345464, "learning_rate": 6.489452495089959e-08, "loss": 0.0018458150327205658, "reward": 3.069223690032959, "reward_std": 0.38708738684654237, "rewards/IngredientFormatReward/mean": 0.9917299151420593, "rewards/IngredientFormatReward/std": 0.048492889106273654, "rewards/IngredientMatchReward/mean": 0.6515817403793335, "rewards/IngredientMatchReward/std": 0.2845022022724152, "rewards/IngredientQuantityMatchReward/mean": 0.6180995345115662, "rewards/IngredientQuantityMatchReward/std": 0.41500294804573057, "rewards/TotalKcalExactMatchReward/mean": 0.8078125, "rewards/TotalKcalExactMatchReward/std": 0.38249378800392153, "step": 3035 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 509.8, "completions/mean_length": 392.3171875, "completions/min_length": 252.6, "epoch": 0.8456189151599444, "frac_reward_zero_std": 0.075, "grad_norm": 8.307910919189453, "kl": 0.07405607257969678, "learning_rate": 6.376604411826069e-08, "loss": 0.002963574230670929, "reward": 3.0168015003204345, "reward_std": 0.45492454767227175, "rewards/IngredientFormatReward/mean": 0.9760770082473755, "rewards/IngredientFormatReward/std": 0.13225289061665535, "rewards/IngredientMatchReward/mean": 0.6269741415977478, "rewards/IngredientMatchReward/std": 0.3054593026638031, "rewards/IngredientQuantityMatchReward/mean": 0.6465628862380981, "rewards/IngredientQuantityMatchReward/std": 0.43766944408416747, "rewards/TotalKcalExactMatchReward/mean": 0.7671875, "rewards/TotalKcalExactMatchReward/std": 0.4169674217700958, "step": 3040 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 507.4, "completions/mean_length": 396.1921875, "completions/min_length": 259.2, "epoch": 0.847009735744089, "frac_reward_zero_std": 0.1, "grad_norm": 0.7011051774024963, "kl": 0.05020157103426755, "learning_rate": 6.264679276151485e-08, "loss": 0.0020083660259842873, "reward": 3.1013489723205567, "reward_std": 0.36317563652992246, "rewards/IngredientFormatReward/mean": 0.9876562356948853, "rewards/IngredientFormatReward/std": 0.08381983656436205, "rewards/IngredientMatchReward/mean": 0.6747749328613282, "rewards/IngredientMatchReward/std": 0.2843883991241455, "rewards/IngredientQuantityMatchReward/mean": 0.6686052441596985, "rewards/IngredientQuantityMatchReward/std": 0.38197819590568544, "rewards/TotalKcalExactMatchReward/mean": 0.7703125, "rewards/TotalKcalExactMatchReward/std": 0.39580188393592836, "step": 3045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 508.2, "completions/mean_length": 393.009375, "completions/min_length": 265.0, "epoch": 0.8484005563282336, "frac_reward_zero_std": 0.075, "grad_norm": 0.6340299844741821, "kl": 0.04501628072466701, "learning_rate": 6.153679456085343e-08, "loss": 0.0018011130392551421, "reward": 3.1332841396331785, "reward_std": 0.33361607491970063, "rewards/IngredientFormatReward/mean": 0.9887760400772094, "rewards/IngredientFormatReward/std": 0.06544025391340255, "rewards/IngredientMatchReward/mean": 0.6738690495491028, "rewards/IngredientMatchReward/std": 0.2672120571136475, "rewards/IngredientQuantityMatchReward/mean": 0.6628265261650086, "rewards/IngredientQuantityMatchReward/std": 0.3794059336185455, "rewards/TotalKcalExactMatchReward/mean": 0.8078125, "rewards/TotalKcalExactMatchReward/std": 0.3943259060382843, "step": 3050 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 501.2, "completions/mean_length": 394.6828125, "completions/min_length": 273.0, "epoch": 0.8497913769123783, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6601257920265198, "kl": 0.04646438907366246, "learning_rate": 6.043607300069653e-08, "loss": 0.0018583284690976143, "reward": 3.096318817138672, "reward_std": 0.3890433728694916, "rewards/IngredientFormatReward/mean": 0.9942968726158142, "rewards/IngredientFormatReward/std": 0.051978741958737376, "rewards/IngredientMatchReward/mean": 0.6110546827316284, "rewards/IngredientMatchReward/std": 0.31328256130218507, "rewards/IngredientQuantityMatchReward/mean": 0.6878422737121582, "rewards/IngredientQuantityMatchReward/std": 0.4182025969028473, "rewards/TotalKcalExactMatchReward/mean": 0.803125, "rewards/TotalKcalExactMatchReward/std": 0.3834996998310089, "step": 3055 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 490.2, "completions/mean_length": 382.29375, "completions/min_length": 262.2, "epoch": 0.8511821974965229, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6450671553611755, "kl": 0.047940224315971135, "learning_rate": 5.934465136919736e-08, "loss": 0.001917634904384613, "reward": 3.150379753112793, "reward_std": 0.35938249826431273, "rewards/IngredientFormatReward/mean": 0.9995814800262451, "rewards/IngredientFormatReward/std": 0.0033425264060497283, "rewards/IngredientMatchReward/mean": 0.6239750862121582, "rewards/IngredientMatchReward/std": 0.2981367170810699, "rewards/IngredientQuantityMatchReward/mean": 0.6846357345581054, "rewards/IngredientQuantityMatchReward/std": 0.4151392221450806, "rewards/TotalKcalExactMatchReward/mean": 0.8421875, "rewards/TotalKcalExactMatchReward/std": 0.35220016837120055, "step": 3060 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 513.0, "completions/mean_length": 392.1671875, "completions/min_length": 276.2, "epoch": 0.8525730180806675, "frac_reward_zero_std": 0.1, "grad_norm": 0.6400882005691528, "kl": 0.046734277112409475, "learning_rate": 5.826255275774861e-08, "loss": 0.0018693899735808372, "reward": 3.0946020126342773, "reward_std": 0.41802999973297117, "rewards/IngredientFormatReward/mean": 0.9781845331192016, "rewards/IngredientFormatReward/std": 0.12996118217706681, "rewards/IngredientMatchReward/mean": 0.6358180522918702, "rewards/IngredientMatchReward/std": 0.3087162494659424, "rewards/IngredientQuantityMatchReward/mean": 0.6821619391441345, "rewards/IngredientQuantityMatchReward/std": 0.41959701776504515, "rewards/TotalKcalExactMatchReward/mean": 0.7984375, "rewards/TotalKcalExactMatchReward/std": 0.39511425495147706, "step": 3065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 513.0, "completions/mean_length": 390.409375, "completions/min_length": 266.6, "epoch": 0.8539638386648123, "frac_reward_zero_std": 0.025, "grad_norm": 0.6598068475723267, "kl": 0.05595761397853494, "learning_rate": 5.718980006049445e-08, "loss": 0.0022384027019143105, "reward": 3.1849377155303955, "reward_std": 0.4088379919528961, "rewards/IngredientFormatReward/mean": 0.991391372680664, "rewards/IngredientFormatReward/std": 0.08908663541078568, "rewards/IngredientMatchReward/mean": 0.6616598486900329, "rewards/IngredientMatchReward/std": 0.2799966514110565, "rewards/IngredientQuantityMatchReward/mean": 0.6928239345550538, "rewards/IngredientQuantityMatchReward/std": 0.4054431617259979, "rewards/TotalKcalExactMatchReward/mean": 0.8390625, "rewards/TotalKcalExactMatchReward/std": 0.3501300573348999, "step": 3070 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 508.4, "completions/mean_length": 397.04375, "completions/min_length": 263.8, "epoch": 0.8553546592489569, "frac_reward_zero_std": 0.025, "grad_norm": 0.6476394534111023, "kl": 0.04536970742046833, "learning_rate": 5.6126415973845875e-08, "loss": 0.00181522686034441, "reward": 3.127425193786621, "reward_std": 0.4319356918334961, "rewards/IngredientFormatReward/mean": 0.9797916650772095, "rewards/IngredientFormatReward/std": 0.11502126976847649, "rewards/IngredientMatchReward/mean": 0.6737530946731567, "rewards/IngredientMatchReward/std": 0.2894026696681976, "rewards/IngredientQuantityMatchReward/mean": 0.6629429399967194, "rewards/IngredientQuantityMatchReward/std": 0.403452605009079, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.3918249785900116, "step": 3075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 506.4, "completions/mean_length": 394.4671875, "completions/min_length": 275.0, "epoch": 0.8567454798331016, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6560621857643127, "kl": 0.050239683431573215, "learning_rate": 5.5072422996000626e-08, "loss": 0.002009827084839344, "reward": 3.2230833530426026, "reward_std": 0.3705843985080719, "rewards/IngredientFormatReward/mean": 0.9926041603088379, "rewards/IngredientFormatReward/std": 0.05480610430240631, "rewards/IngredientMatchReward/mean": 0.6874944806098938, "rewards/IngredientMatchReward/std": 0.28081194758415223, "rewards/IngredientQuantityMatchReward/mean": 0.6945472121238708, "rewards/IngredientQuantityMatchReward/std": 0.39099807739257814, "rewards/TotalKcalExactMatchReward/mean": 0.8484375, "rewards/TotalKcalExactMatchReward/std": 0.35526342391967775, "step": 3080 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 509.4, "completions/mean_length": 391.5140625, "completions/min_length": 271.8, "epoch": 0.8581363004172462, "frac_reward_zero_std": 0.025, "grad_norm": 1.5736770629882812, "kl": 0.04943891023285687, "learning_rate": 5.40278434264671e-08, "loss": 0.001977892965078354, "reward": 3.0344571590423586, "reward_std": 0.34014609456062317, "rewards/IngredientFormatReward/mean": 0.9945535778999328, "rewards/IngredientFormatReward/std": 0.041323404759168625, "rewards/IngredientMatchReward/mean": 0.615483021736145, "rewards/IngredientMatchReward/std": 0.27883716523647306, "rewards/IngredientQuantityMatchReward/mean": 0.6556705832481384, "rewards/IngredientQuantityMatchReward/std": 0.41032910346984863, "rewards/TotalKcalExactMatchReward/mean": 0.76875, "rewards/TotalKcalExactMatchReward/std": 0.42164164781570435, "step": 3085 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 495.2, "completions/mean_length": 386.803125, "completions/min_length": 246.8, "epoch": 0.8595271210013908, "frac_reward_zero_std": 0.0625, "grad_norm": 0.8206337690353394, "kl": 0.051253604143857955, "learning_rate": 5.299269936559275e-08, "loss": 0.0020507143810391424, "reward": 3.12697377204895, "reward_std": 0.40267518162727356, "rewards/IngredientFormatReward/mean": 0.9957291603088378, "rewards/IngredientFormatReward/std": 0.03133343979716301, "rewards/IngredientMatchReward/mean": 0.6811867475509643, "rewards/IngredientMatchReward/std": 0.2661647886037827, "rewards/IngredientQuantityMatchReward/mean": 0.6641203999519348, "rewards/IngredientQuantityMatchReward/std": 0.40845090746879575, "rewards/TotalKcalExactMatchReward/mean": 0.7859375, "rewards/TotalKcalExactMatchReward/std": 0.40290495157241824, "step": 3090 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 506.4, "completions/mean_length": 394.28125, "completions/min_length": 265.0, "epoch": 0.8609179415855355, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7553901672363281, "kl": 0.06073210455942899, "learning_rate": 5.196701271409615e-08, "loss": 0.0024279464036226273, "reward": 3.1119028091430665, "reward_std": 0.47098749279975893, "rewards/IngredientFormatReward/mean": 0.9827256798744202, "rewards/IngredientFormatReward/std": 0.09406303185969592, "rewards/IngredientMatchReward/mean": 0.6565879464149476, "rewards/IngredientMatchReward/std": 0.28705950975418093, "rewards/IngredientQuantityMatchReward/mean": 0.6757142424583436, "rewards/IngredientQuantityMatchReward/std": 0.4014221131801605, "rewards/TotalKcalExactMatchReward/mean": 0.796875, "rewards/TotalKcalExactMatchReward/std": 0.39542931914329527, "step": 3095 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 511.6, "completions/mean_length": 398.4203125, "completions/min_length": 274.6, "epoch": 0.8623087621696801, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7021408081054688, "kl": 0.04793826362583786, "learning_rate": 5.095080517260397e-08, "loss": 0.0019172437489032746, "reward": 3.079103422164917, "reward_std": 0.41487077474594114, "rewards/IngredientFormatReward/mean": 0.9877511143684388, "rewards/IngredientFormatReward/std": 0.0843487460166216, "rewards/IngredientMatchReward/mean": 0.6174058675765991, "rewards/IngredientMatchReward/std": 0.2904114544391632, "rewards/IngredientQuantityMatchReward/mean": 0.6380090236663818, "rewards/IngredientQuantityMatchReward/std": 0.4309688091278076, "rewards/TotalKcalExactMatchReward/mean": 0.8359375, "rewards/TotalKcalExactMatchReward/std": 0.3543159782886505, "step": 3100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 513.0, "completions/mean_length": 390.659375, "completions/min_length": 273.8, "epoch": 0.8636995827538247, "frac_reward_zero_std": 0.05, "grad_norm": 0.6427891254425049, "kl": 0.04723424823023379, "learning_rate": 4.994409824119189e-08, "loss": 0.001889539510011673, "reward": 3.184514856338501, "reward_std": 0.43525434732437135, "rewards/IngredientFormatReward/mean": 0.991796875, "rewards/IngredientFormatReward/std": 0.08889861106872558, "rewards/IngredientMatchReward/mean": 0.6531863808631897, "rewards/IngredientMatchReward/std": 0.28255691528320315, "rewards/IngredientQuantityMatchReward/mean": 0.7442190766334533, "rewards/IngredientQuantityMatchReward/std": 0.37337024211883546, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.39657433032989503, "step": 3105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 505.0, "completions/mean_length": 389.3578125, "completions/min_length": 267.2, "epoch": 0.8650904033379694, "frac_reward_zero_std": 0.05, "grad_norm": 0.6806484460830688, "kl": 0.23146929705981165, "learning_rate": 4.8946913218929466e-08, "loss": 0.009275725483894348, "reward": 3.085928964614868, "reward_std": 0.37679257392883303, "rewards/IngredientFormatReward/mean": 0.9950744032859802, "rewards/IngredientFormatReward/std": 0.04088784530758858, "rewards/IngredientMatchReward/mean": 0.6328473806381225, "rewards/IngredientMatchReward/std": 0.3016449213027954, "rewards/IngredientQuantityMatchReward/mean": 0.6376946866512299, "rewards/IngredientQuantityMatchReward/std": 0.42166988253593446, "rewards/TotalKcalExactMatchReward/mean": 0.8203125, "rewards/TotalKcalExactMatchReward/std": 0.3804445922374725, "step": 3110 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 506.6, "completions/mean_length": 387.0, "completions/min_length": 239.4, "epoch": 0.866481223922114, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6415086984634399, "kl": 0.07196273931767791, "learning_rate": 4.7959271203429405e-08, "loss": 0.002879709005355835, "reward": 3.1648409843444822, "reward_std": 0.3970713198184967, "rewards/IngredientFormatReward/mean": 0.979895830154419, "rewards/IngredientFormatReward/std": 0.10437989309430122, "rewards/IngredientMatchReward/mean": 0.6672836184501648, "rewards/IngredientMatchReward/std": 0.2711143881082535, "rewards/IngredientQuantityMatchReward/mean": 0.6801616072654724, "rewards/IngredientQuantityMatchReward/std": 0.40835112929344175, "rewards/TotalKcalExactMatchReward/mean": 0.8375, "rewards/TotalKcalExactMatchReward/std": 0.3565270006656647, "step": 3115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 504.2, "completions/mean_length": 387.425, "completions/min_length": 270.8, "epoch": 0.8678720445062587, "frac_reward_zero_std": 0.05, "grad_norm": 0.6656712293624878, "kl": 0.046086785243824126, "learning_rate": 4.6981193090401604e-08, "loss": 0.0018435075879096984, "reward": 3.188760280609131, "reward_std": 0.400802755355835, "rewards/IngredientFormatReward/mean": 0.9972656011581421, "rewards/IngredientFormatReward/std": 0.02613703105598688, "rewards/IngredientMatchReward/mean": 0.6525930047035218, "rewards/IngredientMatchReward/std": 0.290238493680954, "rewards/IngredientQuantityMatchReward/mean": 0.701401686668396, "rewards/IngredientQuantityMatchReward/std": 0.39277478456497195, "rewards/TotalKcalExactMatchReward/mean": 0.8375, "rewards/TotalKcalExactMatchReward/std": 0.36130423545837403, "step": 3120 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 497.0, "completions/mean_length": 390.6546875, "completions/min_length": 247.8, "epoch": 0.8692628650904033, "frac_reward_zero_std": 0.075, "grad_norm": 0.6652013659477234, "kl": 0.04941854891367257, "learning_rate": 4.601269957321091e-08, "loss": 0.00197696965187788, "reward": 3.116376447677612, "reward_std": 0.41209186911582946, "rewards/IngredientFormatReward/mean": 0.9899330377578736, "rewards/IngredientFormatReward/std": 0.053738721460103986, "rewards/IngredientMatchReward/mean": 0.6498803496360779, "rewards/IngredientMatchReward/std": 0.2805606722831726, "rewards/IngredientQuantityMatchReward/mean": 0.670313024520874, "rewards/IngredientQuantityMatchReward/std": 0.41227946877479554, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.378220134973526, "step": 3125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 510.2, "completions/mean_length": 391.4265625, "completions/min_length": 255.4, "epoch": 0.8706536856745479, "frac_reward_zero_std": 0.05, "grad_norm": 0.7125198245048523, "kl": 0.04970926723908633, "learning_rate": 4.5053811142439056e-08, "loss": 0.0019885823130607605, "reward": 2.987143039703369, "reward_std": 0.42771296501159667, "rewards/IngredientFormatReward/mean": 0.987291669845581, "rewards/IngredientFormatReward/std": 0.082947389036417, "rewards/IngredientMatchReward/mean": 0.6242001533508301, "rewards/IngredientMatchReward/std": 0.28825379014015196, "rewards/IngredientQuantityMatchReward/mean": 0.6147137761116028, "rewards/IngredientQuantityMatchReward/std": 0.42187526226043703, "rewards/TotalKcalExactMatchReward/mean": 0.7609375, "rewards/TotalKcalExactMatchReward/std": 0.395613294839859, "step": 3130 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 509.6, "completions/mean_length": 387.49375, "completions/min_length": 264.4, "epoch": 0.8720445062586927, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6735195517539978, "kl": 0.048319844016805294, "learning_rate": 4.410454808545144e-08, "loss": 0.0019330214709043503, "reward": 3.125268268585205, "reward_std": 0.42156311869621277, "rewards/IngredientFormatReward/mean": 0.9879371166229248, "rewards/IngredientFormatReward/std": 0.09172794967889786, "rewards/IngredientMatchReward/mean": 0.6832422733306884, "rewards/IngredientMatchReward/std": 0.283637547492981, "rewards/IngredientQuantityMatchReward/mean": 0.652526342868805, "rewards/IngredientQuantityMatchReward/std": 0.4098579943180084, "rewards/TotalKcalExactMatchReward/mean": 0.8015625, "rewards/TotalKcalExactMatchReward/std": 0.3839901268482208, "step": 3135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 504.6, "completions/mean_length": 387.3328125, "completions/min_length": 260.6, "epoch": 0.8734353268428373, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6666101813316345, "kl": 0.04869773075915873, "learning_rate": 4.316493048596786e-08, "loss": 0.0019479818642139436, "reward": 3.1016587257385253, "reward_std": 0.4278287351131439, "rewards/IngredientFormatReward/mean": 0.9838541746139526, "rewards/IngredientFormatReward/std": 0.08848539292812348, "rewards/IngredientMatchReward/mean": 0.6473877549171447, "rewards/IngredientMatchReward/std": 0.3035359501838684, "rewards/IngredientQuantityMatchReward/mean": 0.6501043200492859, "rewards/IngredientQuantityMatchReward/std": 0.4268841803073883, "rewards/TotalKcalExactMatchReward/mean": 0.8203125, "rewards/TotalKcalExactMatchReward/std": 0.37457784414291384, "step": 3140 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 493.6, "completions/mean_length": 387.4078125, "completions/min_length": 255.6, "epoch": 0.874826147426982, "frac_reward_zero_std": 0.075, "grad_norm": 0.6627177596092224, "kl": 0.04510366797912866, "learning_rate": 4.2234978223637365e-08, "loss": 0.0018041422590613365, "reward": 3.216240119934082, "reward_std": 0.36082149744033815, "rewards/IngredientFormatReward/mean": 0.9941145896911621, "rewards/IngredientFormatReward/std": 0.05023606196045875, "rewards/IngredientMatchReward/mean": 0.674437654018402, "rewards/IngredientMatchReward/std": 0.28741170167922975, "rewards/IngredientQuantityMatchReward/mean": 0.7445629119873047, "rewards/IngredientQuantityMatchReward/std": 0.37724101543426514, "rewards/TotalKcalExactMatchReward/mean": 0.803125, "rewards/TotalKcalExactMatchReward/std": 0.39241496920585633, "step": 3145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 490.4, "completions/mean_length": 392.046875, "completions/min_length": 276.2, "epoch": 0.8762169680111266, "frac_reward_zero_std": 0.0375, "grad_norm": 0.609026312828064, "kl": 0.046659936918877064, "learning_rate": 4.1314710973617994e-08, "loss": 0.001866631954908371, "reward": 3.122904109954834, "reward_std": 0.34099590182304385, "rewards/IngredientFormatReward/mean": 0.9961904764175415, "rewards/IngredientFormatReward/std": 0.03287437073886394, "rewards/IngredientMatchReward/mean": 0.6628584146499634, "rewards/IngredientMatchReward/std": 0.28121124804019926, "rewards/IngredientQuantityMatchReward/mean": 0.6841677546501159, "rewards/IngredientQuantityMatchReward/std": 0.39395946860313413, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.4074713945388794, "step": 3150 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 509.8, "completions/mean_length": 390.603125, "completions/min_length": 265.4, "epoch": 0.8776077885952712, "frac_reward_zero_std": 0.05, "grad_norm": 0.7134307622909546, "kl": 0.04822455393150449, "learning_rate": 4.040414820616006e-08, "loss": 0.0019289350137114524, "reward": 3.003550624847412, "reward_std": 0.39522513151168825, "rewards/IngredientFormatReward/mean": 0.985182273387909, "rewards/IngredientFormatReward/std": 0.08801141232252122, "rewards/IngredientMatchReward/mean": 0.6495293974876404, "rewards/IngredientMatchReward/std": 0.2960770189762115, "rewards/IngredientQuantityMatchReward/mean": 0.6282139658927918, "rewards/IngredientQuantityMatchReward/std": 0.4116261124610901, "rewards/TotalKcalExactMatchReward/mean": 0.740625, "rewards/TotalKcalExactMatchReward/std": 0.43001351356506345, "step": 3155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 506.4, "completions/mean_length": 393.2734375, "completions/min_length": 262.8, "epoch": 0.8789986091794159, "frac_reward_zero_std": 0.05, "grad_norm": 0.6307358741760254, "kl": 0.04441832688171417, "learning_rate": 3.950330918619488e-08, "loss": 0.0017767375335097312, "reward": 3.0464088916778564, "reward_std": 0.45417346954345705, "rewards/IngredientFormatReward/mean": 0.9859703660011292, "rewards/IngredientFormatReward/std": 0.10039616972208024, "rewards/IngredientMatchReward/mean": 0.6358823299407959, "rewards/IngredientMatchReward/std": 0.2765005111694336, "rewards/IngredientQuantityMatchReward/mean": 0.6792436122894288, "rewards/IngredientQuantityMatchReward/std": 0.39857603907585143, "rewards/TotalKcalExactMatchReward/mean": 0.7453125, "rewards/TotalKcalExactMatchReward/std": 0.4332336366176605, "step": 3160 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 504.6, "completions/mean_length": 391.35, "completions/min_length": 274.2, "epoch": 0.8803894297635605, "frac_reward_zero_std": 0.075, "grad_norm": 0.6311498880386353, "kl": 0.04784476771019399, "learning_rate": 3.861221297292655e-08, "loss": 0.0019135732203722, "reward": 3.1883824348449705, "reward_std": 0.36939600110054016, "rewards/IngredientFormatReward/mean": 0.9859598278999329, "rewards/IngredientFormatReward/std": 0.10982539653778076, "rewards/IngredientMatchReward/mean": 0.6296354055404663, "rewards/IngredientMatchReward/std": 0.2733914017677307, "rewards/IngredientQuantityMatchReward/mean": 0.7274746179580689, "rewards/IngredientQuantityMatchReward/std": 0.37859617471694945, "rewards/TotalKcalExactMatchReward/mean": 0.8453125, "rewards/TotalKcalExactMatchReward/std": 0.35329467952251437, "step": 3165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 513.0, "completions/mean_length": 390.471875, "completions/min_length": 248.6, "epoch": 0.8817802503477051, "frac_reward_zero_std": 0.05, "grad_norm": 0.5605978965759277, "kl": 0.047909293626435104, "learning_rate": 3.7730878419429066e-08, "loss": 0.0019164547324180604, "reward": 3.0932684421539305, "reward_std": 0.42589526176452636, "rewards/IngredientFormatReward/mean": 0.9879092216491699, "rewards/IngredientFormatReward/std": 0.10345041304826737, "rewards/IngredientMatchReward/mean": 0.6832546114921569, "rewards/IngredientMatchReward/std": 0.2867938756942749, "rewards/IngredientQuantityMatchReward/mean": 0.6080420970916748, "rewards/IngredientQuantityMatchReward/std": 0.4464531302452087, "rewards/TotalKcalExactMatchReward/mean": 0.8140625, "rewards/TotalKcalExactMatchReward/std": 0.3797110438346863, "step": 3170 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 510.8, "completions/mean_length": 392.5828125, "completions/min_length": 275.8, "epoch": 0.8831710709318498, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6873921751976013, "kl": 0.04381905437912792, "learning_rate": 3.685932417224702e-08, "loss": 0.0017528809607028962, "reward": 3.072804307937622, "reward_std": 0.4026482582092285, "rewards/IngredientFormatReward/mean": 0.99140625, "rewards/IngredientFormatReward/std": 0.07906749099493027, "rewards/IngredientMatchReward/mean": 0.6443638324737548, "rewards/IngredientMatchReward/std": 0.28569428324699403, "rewards/IngredientQuantityMatchReward/mean": 0.6745341658592224, "rewards/IngredientQuantityMatchReward/std": 0.4039541006088257, "rewards/TotalKcalExactMatchReward/mean": 0.7625, "rewards/TotalKcalExactMatchReward/std": 0.41969347596168516, "step": 3175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 508.6, "completions/mean_length": 389.384375, "completions/min_length": 250.6, "epoch": 0.8845618915159944, "frac_reward_zero_std": 0.075, "grad_norm": 0.7061554193496704, "kl": 0.04746296023949981, "learning_rate": 3.599756867100184e-08, "loss": 0.0018984757363796235, "reward": 3.1786093711853027, "reward_std": 0.42314148545265196, "rewards/IngredientFormatReward/mean": 0.9849218845367431, "rewards/IngredientFormatReward/std": 0.10753473863005639, "rewards/IngredientMatchReward/mean": 0.7073313474655152, "rewards/IngredientMatchReward/std": 0.26829246878623964, "rewards/IngredientQuantityMatchReward/mean": 0.6582311034202576, "rewards/IngredientQuantityMatchReward/std": 0.3875416398048401, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.3658925950527191, "step": 3180 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.8, "completions/mean_length": 391.9296875, "completions/min_length": 267.0, "epoch": 0.885952712100139, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6179981231689453, "kl": 0.04594576419331133, "learning_rate": 3.5145630148000985e-08, "loss": 0.0018379855901002884, "reward": 3.1979689598083496, "reward_std": 0.3995221495628357, "rewards/IngredientFormatReward/mean": 0.9856621861457825, "rewards/IngredientFormatReward/std": 0.10145160276442766, "rewards/IngredientMatchReward/mean": 0.6657750487327576, "rewards/IngredientMatchReward/std": 0.28745636343955994, "rewards/IngredientQuantityMatchReward/mean": 0.6980942010879516, "rewards/IngredientQuantityMatchReward/std": 0.4006811559200287, "rewards/TotalKcalExactMatchReward/mean": 0.8484375, "rewards/TotalKcalExactMatchReward/std": 0.3429839700460434, "step": 3185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 503.4, "completions/mean_length": 397.546875, "completions/min_length": 297.6, "epoch": 0.8873435326842837, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6583110094070435, "kl": 0.2816643947735429, "learning_rate": 3.430352662785247e-08, "loss": 0.011268769949674606, "reward": 3.1793424606323244, "reward_std": 0.41445272564888, "rewards/IngredientFormatReward/mean": 0.984973955154419, "rewards/IngredientFormatReward/std": 0.09121790900826454, "rewards/IngredientMatchReward/mean": 0.6691094040870667, "rewards/IngredientMatchReward/std": 0.2692943513393402, "rewards/IngredientQuantityMatchReward/mean": 0.6674466133117676, "rewards/IngredientQuantityMatchReward/std": 0.3967915832996368, "rewards/TotalKcalExactMatchReward/mean": 0.8578125, "rewards/TotalKcalExactMatchReward/std": 0.34571763277053835, "step": 3190 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 509.6, "completions/mean_length": 389.496875, "completions/min_length": 263.6, "epoch": 0.8887343532684284, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6590292453765869, "kl": 0.05054741660133004, "learning_rate": 3.347127592708343e-08, "loss": 0.0020219020545482634, "reward": 3.209326696395874, "reward_std": 0.3645615816116333, "rewards/IngredientFormatReward/mean": 0.9900520801544189, "rewards/IngredientFormatReward/std": 0.08570180833339691, "rewards/IngredientMatchReward/mean": 0.6802461624145508, "rewards/IngredientMatchReward/std": 0.2965859532356262, "rewards/IngredientQuantityMatchReward/mean": 0.7109034299850464, "rewards/IngredientQuantityMatchReward/std": 0.38022114634513854, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.3564129650592804, "step": 3195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 511.8, "completions/mean_length": 388.940625, "completions/min_length": 272.0, "epoch": 0.8901251738525731, "frac_reward_zero_std": 0.05, "grad_norm": 0.7058317065238953, "kl": 0.047892588051036, "learning_rate": 3.264889565376339e-08, "loss": 0.0019158922135829926, "reward": 3.0706414222717284, "reward_std": 0.4435698390007019, "rewards/IngredientFormatReward/mean": 0.9884374976158142, "rewards/IngredientFormatReward/std": 0.08515960946679116, "rewards/IngredientMatchReward/mean": 0.6230040967464447, "rewards/IngredientMatchReward/std": 0.2796986043453217, "rewards/IngredientQuantityMatchReward/mean": 0.6310748338699341, "rewards/IngredientQuantityMatchReward/std": 0.4169824719429016, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.3632709324359894, "step": 3200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 506.2, "completions/mean_length": 388.6578125, "completions/min_length": 248.2, "epoch": 0.8915159944367177, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6913332343101501, "kl": 0.05210544446017593, "learning_rate": 3.183640320713138e-08, "loss": 0.0020846392959356307, "reward": 3.0734073162078857, "reward_std": 0.37518334984779356, "rewards/IngredientFormatReward/mean": 0.9895312547683716, "rewards/IngredientFormatReward/std": 0.0965784803032875, "rewards/IngredientMatchReward/mean": 0.6489539980888367, "rewards/IngredientMatchReward/std": 0.2867078483104706, "rewards/IngredientQuantityMatchReward/mean": 0.6114845871925354, "rewards/IngredientQuantityMatchReward/std": 0.43328599333763124, "rewards/TotalKcalExactMatchReward/mean": 0.8234375, "rewards/TotalKcalExactMatchReward/std": 0.37979313135147097, "step": 3205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 509.0, "completions/mean_length": 395.76875, "completions/min_length": 270.6, "epoch": 0.8929068150208623, "frac_reward_zero_std": 0.075, "grad_norm": 0.6467673182487488, "kl": 0.047686185827478765, "learning_rate": 3.103381577722813e-08, "loss": 0.0019076012074947357, "reward": 3.168344497680664, "reward_std": 0.37602880597114563, "rewards/IngredientFormatReward/mean": 0.9888020753860474, "rewards/IngredientFormatReward/std": 0.0813770966604352, "rewards/IngredientMatchReward/mean": 0.6629953980445862, "rewards/IngredientMatchReward/std": 0.28421200513839723, "rewards/IngredientQuantityMatchReward/mean": 0.7024844765663147, "rewards/IngredientQuantityMatchReward/std": 0.39068577289581297, "rewards/TotalKcalExactMatchReward/mean": 0.8140625, "rewards/TotalKcalExactMatchReward/std": 0.3665438830852509, "step": 3210 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 503.0, "completions/mean_length": 394.2890625, "completions/min_length": 269.0, "epoch": 0.894297635605007, "frac_reward_zero_std": 0.025, "grad_norm": 0.644658625125885, "kl": 0.044069158774800596, "learning_rate": 3.024115034453223e-08, "loss": 0.0017627470195293426, "reward": 3.1621095657348635, "reward_std": 0.38854520320892333, "rewards/IngredientFormatReward/mean": 0.9934375047683716, "rewards/IngredientFormatReward/std": 0.053338292986154556, "rewards/IngredientMatchReward/mean": 0.6316596031188965, "rewards/IngredientMatchReward/std": 0.2820505201816559, "rewards/IngredientQuantityMatchReward/mean": 0.6838874697685242, "rewards/IngredientQuantityMatchReward/std": 0.4012498319149017, "rewards/TotalKcalExactMatchReward/mean": 0.853125, "rewards/TotalKcalExactMatchReward/std": 0.3469159841537476, "step": 3215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 499.6, "completions/mean_length": 385.9265625, "completions/min_length": 242.8, "epoch": 0.8956884561891516, "frac_reward_zero_std": 0.1125, "grad_norm": 0.6403445601463318, "kl": 0.049373555136844514, "learning_rate": 2.945842367960083e-08, "loss": 0.0019750598818063735, "reward": 3.1126923084259035, "reward_std": 0.3951055645942688, "rewards/IngredientFormatReward/mean": 0.9936458230018616, "rewards/IngredientFormatReward/std": 0.04035495463758707, "rewards/IngredientMatchReward/mean": 0.6792900562286377, "rewards/IngredientMatchReward/std": 0.27618914246559145, "rewards/IngredientQuantityMatchReward/mean": 0.6788189053535462, "rewards/IngredientQuantityMatchReward/std": 0.41655081510543823, "rewards/TotalKcalExactMatchReward/mean": 0.7609375, "rewards/TotalKcalExactMatchReward/std": 0.4172710537910461, "step": 3220 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 507.4, "completions/mean_length": 390.265625, "completions/min_length": 257.2, "epoch": 0.8970792767732962, "frac_reward_zero_std": 0.0125, "grad_norm": 0.717547595500946, "kl": 0.04932956439442933, "learning_rate": 2.8685652342714862e-08, "loss": 0.0019732709974050523, "reward": 3.148496913909912, "reward_std": 0.35338476300239563, "rewards/IngredientFormatReward/mean": 0.9878236651420593, "rewards/IngredientFormatReward/std": 0.07924887351691723, "rewards/IngredientMatchReward/mean": 0.6488428235054016, "rewards/IngredientMatchReward/std": 0.2869942843914032, "rewards/IngredientQuantityMatchReward/mean": 0.7008928775787353, "rewards/IngredientQuantityMatchReward/std": 0.3995750486850739, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.39261109232902525, "step": 3225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 506.4, "completions/mean_length": 392.1296875, "completions/min_length": 259.8, "epoch": 0.8984700973574409, "frac_reward_zero_std": 0.05, "grad_norm": 0.6339384913444519, "kl": 0.04682166455313563, "learning_rate": 2.7922852683528897e-08, "loss": 0.0018730312585830688, "reward": 3.091051721572876, "reward_std": 0.3852741241455078, "rewards/IngredientFormatReward/mean": 0.983017110824585, "rewards/IngredientFormatReward/std": 0.09547805488109588, "rewards/IngredientMatchReward/mean": 0.6440420389175415, "rewards/IngredientMatchReward/std": 0.29824180603027345, "rewards/IngredientQuantityMatchReward/mean": 0.6530550360679627, "rewards/IngredientQuantityMatchReward/std": 0.4097617506980896, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.36531429886817934, "step": 3230 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 510.8, "completions/mean_length": 393.89375, "completions/min_length": 267.8, "epoch": 0.8998609179415855, "frac_reward_zero_std": 0.05, "grad_norm": 0.5954923629760742, "kl": 0.04409891550894827, "learning_rate": 2.7170040840724716e-08, "loss": 0.0017639545723795891, "reward": 3.036423349380493, "reward_std": 0.4317321002483368, "rewards/IngredientFormatReward/mean": 0.9889583349227905, "rewards/IngredientFormatReward/std": 0.08621723800897599, "rewards/IngredientMatchReward/mean": 0.6406045317649841, "rewards/IngredientMatchReward/std": 0.2745776265859604, "rewards/IngredientQuantityMatchReward/mean": 0.5740479826927185, "rewards/IngredientQuantityMatchReward/std": 0.43019055128097533, "rewards/TotalKcalExactMatchReward/mean": 0.8328125, "rewards/TotalKcalExactMatchReward/std": 0.3640967130661011, "step": 3235 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 512.2, "completions/mean_length": 392.6125, "completions/min_length": 269.6, "epoch": 0.9012517385257302, "frac_reward_zero_std": 0.05, "grad_norm": 0.6983107328414917, "kl": 0.04888640344142914, "learning_rate": 2.6427232741670357e-08, "loss": 0.0019554536789655685, "reward": 3.089215040206909, "reward_std": 0.4097185730934143, "rewards/IngredientFormatReward/mean": 0.9904798984527587, "rewards/IngredientFormatReward/std": 0.07944944500923157, "rewards/IngredientMatchReward/mean": 0.6514082670211792, "rewards/IngredientMatchReward/std": 0.28968934416770936, "rewards/IngredientQuantityMatchReward/mean": 0.6317018032073974, "rewards/IngredientQuantityMatchReward/std": 0.42319734692573546, "rewards/TotalKcalExactMatchReward/mean": 0.815625, "rewards/TotalKcalExactMatchReward/std": 0.38177411556243895, "step": 3240 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 509.4, "completions/mean_length": 394.903125, "completions/min_length": 273.4, "epoch": 0.9026425591098748, "frac_reward_zero_std": 0.05, "grad_norm": 0.5831993222236633, "kl": 0.04227819531224668, "learning_rate": 2.5694444102082936e-08, "loss": 0.0016911562532186508, "reward": 3.1611047267913817, "reward_std": 0.3636060357093811, "rewards/IngredientFormatReward/mean": 0.99296875, "rewards/IngredientFormatReward/std": 0.06629023440182209, "rewards/IngredientMatchReward/mean": 0.6555952429771423, "rewards/IngredientMatchReward/std": 0.2678823709487915, "rewards/IngredientQuantityMatchReward/mean": 0.7016033172607422, "rewards/IngredientQuantityMatchReward/std": 0.388336056470871, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.3912253439426422, "step": 3245 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 507.4, "completions/mean_length": 392.5296875, "completions/min_length": 274.6, "epoch": 0.9040333796940194, "frac_reward_zero_std": 0.075, "grad_norm": 0.641535758972168, "kl": 0.0460631252033636, "learning_rate": 2.49716904256963e-08, "loss": 0.0018425412476062776, "reward": 3.134316158294678, "reward_std": 0.3648345053195953, "rewards/IngredientFormatReward/mean": 0.995892858505249, "rewards/IngredientFormatReward/std": 0.03714258074760437, "rewards/IngredientMatchReward/mean": 0.6300421595573426, "rewards/IngredientMatchReward/std": 0.2943162739276886, "rewards/IngredientQuantityMatchReward/mean": 0.6911936283111573, "rewards/IngredientQuantityMatchReward/std": 0.39555092453956603, "rewards/TotalKcalExactMatchReward/mean": 0.8171875, "rewards/TotalKcalExactMatchReward/std": 0.3763595879077911, "step": 3250 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 507.8, "completions/mean_length": 392.590625, "completions/min_length": 259.0, "epoch": 0.9054242002781642, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6454972624778748, "kl": 0.04925240727607161, "learning_rate": 2.425898700393253e-08, "loss": 0.0019701555371284484, "reward": 2.9525651931762695, "reward_std": 0.43930028676986693, "rewards/IngredientFormatReward/mean": 0.9813281297683716, "rewards/IngredientFormatReward/std": 0.11844980865716934, "rewards/IngredientMatchReward/mean": 0.5760538458824158, "rewards/IngredientMatchReward/std": 0.29196472764015197, "rewards/IngredientQuantityMatchReward/mean": 0.6451832890510559, "rewards/IngredientQuantityMatchReward/std": 0.42400979399681094, "rewards/TotalKcalExactMatchReward/mean": 0.75, "rewards/TotalKcalExactMatchReward/std": 0.4188802659511566, "step": 3255 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 503.4, "completions/mean_length": 389.4859375, "completions/min_length": 275.2, "epoch": 0.9068150208623088, "frac_reward_zero_std": 0.1, "grad_norm": 0.6864833235740662, "kl": 0.045732753025367855, "learning_rate": 2.355634891557906e-08, "loss": 0.0018294647336006165, "reward": 3.015725040435791, "reward_std": 0.40643337965011594, "rewards/IngredientFormatReward/mean": 0.9921875, "rewards/IngredientFormatReward/std": 0.05261292904615402, "rewards/IngredientMatchReward/mean": 0.6334901928901673, "rewards/IngredientMatchReward/std": 0.3036071300506592, "rewards/IngredientQuantityMatchReward/mean": 0.5759848117828369, "rewards/IngredientQuantityMatchReward/std": 0.4259013533592224, "rewards/TotalKcalExactMatchReward/mean": 0.8140625, "rewards/TotalKcalExactMatchReward/std": 0.3635615885257721, "step": 3260 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 503.6, "completions/mean_length": 388.340625, "completions/min_length": 252.6, "epoch": 0.9082058414464534, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7106591463088989, "kl": 0.04727880118880421, "learning_rate": 2.2863791026469235e-08, "loss": 0.0018911713734269143, "reward": 3.2334872245788575, "reward_std": 0.37439358830451963, "rewards/IngredientFormatReward/mean": 0.9897135257720947, "rewards/IngredientFormatReward/std": 0.05642170626670122, "rewards/IngredientMatchReward/mean": 0.653626000881195, "rewards/IngredientMatchReward/std": 0.2840902924537659, "rewards/IngredientQuantityMatchReward/mean": 0.7245226621627807, "rewards/IngredientQuantityMatchReward/std": 0.38535062670707704, "rewards/TotalKcalExactMatchReward/mean": 0.865625, "rewards/TotalKcalExactMatchReward/std": 0.32322187423706056, "step": 3265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 511.4, "completions/mean_length": 393.809375, "completions/min_length": 271.6, "epoch": 0.9095966620305981, "frac_reward_zero_std": 0.0875, "grad_norm": 0.626308023929596, "kl": 0.047429022891446945, "learning_rate": 2.2181327989167998e-08, "loss": 0.0018970862030982972, "reward": 3.040598201751709, "reward_std": 0.42209060192108155, "rewards/IngredientFormatReward/mean": 0.9875, "rewards/IngredientFormatReward/std": 0.08148004412651062, "rewards/IngredientMatchReward/mean": 0.6066759705543519, "rewards/IngredientMatchReward/std": 0.3146925806999207, "rewards/IngredientQuantityMatchReward/mean": 0.637047290802002, "rewards/IngredientQuantityMatchReward/std": 0.4168846786022186, "rewards/TotalKcalExactMatchReward/mean": 0.809375, "rewards/TotalKcalExactMatchReward/std": 0.38911654353141784, "step": 3270 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 508.4, "completions/mean_length": 393.825, "completions/min_length": 284.8, "epoch": 0.9109874826147427, "frac_reward_zero_std": 0.0625, "grad_norm": 0.5958412289619446, "kl": 0.049182718200609085, "learning_rate": 2.1508974242661628e-08, "loss": 0.0019672702997922896, "reward": 3.096751403808594, "reward_std": 0.3994257628917694, "rewards/IngredientFormatReward/mean": 0.9866145730018616, "rewards/IngredientFormatReward/std": 0.0923412561416626, "rewards/IngredientMatchReward/mean": 0.6960999369621277, "rewards/IngredientMatchReward/std": 0.27538976073265076, "rewards/IngredientQuantityMatchReward/mean": 0.6546619057655334, "rewards/IngredientQuantityMatchReward/std": 0.37647982835769656, "rewards/TotalKcalExactMatchReward/mean": 0.759375, "rewards/TotalKcalExactMatchReward/std": 0.41687182784080506, "step": 3275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 503.8, "completions/mean_length": 387.3296875, "completions/min_length": 251.0, "epoch": 0.9123783031988874, "frac_reward_zero_std": 0.1, "grad_norm": 0.6608989238739014, "kl": 0.04685659667011351, "learning_rate": 2.084674401205261e-08, "loss": 0.0018743416294455528, "reward": 3.0759621143341063, "reward_std": 0.3722398400306702, "rewards/IngredientFormatReward/mean": 0.9926562428474426, "rewards/IngredientFormatReward/std": 0.05678885132074356, "rewards/IngredientMatchReward/mean": 0.6163002371788024, "rewards/IngredientMatchReward/std": 0.2801432520151138, "rewards/IngredientQuantityMatchReward/mean": 0.6420055627822876, "rewards/IngredientQuantityMatchReward/std": 0.435650622844696, "rewards/TotalKcalExactMatchReward/mean": 0.825, "rewards/TotalKcalExactMatchReward/std": 0.35427531599998474, "step": 3280 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 496.4, "completions/mean_length": 388.78125, "completions/min_length": 271.8, "epoch": 0.913769123783032, "frac_reward_zero_std": 0.05, "grad_norm": 0.6058049201965332, "kl": 0.04507222096435726, "learning_rate": 2.019465130825837e-08, "loss": 0.001802671328186989, "reward": 3.2701576232910154, "reward_std": 0.345516049861908, "rewards/IngredientFormatReward/mean": 0.9930729150772095, "rewards/IngredientFormatReward/std": 0.048128366470336914, "rewards/IngredientMatchReward/mean": 0.664314353466034, "rewards/IngredientMatchReward/std": 0.2860167622566223, "rewards/IngredientQuantityMatchReward/mean": 0.7299577355384826, "rewards/IngredientQuantityMatchReward/std": 0.3946863770484924, "rewards/TotalKcalExactMatchReward/mean": 0.8828125, "rewards/TotalKcalExactMatchReward/std": 0.31640257239341735, "step": 3285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 507.8, "completions/mean_length": 396.428125, "completions/min_length": 276.4, "epoch": 0.9151599443671766, "frac_reward_zero_std": 0.1, "grad_norm": 0.6639472246170044, "kl": 0.0454582569655031, "learning_rate": 1.9552709927715072e-08, "loss": 0.0018183968961238862, "reward": 3.0982945442199705, "reward_std": 0.3722270905971527, "rewards/IngredientFormatReward/mean": 0.9878385543823243, "rewards/IngredientFormatReward/std": 0.07973492741584778, "rewards/IngredientMatchReward/mean": 0.6462369799613953, "rewards/IngredientMatchReward/std": 0.289064759016037, "rewards/IngredientQuantityMatchReward/mean": 0.6142190217971801, "rewards/IngredientQuantityMatchReward/std": 0.42048256993293764, "rewards/TotalKcalExactMatchReward/mean": 0.85, "rewards/TotalKcalExactMatchReward/std": 0.32386162132024765, "step": 3290 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 512.0, "completions/mean_length": 392.221875, "completions/min_length": 265.8, "epoch": 0.9165507649513213, "frac_reward_zero_std": 0.05, "grad_norm": 0.591355562210083, "kl": 0.04644181546755135, "learning_rate": 1.8920933452085396e-08, "loss": 0.0018579160794615745, "reward": 2.946201133728027, "reward_std": 0.4095645725727081, "rewards/IngredientFormatReward/mean": 0.9810007452964783, "rewards/IngredientFormatReward/std": 0.11646474674344062, "rewards/IngredientMatchReward/mean": 0.6444252252578735, "rewards/IngredientMatchReward/std": 0.3023004472255707, "rewards/IngredientQuantityMatchReward/mean": 0.6270251989364624, "rewards/IngredientQuantityMatchReward/std": 0.4267409145832062, "rewards/TotalKcalExactMatchReward/mean": 0.69375, "rewards/TotalKcalExactMatchReward/std": 0.44755579829216, "step": 3295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 502.8, "completions/mean_length": 391.75625, "completions/min_length": 262.0, "epoch": 0.9179415855354659, "frac_reward_zero_std": 0.05, "grad_norm": 0.6602470874786377, "kl": 0.04800860662944615, "learning_rate": 1.829933524797156e-08, "loss": 0.0019200056791305541, "reward": 3.1091113567352293, "reward_std": 0.396321576833725, "rewards/IngredientFormatReward/mean": 0.9884375095367431, "rewards/IngredientFormatReward/std": 0.08126041516661645, "rewards/IngredientMatchReward/mean": 0.6062586784362793, "rewards/IngredientMatchReward/std": 0.26619652807712557, "rewards/IngredientQuantityMatchReward/mean": 0.6925402522087097, "rewards/IngredientQuantityMatchReward/std": 0.39906328320503237, "rewards/TotalKcalExactMatchReward/mean": 0.821875, "rewards/TotalKcalExactMatchReward/std": 0.37994712591171265, "step": 3300 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 506.2, "completions/mean_length": 392.1375, "completions/min_length": 275.0, "epoch": 0.9193324061196105, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6438077688217163, "kl": 0.05436852378770709, "learning_rate": 1.768792846663242e-08, "loss": 0.002175022102892399, "reward": 3.1007835388183596, "reward_std": 0.3859154999256134, "rewards/IngredientFormatReward/mean": 0.9934895753860473, "rewards/IngredientFormatReward/std": 0.0604776605963707, "rewards/IngredientMatchReward/mean": 0.6451841711997985, "rewards/IngredientMatchReward/std": 0.2823725491762161, "rewards/IngredientQuantityMatchReward/mean": 0.6699223518371582, "rewards/IngredientQuantityMatchReward/std": 0.41072106957435606, "rewards/TotalKcalExactMatchReward/mean": 0.7921875, "rewards/TotalKcalExactMatchReward/std": 0.4017234563827515, "step": 3305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 507.0, "completions/mean_length": 386.634375, "completions/min_length": 266.6, "epoch": 0.9207232267037552, "frac_reward_zero_std": 0.025, "grad_norm": 1.0080318450927734, "kl": 0.05075234514661133, "learning_rate": 1.708672604370509e-08, "loss": 0.002030384540557861, "reward": 2.9925003051757812, "reward_std": 0.419496089220047, "rewards/IngredientFormatReward/mean": 0.9936458349227906, "rewards/IngredientFormatReward/std": 0.05940539613366127, "rewards/IngredientMatchReward/mean": 0.6513331174850464, "rewards/IngredientMatchReward/std": 0.30027254223823546, "rewards/IngredientQuantityMatchReward/mean": 0.5662714242935181, "rewards/IngredientQuantityMatchReward/std": 0.43809983134269714, "rewards/TotalKcalExactMatchReward/mean": 0.78125, "rewards/TotalKcalExactMatchReward/std": 0.401913982629776, "step": 3310 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 495.2, "completions/mean_length": 390.7421875, "completions/min_length": 254.4, "epoch": 0.9221140472878998, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6824638247489929, "kl": 0.04954786649905145, "learning_rate": 1.6495740698931282e-08, "loss": 0.0019818780943751334, "reward": 3.1271950244903564, "reward_std": 0.41434894800186156, "rewards/IngredientFormatReward/mean": 0.9919791698455811, "rewards/IngredientFormatReward/std": 0.05760362073779106, "rewards/IngredientMatchReward/mean": 0.6849448323249817, "rewards/IngredientMatchReward/std": 0.27029902338981626, "rewards/IngredientQuantityMatchReward/mean": 0.6799584984779358, "rewards/IngredientQuantityMatchReward/std": 0.3997422754764557, "rewards/TotalKcalExactMatchReward/mean": 0.7703125, "rewards/TotalKcalExactMatchReward/std": 0.41917206048965455, "step": 3315 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 506.8, "completions/mean_length": 393.246875, "completions/min_length": 258.2, "epoch": 0.9235048678720446, "frac_reward_zero_std": 0.05, "grad_norm": 0.6570489406585693, "kl": 0.09846211015246809, "learning_rate": 1.5914984935888277e-08, "loss": 0.0039469413459300995, "reward": 3.0435446739196776, "reward_std": 0.44498496055603026, "rewards/IngredientFormatReward/mean": 0.9882663726806641, "rewards/IngredientFormatReward/std": 0.07785128802061081, "rewards/IngredientMatchReward/mean": 0.6336315870285034, "rewards/IngredientMatchReward/std": 0.3072134733200073, "rewards/IngredientQuantityMatchReward/mean": 0.6185217320919036, "rewards/IngredientQuantityMatchReward/std": 0.417307448387146, "rewards/TotalKcalExactMatchReward/mean": 0.803125, "rewards/TotalKcalExactMatchReward/std": 0.37303715348243716, "step": 3320 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 507.4, "completions/mean_length": 397.1890625, "completions/min_length": 289.6, "epoch": 0.9248956884561892, "frac_reward_zero_std": 0.075, "grad_norm": 0.7150806188583374, "kl": 0.046813591942191125, "learning_rate": 1.5344471041724482e-08, "loss": 0.0018727334216237068, "reward": 3.0286718368530274, "reward_std": 0.37150052189826965, "rewards/IngredientFormatReward/mean": 0.9911458253860473, "rewards/IngredientFormatReward/std": 0.06259849891066552, "rewards/IngredientMatchReward/mean": 0.6338678121566772, "rewards/IngredientMatchReward/std": 0.29035924673080443, "rewards/IngredientQuantityMatchReward/mean": 0.6130332827568055, "rewards/IngredientQuantityMatchReward/std": 0.4491858184337616, "rewards/TotalKcalExactMatchReward/mean": 0.790625, "rewards/TotalKcalExactMatchReward/std": 0.39557355642318726, "step": 3325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 499.4, "completions/mean_length": 387.3109375, "completions/min_length": 264.4, "epoch": 0.9262865090403338, "frac_reward_zero_std": 0.05, "grad_norm": 0.6601356267929077, "kl": 0.0573459361679852, "learning_rate": 1.4784211086899146e-08, "loss": 0.0022935055196285246, "reward": 3.205912399291992, "reward_std": 0.40079067945480346, "rewards/IngredientFormatReward/mean": 0.9884375095367431, "rewards/IngredientFormatReward/std": 0.08228911980986595, "rewards/IngredientMatchReward/mean": 0.6975455045700073, "rewards/IngredientMatchReward/std": 0.2682806015014648, "rewards/IngredientQuantityMatchReward/mean": 0.6918044209480285, "rewards/IngredientQuantityMatchReward/std": 0.39961135387420654, "rewards/TotalKcalExactMatchReward/mean": 0.828125, "rewards/TotalKcalExactMatchReward/std": 0.3684461772441864, "step": 3330 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.8, "completions/mean_length": 391.8828125, "completions/min_length": 260.2, "epoch": 0.9276773296244785, "frac_reward_zero_std": 0.05, "grad_norm": 3.000927209854126, "kl": 0.05884796802420169, "learning_rate": 1.4234216924927377e-08, "loss": 0.0023534022271633147, "reward": 3.089488124847412, "reward_std": 0.4574960291385651, "rewards/IngredientFormatReward/mean": 0.9880059480667114, "rewards/IngredientFormatReward/std": 0.09700787290930749, "rewards/IngredientMatchReward/mean": 0.6292633891105652, "rewards/IngredientMatchReward/std": 0.2786643743515015, "rewards/IngredientQuantityMatchReward/mean": 0.6644062876701355, "rewards/IngredientQuantityMatchReward/std": 0.40214182138442994, "rewards/TotalKcalExactMatchReward/mean": 0.8078125, "rewards/TotalKcalExactMatchReward/std": 0.36265517473220826, "step": 3335 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 513.0, "completions/mean_length": 392.290625, "completions/min_length": 271.6, "epoch": 0.9290681502086231, "frac_reward_zero_std": 0.05, "grad_norm": 0.6722849011421204, "kl": 0.04828175606671721, "learning_rate": 1.3694500192128977e-08, "loss": 0.0019316310063004494, "reward": 2.968636226654053, "reward_std": 0.45129722356796265, "rewards/IngredientFormatReward/mean": 0.9836458206176758, "rewards/IngredientFormatReward/std": 0.11726146638393402, "rewards/IngredientMatchReward/mean": 0.6276495575904846, "rewards/IngredientMatchReward/std": 0.28815919160842896, "rewards/IngredientQuantityMatchReward/mean": 0.6167159080505371, "rewards/IngredientQuantityMatchReward/std": 0.417825722694397, "rewards/TotalKcalExactMatchReward/mean": 0.740625, "rewards/TotalKcalExactMatchReward/std": 0.4330654442310333, "step": 3340 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 511.8, "completions/mean_length": 394.0671875, "completions/min_length": 278.8, "epoch": 0.9304589707927677, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6125572323799133, "kl": 0.04887234810739756, "learning_rate": 1.3165072307382619e-08, "loss": 0.0019550468772649766, "reward": 3.2130596160888674, "reward_std": 0.39644322991371156, "rewards/IngredientFormatReward/mean": 0.9825446486473084, "rewards/IngredientFormatReward/std": 0.1088470995426178, "rewards/IngredientMatchReward/mean": 0.6876734018325805, "rewards/IngredientMatchReward/std": 0.29353204369544983, "rewards/IngredientQuantityMatchReward/mean": 0.6959666013717651, "rewards/IngredientQuantityMatchReward/std": 0.3904150426387787, "rewards/TotalKcalExactMatchReward/mean": 0.846875, "rewards/TotalKcalExactMatchReward/std": 0.33690011501312256, "step": 3345 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 506.2, "completions/mean_length": 394.6640625, "completions/min_length": 282.6, "epoch": 0.9318497913769124, "frac_reward_zero_std": 0.075, "grad_norm": 0.623755693435669, "kl": 0.045136950956657526, "learning_rate": 1.2645944471883995e-08, "loss": 0.0018055334687232972, "reward": 3.077497720718384, "reward_std": 0.3821293354034424, "rewards/IngredientFormatReward/mean": 0.9941257476806641, "rewards/IngredientFormatReward/std": 0.05168246813118458, "rewards/IngredientMatchReward/mean": 0.6398034691810608, "rewards/IngredientMatchReward/std": 0.2708088785409927, "rewards/IngredientQuantityMatchReward/mean": 0.6232560276985168, "rewards/IngredientQuantityMatchReward/std": 0.4233528017997742, "rewards/TotalKcalExactMatchReward/mean": 0.8203125, "rewards/TotalKcalExactMatchReward/std": 0.3787771940231323, "step": 3350 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 507.4, "completions/mean_length": 397.6359375, "completions/min_length": 283.0, "epoch": 0.933240611961057, "frac_reward_zero_std": 0.05, "grad_norm": 0.615950345993042, "kl": 0.042504592309705914, "learning_rate": 1.2137127668908732e-08, "loss": 0.0017003992572426796, "reward": 3.1166717052459716, "reward_std": 0.3788813233375549, "rewards/IngredientFormatReward/mean": 0.9865625023841857, "rewards/IngredientFormatReward/std": 0.08676649853587151, "rewards/IngredientMatchReward/mean": 0.6693539142608642, "rewards/IngredientMatchReward/std": 0.283643913269043, "rewards/IngredientQuantityMatchReward/mean": 0.6466927766799927, "rewards/IngredientQuantityMatchReward/std": 0.42546847462654114, "rewards/TotalKcalExactMatchReward/mean": 0.8140625, "rewards/TotalKcalExactMatchReward/std": 0.3705523431301117, "step": 3355 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 504.2, "completions/mean_length": 395.6390625, "completions/min_length": 281.0, "epoch": 0.9346314325452016, "frac_reward_zero_std": 0.0875, "grad_norm": 0.5828317403793335, "kl": 0.044918736536055805, "learning_rate": 1.163863266358045e-08, "loss": 0.001797039434313774, "reward": 3.166773557662964, "reward_std": 0.367119961977005, "rewards/IngredientFormatReward/mean": 0.985364580154419, "rewards/IngredientFormatReward/std": 0.08733292371034622, "rewards/IngredientMatchReward/mean": 0.6989391088485718, "rewards/IngredientMatchReward/std": 0.2778316617012024, "rewards/IngredientQuantityMatchReward/mean": 0.6699697375297546, "rewards/IngredientQuantityMatchReward/std": 0.39433544874191284, "rewards/TotalKcalExactMatchReward/mean": 0.8125, "rewards/TotalKcalExactMatchReward/std": 0.388651567697525, "step": 3360 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 501.0, "completions/mean_length": 390.125, "completions/min_length": 260.2, "epoch": 0.9360222531293463, "frac_reward_zero_std": 0.0125, "grad_norm": 0.627528190612793, "kl": 0.0473182033514604, "learning_rate": 1.1150470002642686e-08, "loss": 0.0018929582089185714, "reward": 2.9560650825500487, "reward_std": 0.39082145094871523, "rewards/IngredientFormatReward/mean": 0.9876562595367432, "rewards/IngredientFormatReward/std": 0.058681896328926085, "rewards/IngredientMatchReward/mean": 0.6209102034568786, "rewards/IngredientMatchReward/std": 0.2951041400432587, "rewards/IngredientQuantityMatchReward/mean": 0.6381236493587494, "rewards/IngredientQuantityMatchReward/std": 0.4160947024822235, "rewards/TotalKcalExactMatchReward/mean": 0.709375, "rewards/TotalKcalExactMatchReward/std": 0.45090446472167967, "step": 3365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 512.2, "completions/mean_length": 386.2125, "completions/min_length": 261.6, "epoch": 0.9374130737134909, "frac_reward_zero_std": 0.075, "grad_norm": 0.6412569284439087, "kl": 0.04862490389496088, "learning_rate": 1.0672650014235729e-08, "loss": 0.0019449492916464805, "reward": 3.171107292175293, "reward_std": 0.34414549469947814, "rewards/IngredientFormatReward/mean": 0.995520830154419, "rewards/IngredientFormatReward/std": 0.0420041311532259, "rewards/IngredientMatchReward/mean": 0.6497817516326905, "rewards/IngredientMatchReward/std": 0.2970588386058807, "rewards/IngredientQuantityMatchReward/mean": 0.7070546865463256, "rewards/IngredientQuantityMatchReward/std": 0.38046914935112, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.37668890953063966, "step": 3370 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 513.0, "completions/mean_length": 396.796875, "completions/min_length": 276.8, "epoch": 0.9388038942976356, "frac_reward_zero_std": 0.1125, "grad_norm": 0.5388448238372803, "kl": 0.04376723191235214, "learning_rate": 1.0205182807678181e-08, "loss": 0.0017541015520691871, "reward": 3.140538787841797, "reward_std": 0.353511369228363, "rewards/IngredientFormatReward/mean": 0.9919270753860474, "rewards/IngredientFormatReward/std": 0.08860928863286972, "rewards/IngredientMatchReward/mean": 0.6582719564437867, "rewards/IngredientMatchReward/std": 0.256216949224472, "rewards/IngredientQuantityMatchReward/mean": 0.7153397917747497, "rewards/IngredientQuantityMatchReward/std": 0.3973944127559662, "rewards/TotalKcalExactMatchReward/mean": 0.775, "rewards/TotalKcalExactMatchReward/std": 0.41154061555862426, "step": 3375 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 495.2, "completions/mean_length": 392.421875, "completions/min_length": 257.2, "epoch": 0.9401947148817803, "frac_reward_zero_std": 0.05, "grad_norm": 0.660618245601654, "kl": 0.05038860326167196, "learning_rate": 9.748078273253136e-09, "loss": 0.002015620470046997, "reward": 3.008139133453369, "reward_std": 0.4135745406150818, "rewards/IngredientFormatReward/mean": 0.9920312523841858, "rewards/IngredientFormatReward/std": 0.05728701539337635, "rewards/IngredientMatchReward/mean": 0.6056801795959472, "rewards/IngredientMatchReward/std": 0.3066266417503357, "rewards/IngredientQuantityMatchReward/mean": 0.624490213394165, "rewards/IngredientQuantityMatchReward/std": 0.42028595209121705, "rewards/TotalKcalExactMatchReward/mean": 0.7859375, "rewards/TotalKcalExactMatchReward/std": 0.4056543827056885, "step": 3380 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.01875, "completions/max_length": 510.4, "completions/mean_length": 389.103125, "completions/min_length": 268.6, "epoch": 0.9415855354659249, "frac_reward_zero_std": 0.0125, "grad_norm": 0.7143226861953735, "kl": 0.05210619480349123, "learning_rate": 9.30134608199884e-09, "loss": 0.0020843889564275742, "reward": 3.1568170547485352, "reward_std": 0.439932119846344, "rewards/IngredientFormatReward/mean": 0.9773883819580078, "rewards/IngredientFormatReward/std": 0.12660152986645698, "rewards/IngredientMatchReward/mean": 0.6738641619682312, "rewards/IngredientMatchReward/std": 0.2843153178691864, "rewards/IngredientQuantityMatchReward/mean": 0.6915020108222961, "rewards/IngredientQuantityMatchReward/std": 0.40528308153152465, "rewards/TotalKcalExactMatchReward/mean": 0.8140625, "rewards/TotalKcalExactMatchReward/std": 0.3795959770679474, "step": 3385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 506.6, "completions/mean_length": 391.2890625, "completions/min_length": 265.2, "epoch": 0.9429763560500696, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6121615767478943, "kl": 0.04772463580593467, "learning_rate": 8.864995685504251e-09, "loss": 0.0019091980531811715, "reward": 2.93808856010437, "reward_std": 0.39786078333854674, "rewards/IngredientFormatReward/mean": 0.9845312595367431, "rewards/IngredientFormatReward/std": 0.0779263861477375, "rewards/IngredientMatchReward/mean": 0.5765872240066529, "rewards/IngredientMatchReward/std": 0.29070484042167666, "rewards/IngredientQuantityMatchReward/mean": 0.5816575646400451, "rewards/IngredientQuantityMatchReward/std": 0.42740336060523987, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.39443774819374083, "step": 3390 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 495.2, "completions/mean_length": 387.0484375, "completions/min_length": 239.2, "epoch": 0.9443671766342142, "frac_reward_zero_std": 0.0375, "grad_norm": 0.7277557849884033, "kl": 0.047656627488322556, "learning_rate": 8.439036315708691e-09, "loss": 0.001906503364443779, "reward": 3.1081774711608885, "reward_std": 0.3849943280220032, "rewards/IngredientFormatReward/mean": 0.9951562523841858, "rewards/IngredientFormatReward/std": 0.039984484761953355, "rewards/IngredientMatchReward/mean": 0.6685410618782044, "rewards/IngredientMatchReward/std": 0.2922262877225876, "rewards/IngredientQuantityMatchReward/mean": 0.6538552045822144, "rewards/IngredientQuantityMatchReward/std": 0.4168520152568817, "rewards/TotalKcalExactMatchReward/mean": 0.790625, "rewards/TotalKcalExactMatchReward/std": 0.3908423066139221, "step": 3395 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 509.4, "completions/mean_length": 391.184375, "completions/min_length": 278.2, "epoch": 0.9457579972183588, "frac_reward_zero_std": 0.075, "grad_norm": 0.6436322331428528, "kl": 0.04723382429219782, "learning_rate": 8.023476984706956e-09, "loss": 0.001889285072684288, "reward": 3.161758041381836, "reward_std": 0.3478512942790985, "rewards/IngredientFormatReward/mean": 0.9847854614257813, "rewards/IngredientFormatReward/std": 0.0928487703204155, "rewards/IngredientMatchReward/mean": 0.635439395904541, "rewards/IngredientMatchReward/std": 0.27520969808101653, "rewards/IngredientQuantityMatchReward/mean": 0.7024707198143005, "rewards/IngredientQuantityMatchReward/std": 0.3910308003425598, "rewards/TotalKcalExactMatchReward/mean": 0.8390625, "rewards/TotalKcalExactMatchReward/std": 0.3589689821004868, "step": 3400 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 507.6, "completions/mean_length": 386.3875, "completions/min_length": 266.6, "epoch": 0.9471488178025035, "frac_reward_zero_std": 0.025, "grad_norm": 0.7100093364715576, "kl": 0.049441801151260734, "learning_rate": 7.618326484558401e-09, "loss": 0.001977815106511116, "reward": 3.023116683959961, "reward_std": 0.3968947231769562, "rewards/IngredientFormatReward/mean": 0.9973437547683716, "rewards/IngredientFormatReward/std": 0.023174207657575607, "rewards/IngredientMatchReward/mean": 0.6149442076683045, "rewards/IngredientMatchReward/std": 0.2917324811220169, "rewards/IngredientQuantityMatchReward/mean": 0.6452036738395691, "rewards/IngredientQuantityMatchReward/std": 0.41264580488204955, "rewards/TotalKcalExactMatchReward/mean": 0.765625, "rewards/TotalKcalExactMatchReward/std": 0.416858971118927, "step": 3405 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 497.8, "completions/mean_length": 390.4578125, "completions/min_length": 275.0, "epoch": 0.9485396383866481, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6251751780509949, "kl": 0.046490383613854644, "learning_rate": 7.2235933871008795e-09, "loss": 0.0018598001450300216, "reward": 3.1199548721313475, "reward_std": 0.355280739068985, "rewards/IngredientFormatReward/mean": 0.994527530670166, "rewards/IngredientFormatReward/std": 0.040901456214487555, "rewards/IngredientMatchReward/mean": 0.6296986818313599, "rewards/IngredientMatchReward/std": 0.2927346646785736, "rewards/IngredientQuantityMatchReward/mean": 0.6769786238670349, "rewards/IngredientQuantityMatchReward/std": 0.4171420454978943, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.3693612217903137, "step": 3410 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 508.6, "completions/mean_length": 392.115625, "completions/min_length": 272.2, "epoch": 0.9499304589707928, "frac_reward_zero_std": 0.05, "grad_norm": 0.6708676815032959, "kl": 0.047037134994752705, "learning_rate": 6.839286043769654e-09, "loss": 0.0018814524635672569, "reward": 3.0176434993743895, "reward_std": 0.4097278594970703, "rewards/IngredientFormatReward/mean": 0.9863913774490356, "rewards/IngredientFormatReward/std": 0.07142920605838299, "rewards/IngredientMatchReward/mean": 0.5660413384437561, "rewards/IngredientMatchReward/std": 0.2902086913585663, "rewards/IngredientQuantityMatchReward/mean": 0.6245858073234558, "rewards/IngredientQuantityMatchReward/std": 0.429430627822876, "rewards/TotalKcalExactMatchReward/mean": 0.840625, "rewards/TotalKcalExactMatchReward/std": 0.3561762750148773, "step": 3415 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 495.2, "completions/mean_length": 387.44375, "completions/min_length": 258.8, "epoch": 0.9513212795549374, "frac_reward_zero_std": 0.05, "grad_norm": 0.6568620204925537, "kl": 0.04582350081764162, "learning_rate": 6.4654125854204374e-09, "loss": 0.0018331578001379966, "reward": 3.1052068710327148, "reward_std": 0.4515254557132721, "rewards/IngredientFormatReward/mean": 0.9904017925262452, "rewards/IngredientFormatReward/std": 0.05673844218254089, "rewards/IngredientMatchReward/mean": 0.6195003747940063, "rewards/IngredientMatchReward/std": 0.27675817608833314, "rewards/IngredientQuantityMatchReward/mean": 0.7015547275543212, "rewards/IngredientQuantityMatchReward/std": 0.40126625299453733, "rewards/TotalKcalExactMatchReward/mean": 0.79375, "rewards/TotalKcalExactMatchReward/std": 0.38862606287002566, "step": 3420 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 510.2, "completions/mean_length": 387.96875, "completions/min_length": 266.2, "epoch": 0.952712100139082, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6748784780502319, "kl": 0.0493265719152987, "learning_rate": 6.101980922157524e-09, "loss": 0.001973242685198784, "reward": 3.1050642013549803, "reward_std": 0.38432427048683165, "rewards/IngredientFormatReward/mean": 0.99296875, "rewards/IngredientFormatReward/std": 0.06092705130577088, "rewards/IngredientMatchReward/mean": 0.6292460322380066, "rewards/IngredientMatchReward/std": 0.28708921670913695, "rewards/IngredientQuantityMatchReward/mean": 0.6562868833541871, "rewards/IngredientQuantityMatchReward/std": 0.4298116385936737, "rewards/TotalKcalExactMatchReward/mean": 0.8265625, "rewards/TotalKcalExactMatchReward/std": 0.363370805978775, "step": 3425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 508.2, "completions/mean_length": 393.5328125, "completions/min_length": 271.0, "epoch": 0.9541029207232267, "frac_reward_zero_std": 0.0375, "grad_norm": 0.663008987903595, "kl": 0.048660345608368516, "learning_rate": 5.748998743166256e-09, "loss": 0.0019468117505311966, "reward": 3.1665676116943358, "reward_std": 0.3740094184875488, "rewards/IngredientFormatReward/mean": 0.9934895753860473, "rewards/IngredientFormatReward/std": 0.052749037928879264, "rewards/IngredientMatchReward/mean": 0.6485987067222595, "rewards/IngredientMatchReward/std": 0.2870753645896912, "rewards/IngredientQuantityMatchReward/mean": 0.7197918772697449, "rewards/IngredientQuantityMatchReward/std": 0.3927998900413513, "rewards/TotalKcalExactMatchReward/mean": 0.8046875, "rewards/TotalKcalExactMatchReward/std": 0.39596582651138307, "step": 3430 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 503.2, "completions/mean_length": 389.634375, "completions/min_length": 251.6, "epoch": 0.9554937413073713, "frac_reward_zero_std": 0.0375, "grad_norm": 0.65153968334198, "kl": 0.04435617581475526, "learning_rate": 5.406473516550603e-09, "loss": 0.001774563454091549, "reward": 3.1701682090759276, "reward_std": 0.3287068665027618, "rewards/IngredientFormatReward/mean": 0.9945163607597352, "rewards/IngredientFormatReward/std": 0.0458319054916501, "rewards/IngredientMatchReward/mean": 0.6381113409996033, "rewards/IngredientMatchReward/std": 0.29567062854766846, "rewards/IngredientQuantityMatchReward/mean": 0.6484779477119446, "rewards/IngredientQuantityMatchReward/std": 0.4112301588058472, "rewards/TotalKcalExactMatchReward/mean": 0.8890625, "rewards/TotalKcalExactMatchReward/std": 0.2916980266571045, "step": 3435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.8, "completions/mean_length": 395.403125, "completions/min_length": 261.4, "epoch": 0.9568845618915159, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6878206133842468, "kl": 0.04364196332171559, "learning_rate": 5.074412489174895e-09, "loss": 0.0017459928989410401, "reward": 3.17191801071167, "reward_std": 0.3969049096107483, "rewards/IngredientFormatReward/mean": 0.9871874809265136, "rewards/IngredientFormatReward/std": 0.093497334420681, "rewards/IngredientMatchReward/mean": 0.6249485611915588, "rewards/IngredientMatchReward/std": 0.2994285225868225, "rewards/IngredientQuantityMatchReward/mean": 0.7285319924354553, "rewards/IngredientQuantityMatchReward/std": 0.3931278049945831, "rewards/TotalKcalExactMatchReward/mean": 0.83125, "rewards/TotalKcalExactMatchReward/std": 0.37189761996269227, "step": 3440 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 503.8, "completions/mean_length": 390.659375, "completions/min_length": 263.8, "epoch": 0.9582753824756607, "frac_reward_zero_std": 0.05, "grad_norm": 0.6432104706764221, "kl": 0.04477330117952079, "learning_rate": 4.752822686510727e-09, "loss": 0.0017909301444888116, "reward": 3.060155248641968, "reward_std": 0.3750399827957153, "rewards/IngredientFormatReward/mean": 0.9862499952316284, "rewards/IngredientFormatReward/std": 0.07456732466816902, "rewards/IngredientMatchReward/mean": 0.6478366851806641, "rewards/IngredientMatchReward/std": 0.2967373102903366, "rewards/IngredientQuantityMatchReward/mean": 0.6291934847831726, "rewards/IngredientQuantityMatchReward/std": 0.42095019221305846, "rewards/TotalKcalExactMatchReward/mean": 0.796875, "rewards/TotalKcalExactMatchReward/std": 0.39375485181808473, "step": 3445 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 508.4, "completions/mean_length": 397.0875, "completions/min_length": 242.2, "epoch": 0.9596662030598053, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6475841403007507, "kl": 0.04543045272585004, "learning_rate": 4.441710912488239e-09, "loss": 0.0018170112743973732, "reward": 3.0488508701324464, "reward_std": 0.37055699825286864, "rewards/IngredientFormatReward/mean": 0.9900520801544189, "rewards/IngredientFormatReward/std": 0.08570180833339691, "rewards/IngredientMatchReward/mean": 0.6155338644981384, "rewards/IngredientMatchReward/std": 0.3055759608745575, "rewards/IngredientQuantityMatchReward/mean": 0.6213900089263916, "rewards/IngredientQuantityMatchReward/std": 0.4131564795970917, "rewards/TotalKcalExactMatchReward/mean": 0.821875, "rewards/TotalKcalExactMatchReward/std": 0.3545024037361145, "step": 3450 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 504.2, "completions/mean_length": 386.834375, "completions/min_length": 243.4, "epoch": 0.96105702364395, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6767144799232483, "kl": 0.04825890299398452, "learning_rate": 4.141083749351959e-09, "loss": 0.0019302651286125182, "reward": 3.10237512588501, "reward_std": 0.41013769507408143, "rewards/IngredientFormatReward/mean": 0.9851934432983398, "rewards/IngredientFormatReward/std": 0.10111989080905914, "rewards/IngredientMatchReward/mean": 0.6478317260742188, "rewards/IngredientMatchReward/std": 0.2991852521896362, "rewards/IngredientQuantityMatchReward/mean": 0.6787248849868774, "rewards/IngredientQuantityMatchReward/std": 0.40713335275650026, "rewards/TotalKcalExactMatchReward/mean": 0.790625, "rewards/TotalKcalExactMatchReward/std": 0.398161518573761, "step": 3455 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 507.8, "completions/mean_length": 402.75, "completions/min_length": 274.4, "epoch": 0.9624478442280946, "frac_reward_zero_std": 0.05, "grad_norm": 0.6345236897468567, "kl": 0.046873748395591976, "learning_rate": 3.8509475575219105e-09, "loss": 0.0018753422424197196, "reward": 3.0807044982910154, "reward_std": 0.426052987575531, "rewards/IngredientFormatReward/mean": 0.9768452405929565, "rewards/IngredientFormatReward/std": 0.11105858944356442, "rewards/IngredientMatchReward/mean": 0.628242838382721, "rewards/IngredientMatchReward/std": 0.3001261740922928, "rewards/IngredientQuantityMatchReward/mean": 0.6553040742874146, "rewards/IngredientQuantityMatchReward/std": 0.4021215379238129, "rewards/TotalKcalExactMatchReward/mean": 0.8203125, "rewards/TotalKcalExactMatchReward/std": 0.371417498588562, "step": 3460 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 505.0, "completions/mean_length": 388.6921875, "completions/min_length": 260.0, "epoch": 0.9638386648122392, "frac_reward_zero_std": 0.05, "grad_norm": 0.7613881826400757, "kl": 0.04958433983847499, "learning_rate": 3.571308475458723e-09, "loss": 0.0019834831357002257, "reward": 3.1493412971496584, "reward_std": 0.4270430862903595, "rewards/IngredientFormatReward/mean": 0.9888020753860474, "rewards/IngredientFormatReward/std": 0.0908542349934578, "rewards/IngredientMatchReward/mean": 0.6713510751724243, "rewards/IngredientMatchReward/std": 0.3045841991901398, "rewards/IngredientQuantityMatchReward/mean": 0.6938755869865417, "rewards/IngredientQuantityMatchReward/std": 0.4049495041370392, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.39825695753097534, "step": 3465 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 503.2, "completions/mean_length": 387.459375, "completions/min_length": 250.8, "epoch": 0.9652294853963839, "frac_reward_zero_std": 0.025, "grad_norm": 0.6586174964904785, "kl": 0.04571211233269423, "learning_rate": 3.3021724195340107e-09, "loss": 0.001828974112868309, "reward": 3.149783134460449, "reward_std": 0.4005468308925629, "rewards/IngredientFormatReward/mean": 0.9925000071525574, "rewards/IngredientFormatReward/std": 0.052525454014539716, "rewards/IngredientMatchReward/mean": 0.6844698548316955, "rewards/IngredientMatchReward/std": 0.2855565369129181, "rewards/IngredientQuantityMatchReward/mean": 0.6696882605552673, "rewards/IngredientQuantityMatchReward/std": 0.4112094700336456, "rewards/TotalKcalExactMatchReward/mean": 0.803125, "rewards/TotalKcalExactMatchReward/std": 0.3755046874284744, "step": 3470 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 507.8, "completions/mean_length": 388.734375, "completions/min_length": 249.2, "epoch": 0.9666203059805285, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6234393119812012, "kl": 0.049782629730179905, "learning_rate": 3.0435450839049194e-09, "loss": 0.001991119608283043, "reward": 3.1384727478027346, "reward_std": 0.34530335664749146, "rewards/IngredientFormatReward/mean": 0.9880580425262451, "rewards/IngredientFormatReward/std": 0.07789933532476426, "rewards/IngredientMatchReward/mean": 0.6632006525993347, "rewards/IngredientMatchReward/std": 0.27832908630371095, "rewards/IngredientQuantityMatchReward/mean": 0.637213945388794, "rewards/IngredientQuantityMatchReward/std": 0.41195704936981203, "rewards/TotalKcalExactMatchReward/mean": 0.85, "rewards/TotalKcalExactMatchReward/std": 0.32829630076885224, "step": 3475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 505.0, "completions/mean_length": 385.3453125, "completions/min_length": 241.8, "epoch": 0.9680111265646731, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6432631015777588, "kl": 0.747702275426127, "learning_rate": 2.7954319403940553e-09, "loss": 0.029897397756576537, "reward": 3.144414186477661, "reward_std": 0.40637815594673155, "rewards/IngredientFormatReward/mean": 0.9936718821525574, "rewards/IngredientFormatReward/std": 0.05400313213467598, "rewards/IngredientMatchReward/mean": 0.6715754985809326, "rewards/IngredientMatchReward/std": 0.28783435225486753, "rewards/IngredientQuantityMatchReward/mean": 0.708854329586029, "rewards/IngredientQuantityMatchReward/std": 0.3918882191181183, "rewards/TotalKcalExactMatchReward/mean": 0.7703125, "rewards/TotalKcalExactMatchReward/std": 0.40544432401657104, "step": 3480 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 503.6, "completions/mean_length": 381.9390625, "completions/min_length": 245.4, "epoch": 0.9694019471488178, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6335321664810181, "kl": 0.04744428307749331, "learning_rate": 2.5578382383732442e-09, "loss": 0.0018980761989951134, "reward": 3.157121992111206, "reward_std": 0.3908619523048401, "rewards/IngredientFormatReward/mean": 0.9844103455543518, "rewards/IngredientFormatReward/std": 0.08667321354150773, "rewards/IngredientMatchReward/mean": 0.6567723035812378, "rewards/IngredientMatchReward/std": 0.3101677715778351, "rewards/IngredientQuantityMatchReward/mean": 0.6675018310546875, "rewards/IngredientQuantityMatchReward/std": 0.43211646676063536, "rewards/TotalKcalExactMatchReward/mean": 0.8484375, "rewards/TotalKcalExactMatchReward/std": 0.35224978923797606, "step": 3485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 508.8, "completions/mean_length": 389.928125, "completions/min_length": 236.0, "epoch": 0.9707927677329624, "frac_reward_zero_std": 0.075, "grad_norm": 0.6075333952903748, "kl": 0.04749975092709065, "learning_rate": 2.3307690046527883e-09, "loss": 0.00190051831305027, "reward": 3.052085304260254, "reward_std": 0.3895645797252655, "rewards/IngredientFormatReward/mean": 0.9751562595367431, "rewards/IngredientFormatReward/std": 0.12122566476464272, "rewards/IngredientMatchReward/mean": 0.6185956478118897, "rewards/IngredientMatchReward/std": 0.26451897621154785, "rewards/IngredientQuantityMatchReward/mean": 0.6395834565162659, "rewards/IngredientQuantityMatchReward/std": 0.4328662157058716, "rewards/TotalKcalExactMatchReward/mean": 0.81875, "rewards/TotalKcalExactMatchReward/std": 0.37133659422397614, "step": 3490 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 507.6, "completions/mean_length": 388.0703125, "completions/min_length": 274.4, "epoch": 0.972183588317107, "frac_reward_zero_std": 0.075, "grad_norm": 0.6728726625442505, "kl": 0.047019163495860994, "learning_rate": 2.1142290433750486e-09, "loss": 0.0018808633089065553, "reward": 3.1073230266571046, "reward_std": 0.367638099193573, "rewards/IngredientFormatReward/mean": 0.9906510472297668, "rewards/IngredientFormatReward/std": 0.07325609177350997, "rewards/IngredientMatchReward/mean": 0.6417634010314941, "rewards/IngredientMatchReward/std": 0.28443962931632993, "rewards/IngredientQuantityMatchReward/mean": 0.6686586856842041, "rewards/IngredientQuantityMatchReward/std": 0.41565218567848206, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.3815740138292313, "step": 3495 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 502.0, "completions/mean_length": 383.71875, "completions/min_length": 259.0, "epoch": 0.9735744089012517, "frac_reward_zero_std": 0.0625, "grad_norm": 0.7703426480293274, "kl": 0.04528216728940606, "learning_rate": 1.908222935912751e-09, "loss": 0.001811334304511547, "reward": 3.0197416305541993, "reward_std": 0.4097736120223999, "rewards/IngredientFormatReward/mean": 0.9861718654632569, "rewards/IngredientFormatReward/std": 0.06792713664472103, "rewards/IngredientMatchReward/mean": 0.6406938433647156, "rewards/IngredientMatchReward/std": 0.2943361341953278, "rewards/IngredientQuantityMatchReward/mean": 0.57568838596344, "rewards/IngredientQuantityMatchReward/std": 0.4376201629638672, "rewards/TotalKcalExactMatchReward/mean": 0.8171875, "rewards/TotalKcalExactMatchReward/std": 0.3830027997493744, "step": 3500 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 501.6, "completions/mean_length": 388.3625, "completions/min_length": 245.2, "epoch": 0.9749652294853964, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6825488805770874, "kl": 0.052455898956395686, "learning_rate": 1.7127550407721181e-09, "loss": 0.0020987868309020998, "reward": 3.1052551746368406, "reward_std": 0.421875661611557, "rewards/IngredientFormatReward/mean": 0.9838802099227906, "rewards/IngredientFormatReward/std": 0.10591072142124176, "rewards/IngredientMatchReward/mean": 0.6220045924186707, "rewards/IngredientMatchReward/std": 0.2856017410755157, "rewards/IngredientQuantityMatchReward/mean": 0.6649953603744507, "rewards/IngredientQuantityMatchReward/std": 0.4061083495616913, "rewards/TotalKcalExactMatchReward/mean": 0.834375, "rewards/TotalKcalExactMatchReward/std": 0.3393280774354935, "step": 3505 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.003125, "completions/max_length": 502.8, "completions/mean_length": 390.6109375, "completions/min_length": 269.4, "epoch": 0.9763560500695411, "frac_reward_zero_std": 0.05, "grad_norm": 0.6667228937149048, "kl": 0.04870023645926267, "learning_rate": 1.5278294935006098e-09, "loss": 0.0019479893147945405, "reward": 3.113843631744385, "reward_std": 0.35846092104911803, "rewards/IngredientFormatReward/mean": 0.9938801884651184, "rewards/IngredientFormatReward/std": 0.05015102569013834, "rewards/IngredientMatchReward/mean": 0.6636923670768737, "rewards/IngredientMatchReward/std": 0.27049159705638887, "rewards/IngredientQuantityMatchReward/mean": 0.6203335046768188, "rewards/IngredientQuantityMatchReward/std": 0.41901023387908937, "rewards/TotalKcalExactMatchReward/mean": 0.8359375, "rewards/TotalKcalExactMatchReward/std": 0.3319972887635231, "step": 3510 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 512.4, "completions/mean_length": 389.965625, "completions/min_length": 265.8, "epoch": 0.9777468706536857, "frac_reward_zero_std": 0.1125, "grad_norm": 0.7143624424934387, "kl": 0.046095013013109566, "learning_rate": 1.3534502065993826e-09, "loss": 0.001844063401222229, "reward": 3.1194585800170898, "reward_std": 0.45072959661483764, "rewards/IngredientFormatReward/mean": 0.987109375, "rewards/IngredientFormatReward/std": 0.09570224285125732, "rewards/IngredientMatchReward/mean": 0.6516096472740174, "rewards/IngredientMatchReward/std": 0.29113835096359253, "rewards/IngredientQuantityMatchReward/mean": 0.6744894981384277, "rewards/IngredientQuantityMatchReward/std": 0.40634567141532896, "rewards/TotalKcalExactMatchReward/mean": 0.80625, "rewards/TotalKcalExactMatchReward/std": 0.3906187415122986, "step": 3515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 493.0, "completions/mean_length": 382.9546875, "completions/min_length": 244.8, "epoch": 0.9791376912378303, "frac_reward_zero_std": 0.0375, "grad_norm": 0.683193027973175, "kl": 0.050947370985522863, "learning_rate": 1.1896208694406884e-09, "loss": 0.002037731558084488, "reward": 3.0674956798553468, "reward_std": 0.45058631896972656, "rewards/IngredientFormatReward/mean": 0.9909598112106324, "rewards/IngredientFormatReward/std": 0.08356152959167958, "rewards/IngredientMatchReward/mean": 0.6089496612548828, "rewards/IngredientMatchReward/std": 0.29325642585754397, "rewards/IngredientQuantityMatchReward/mean": 0.6363362312316895, "rewards/IngredientQuantityMatchReward/std": 0.4325535476207733, "rewards/TotalKcalExactMatchReward/mean": 0.83125, "rewards/TotalKcalExactMatchReward/std": 0.3584339439868927, "step": 3520 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0078125, "completions/max_length": 504.4, "completions/mean_length": 396.371875, "completions/min_length": 274.4, "epoch": 0.980528511821975, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6610113382339478, "kl": 0.04606618359684944, "learning_rate": 1.0363449481896603e-09, "loss": 0.0018428176641464233, "reward": 3.0702935695648192, "reward_std": 0.3585042953491211, "rewards/IngredientFormatReward/mean": 0.9895312428474426, "rewards/IngredientFormatReward/std": 0.0739785360172391, "rewards/IngredientMatchReward/mean": 0.647214138507843, "rewards/IngredientMatchReward/std": 0.2919449031352997, "rewards/IngredientQuantityMatchReward/mean": 0.650735592842102, "rewards/IngredientQuantityMatchReward/std": 0.38301952481269835, "rewards/TotalKcalExactMatchReward/mean": 0.7828125, "rewards/TotalKcalExactMatchReward/std": 0.4121102452278137, "step": 3525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 494.4, "completions/mean_length": 381.7484375, "completions/min_length": 251.4, "epoch": 0.9819193324061196, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6419509053230286, "kl": 0.04725923398509622, "learning_rate": 8.9362568573087e-10, "loss": 0.0018903814256191255, "reward": 3.1441524028778076, "reward_std": 0.4015710175037384, "rewards/IngredientFormatReward/mean": 0.993958342075348, "rewards/IngredientFormatReward/std": 0.04699918553233147, "rewards/IngredientMatchReward/mean": 0.6546136975288391, "rewards/IngredientMatchReward/std": 0.29563533067703246, "rewards/IngredientQuantityMatchReward/mean": 0.7018303513526917, "rewards/IngredientQuantityMatchReward/std": 0.3950989067554474, "rewards/TotalKcalExactMatchReward/mean": 0.79375, "rewards/TotalKcalExactMatchReward/std": 0.39758574962615967, "step": 3530 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 511.4, "completions/mean_length": 390.678125, "completions/min_length": 256.0, "epoch": 0.9833101529902643, "frac_reward_zero_std": 0.075, "grad_norm": 0.5843694806098938, "kl": 0.05070257929619402, "learning_rate": 7.614661016001056e-10, "loss": 0.002028765343129635, "reward": 3.0022071838378905, "reward_std": 0.4141098916530609, "rewards/IngredientFormatReward/mean": 0.9861049175262451, "rewards/IngredientFormatReward/std": 0.11176733374595642, "rewards/IngredientMatchReward/mean": 0.6246558666229248, "rewards/IngredientMatchReward/std": 0.3036438703536987, "rewards/IngredientQuantityMatchReward/mean": 0.6586337327957154, "rewards/IngredientQuantityMatchReward/std": 0.41347108483314515, "rewards/TotalKcalExactMatchReward/mean": 0.7328125, "rewards/TotalKcalExactMatchReward/std": 0.4390223383903503, "step": 3535 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 508.6, "completions/mean_length": 393.465625, "completions/min_length": 254.4, "epoch": 0.9847009735744089, "frac_reward_zero_std": 0.1375, "grad_norm": 0.808208703994751, "kl": 0.05155458953231573, "learning_rate": 6.398689919201449e-10, "loss": 0.0020628392696380613, "reward": 3.1207000732421877, "reward_std": 0.3916473984718323, "rewards/IngredientFormatReward/mean": 0.9782217264175415, "rewards/IngredientFormatReward/std": 0.1250761903822422, "rewards/IngredientMatchReward/mean": 0.6547699689865112, "rewards/IngredientMatchReward/std": 0.2909015595912933, "rewards/IngredientQuantityMatchReward/mean": 0.7017708659172058, "rewards/IngredientQuantityMatchReward/std": 0.3939378708600998, "rewards/TotalKcalExactMatchReward/mean": 0.7859375, "rewards/TotalKcalExactMatchReward/std": 0.40576318502426145, "step": 3540 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 511.6, "completions/mean_length": 389.2640625, "completions/min_length": 263.8, "epoch": 0.9860917941585535, "frac_reward_zero_std": 0.1, "grad_norm": 0.6409344673156738, "kl": 0.04587732988875359, "learning_rate": 5.288369293415807e-10, "loss": 0.0018349820747971535, "reward": 3.138898420333862, "reward_std": 0.4375850737094879, "rewards/IngredientFormatReward/mean": 0.987485122680664, "rewards/IngredientFormatReward/std": 0.09236232042312623, "rewards/IngredientMatchReward/mean": 0.6686526775360108, "rewards/IngredientMatchReward/std": 0.2893941581249237, "rewards/IngredientQuantityMatchReward/mean": 0.6874481797218323, "rewards/IngredientQuantityMatchReward/std": 0.40284740924835205, "rewards/TotalKcalExactMatchReward/mean": 0.7953125, "rewards/TotalKcalExactMatchReward/std": 0.39978498220443726, "step": 3545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 507.4, "completions/mean_length": 391.2296875, "completions/min_length": 273.2, "epoch": 0.9874826147426982, "frac_reward_zero_std": 0.05, "grad_norm": 0.6869775652885437, "kl": 0.048427975224331024, "learning_rate": 4.2837226298875206e-10, "loss": 0.0019370142370462417, "reward": 3.165076398849487, "reward_std": 0.41075287461280824, "rewards/IngredientFormatReward/mean": 0.9800595283508301, "rewards/IngredientFormatReward/std": 0.12871189117431642, "rewards/IngredientMatchReward/mean": 0.657729423046112, "rewards/IngredientMatchReward/std": 0.2861269056797028, "rewards/IngredientQuantityMatchReward/mean": 0.7179124355316162, "rewards/IngredientQuantityMatchReward/std": 0.3858673632144928, "rewards/TotalKcalExactMatchReward/mean": 0.809375, "rewards/TotalKcalExactMatchReward/std": 0.3873170852661133, "step": 3550 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0015625, "completions/max_length": 495.4, "completions/mean_length": 382.4734375, "completions/min_length": 269.8, "epoch": 0.9888734353268428, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6728901863098145, "kl": 0.04554827213287353, "learning_rate": 3.384771184095081e-10, "loss": 0.0018219701945781709, "reward": 3.1922850608825684, "reward_std": 0.38457239866256715, "rewards/IngredientFormatReward/mean": 0.996875, "rewards/IngredientFormatReward/std": 0.03535533845424652, "rewards/IngredientMatchReward/mean": 0.7049667954444885, "rewards/IngredientMatchReward/std": 0.2878182172775269, "rewards/IngredientQuantityMatchReward/mean": 0.652943241596222, "rewards/IngredientQuantityMatchReward/std": 0.4103390693664551, "rewards/TotalKcalExactMatchReward/mean": 0.8375, "rewards/TotalKcalExactMatchReward/std": 0.36109898686409, "step": 3555 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 506.8, "completions/mean_length": 390.6703125, "completions/min_length": 256.2, "epoch": 0.9902642559109874, "frac_reward_zero_std": 0.075, "grad_norm": 7.2769775390625, "kl": 0.0716410128865391, "learning_rate": 2.5915339753085354e-10, "loss": 0.0028650620952248573, "reward": 3.1179696559906005, "reward_std": 0.3656330227851868, "rewards/IngredientFormatReward/mean": 0.9906473159790039, "rewards/IngredientFormatReward/std": 0.0729976112022996, "rewards/IngredientMatchReward/mean": 0.6654321789741516, "rewards/IngredientMatchReward/std": 0.30215807557106017, "rewards/IngredientQuantityMatchReward/mean": 0.6509526252746582, "rewards/IngredientQuantityMatchReward/std": 0.42949780225753786, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.38284428119659425, "step": 3560 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0140625, "completions/max_length": 508.8, "completions/mean_length": 384.6453125, "completions/min_length": 228.2, "epoch": 0.9916550764951322, "frac_reward_zero_std": 0.1, "grad_norm": 0.6099084615707397, "kl": 0.04533320358023048, "learning_rate": 1.9040277861814835e-10, "loss": 0.0018132233992218972, "reward": 3.1440773010253906, "reward_std": 0.3812756478786469, "rewards/IngredientFormatReward/mean": 0.9850111722946167, "rewards/IngredientFormatReward/std": 0.10009037107229232, "rewards/IngredientMatchReward/mean": 0.646268618106842, "rewards/IngredientMatchReward/std": 0.28203052580356597, "rewards/IngredientQuantityMatchReward/mean": 0.6581100583076477, "rewards/IngredientQuantityMatchReward/std": 0.39739391803741453, "rewards/TotalKcalExactMatchReward/mean": 0.8546875, "rewards/TotalKcalExactMatchReward/std": 0.34881080985069274, "step": 3565 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 513.0, "completions/mean_length": 392.1140625, "completions/min_length": 230.4, "epoch": 0.9930458970792768, "frac_reward_zero_std": 0.1125, "grad_norm": 0.6250317096710205, "kl": 0.04590991784352809, "learning_rate": 1.3222671623991377e-10, "loss": 0.0018364254385232926, "reward": 3.1440375804901124, "reward_std": 0.4134489595890045, "rewards/IngredientFormatReward/mean": 0.9813243865966796, "rewards/IngredientFormatReward/std": 0.12313078790903091, "rewards/IngredientMatchReward/mean": 0.6408395528793335, "rewards/IngredientMatchReward/std": 0.29529817700386046, "rewards/IngredientQuantityMatchReward/mean": 0.6984362483024598, "rewards/IngredientQuantityMatchReward/std": 0.41224990487098695, "rewards/TotalKcalExactMatchReward/mean": 0.8234375, "rewards/TotalKcalExactMatchReward/std": 0.38107048273086547, "step": 3570 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 503.4, "completions/mean_length": 392.796875, "completions/min_length": 257.0, "epoch": 0.9944367176634215, "frac_reward_zero_std": 0.0875, "grad_norm": 0.6811825633049011, "kl": 0.04669976490549743, "learning_rate": 8.462644123696794e-11, "loss": 0.001867879182100296, "reward": 3.0588041305541993, "reward_std": 0.4007099688053131, "rewards/IngredientFormatReward/mean": 0.992005217075348, "rewards/IngredientFormatReward/std": 0.06722570732235908, "rewards/IngredientMatchReward/mean": 0.6472081303596496, "rewards/IngredientMatchReward/std": 0.29389954805374147, "rewards/IngredientQuantityMatchReward/mean": 0.6477158069610596, "rewards/IngredientQuantityMatchReward/std": 0.4318304181098938, "rewards/TotalKcalExactMatchReward/mean": 0.771875, "rewards/TotalKcalExactMatchReward/std": 0.41408060789108275, "step": 3575 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0046875, "completions/max_length": 510.0, "completions/mean_length": 387.1984375, "completions/min_length": 256.8, "epoch": 0.9958275382475661, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6889552474021912, "kl": 0.04746832814998925, "learning_rate": 4.7602960696391246e-11, "loss": 0.001899052783846855, "reward": 3.145450496673584, "reward_std": 0.3926717579364777, "rewards/IngredientFormatReward/mean": 0.9950000047683716, "rewards/IngredientFormatReward/std": 0.05335577577352524, "rewards/IngredientMatchReward/mean": 0.6480660915374756, "rewards/IngredientMatchReward/std": 0.2908414363861084, "rewards/IngredientQuantityMatchReward/mean": 0.6492594242095947, "rewards/IngredientQuantityMatchReward/std": 0.4212400853633881, "rewards/TotalKcalExactMatchReward/mean": 0.853125, "rewards/TotalKcalExactMatchReward/std": 0.3392543256282806, "step": 3580 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.009375, "completions/max_length": 509.2, "completions/mean_length": 387.4125, "completions/min_length": 249.2, "epoch": 0.9972183588317107, "frac_reward_zero_std": 0.025, "grad_norm": 0.7055279612541199, "kl": 0.04739529467187822, "learning_rate": 2.1157057930321077e-11, "loss": 0.0018960034474730491, "reward": 3.0336164474487304, "reward_std": 0.4345869064331055, "rewards/IngredientFormatReward/mean": 0.9876562476158142, "rewards/IngredientFormatReward/std": 0.0809405880048871, "rewards/IngredientMatchReward/mean": 0.6347569465637207, "rewards/IngredientMatchReward/std": 0.2958140969276428, "rewards/IngredientQuantityMatchReward/mean": 0.6315157651901245, "rewards/IngredientQuantityMatchReward/std": 0.42673113346099856, "rewards/TotalKcalExactMatchReward/mean": 0.7796875, "rewards/TotalKcalExactMatchReward/std": 0.40893966555595396, "step": 3585 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0109375, "completions/max_length": 508.8, "completions/mean_length": 389.1203125, "completions/min_length": 255.8, "epoch": 0.9986091794158554, "frac_reward_zero_std": 0.0625, "grad_norm": 0.6596475839614868, "kl": 0.05180376642383635, "learning_rate": 5.289292459187411e-12, "loss": 0.002072012796998024, "reward": 3.072105598449707, "reward_std": 0.37645214796066284, "rewards/IngredientFormatReward/mean": 0.9853515744209289, "rewards/IngredientFormatReward/std": 0.0902503028512001, "rewards/IngredientMatchReward/mean": 0.6567869186401367, "rewards/IngredientMatchReward/std": 0.28849694728851316, "rewards/IngredientQuantityMatchReward/mean": 0.6143420994281769, "rewards/IngredientQuantityMatchReward/std": 0.4065409183502197, "rewards/TotalKcalExactMatchReward/mean": 0.815625, "rewards/TotalKcalExactMatchReward/std": 0.37442034780979155, "step": 3590 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.00625, "completions/max_length": 508.0, "completions/mean_length": 393.1953125, "completions/min_length": 273.2, "epoch": 1.0, "frac_reward_zero_std": 0.0375, "grad_norm": 0.6491743326187134, "kl": 0.046436529234051706, "learning_rate": 0.0, "loss": 0.0018576562404632568, "reward": 3.1264251708984374, "reward_std": 0.3975203990936279, "rewards/IngredientFormatReward/mean": 0.988683032989502, "rewards/IngredientFormatReward/std": 0.07184891402721405, "rewards/IngredientMatchReward/mean": 0.6666737198829651, "rewards/IngredientMatchReward/std": 0.2867902278900146, "rewards/IngredientQuantityMatchReward/mean": 0.6601309657096863, "rewards/IngredientQuantityMatchReward/std": 0.4129492461681366, "rewards/TotalKcalExactMatchReward/mean": 0.8109375, "rewards/TotalKcalExactMatchReward/std": 0.38998509049415586, "step": 3595 } ], "logging_steps": 5, "max_steps": 3595, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 1, "trial_name": null, "trial_params": null }