{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.046875, "eval_steps": 10000, "global_step": 960, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2515625, "completions/max_length": 896.0, "completions/max_terminated_length": 800.6, "completions/mean_length": 444.8359375, "completions/mean_terminated_length": 293.9584396362305, "completions/min_length": 18.6, "completions/min_terminated_length": 18.6, "entropy": 2.1421017169952394, "epoch": 0.00048828125, "frac_reward_zero_std": 0.58125, "grad_norm": 8.75, "learning_rate": 9.925e-07, "loss": 0.5706, "num_tokens": 389999.0, "reward": 0.014375000121071934, "reward_std": 0.024233440216630698, "rewards/countdown_reward/mean": 0.014375000121071934, "rewards/countdown_reward/std": 0.04283876046538353, "step": 10, "step_time": 26.44712580945343 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.240625, "completions/max_length": 896.0, "completions/max_terminated_length": 856.9, "completions/mean_length": 440.240625, "completions/mean_terminated_length": 296.003157043457, "completions/min_length": 9.2, "completions/min_terminated_length": 9.2, "entropy": 2.083886134624481, "epoch": 0.0009765625, "frac_reward_zero_std": 0.46875, "grad_norm": 7.75, "learning_rate": 9.841666666666666e-07, "loss": 0.5938, "num_tokens": 777029.0, "reward": 0.020000000298023225, "reward_std": 0.03310603573918343, "rewards/countdown_reward/mean": 0.020000000298023225, "rewards/countdown_reward/std": 0.05562963094562292, "step": 20, "step_time": 25.560097302123904 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2234375, "completions/max_length": 896.0, "completions/max_terminated_length": 839.3, "completions/mean_length": 418.6140625, "completions/mean_terminated_length": 281.0589340209961, "completions/min_length": 9.4, "completions/min_terminated_length": 9.4, "entropy": 2.1231595039367677, "epoch": 0.00146484375, "frac_reward_zero_std": 0.5375, "grad_norm": 9.5, "learning_rate": 9.758333333333332e-07, "loss": 0.5994, "num_tokens": 1150222.0, "reward": 0.014843750325962902, "reward_std": 0.0238501594401896, "rewards/countdown_reward/mean": 0.014843750325962902, "rewards/countdown_reward/std": 0.03515603132545948, "step": 30, "step_time": 25.725066986307503 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.20625, "completions/max_length": 896.0, "completions/max_terminated_length": 840.2, "completions/mean_length": 410.6921875, "completions/mean_terminated_length": 284.03387756347655, "completions/min_length": 12.8, "completions/min_terminated_length": 12.8, "entropy": 1.8003376007080079, "epoch": 0.001953125, "frac_reward_zero_std": 0.4375, "grad_norm": 6.90625, "learning_rate": 9.675e-07, "loss": 0.6123, "num_tokens": 1518313.0, "reward": 0.020937500335276128, "reward_std": 0.03471687939018011, "rewards/countdown_reward/mean": 0.020937500335276128, "rewards/countdown_reward/std": 0.05617134161293506, "step": 40, "step_time": 25.52387049738318 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2328125, "completions/max_length": 896.0, "completions/max_terminated_length": 796.2, "completions/mean_length": 434.109375, "completions/mean_terminated_length": 293.4788452148438, "completions/min_length": 16.2, "completions/min_terminated_length": 16.2, "entropy": 1.8691810965538025, "epoch": 0.00244140625, "frac_reward_zero_std": 0.43125, "grad_norm": 7.3125, "learning_rate": 9.591666666666667e-07, "loss": 0.6038, "num_tokens": 1901379.0, "reward": 0.021406250447034834, "reward_std": 0.035126067139208315, "rewards/countdown_reward/mean": 0.021406250447034834, "rewards/countdown_reward/std": 0.056653326377272606, "step": 50, "step_time": 25.44698301460594 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2234375, "completions/max_length": 896.0, "completions/max_terminated_length": 846.4, "completions/mean_length": 425.2890625, "completions/mean_terminated_length": 288.93585815429685, "completions/min_length": 12.2, "completions/min_terminated_length": 12.2, "entropy": 1.8126930356025697, "epoch": 0.0029296875, "frac_reward_zero_std": 0.4875, "grad_norm": 7.6875, "learning_rate": 9.508333333333333e-07, "loss": 0.58, "num_tokens": 2278840.0, "reward": 0.01937500052154064, "reward_std": 0.031982014793902634, "rewards/countdown_reward/mean": 0.01937500052154064, "rewards/countdown_reward/std": 0.05487028174102306, "step": 60, "step_time": 25.180531426519156 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2453125, "completions/max_length": 896.0, "completions/max_terminated_length": 818.2, "completions/mean_length": 427.7109375, "completions/mean_terminated_length": 275.73370056152345, "completions/min_length": 15.2, "completions/min_terminated_length": 15.2, "entropy": 1.6942297577857972, "epoch": 0.00341796875, "frac_reward_zero_std": 0.475, "grad_norm": 9.875, "learning_rate": 9.425e-07, "loss": 0.5833, "num_tokens": 2657855.0, "reward": 0.01796875037252903, "reward_std": 0.027361910790205002, "rewards/countdown_reward/mean": 0.01796875037252903, "rewards/countdown_reward/std": 0.03823279850184917, "step": 70, "step_time": 25.362101176008583 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.253125, "completions/max_length": 896.0, "completions/max_terminated_length": 828.2, "completions/mean_length": 429.5296875, "completions/mean_terminated_length": 272.2754821777344, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 1.7523829221725464, "epoch": 0.00390625, "frac_reward_zero_std": 0.45, "grad_norm": 8.5, "learning_rate": 9.341666666666667e-07, "loss": 0.5849, "num_tokens": 3038066.0, "reward": 0.02109375037252903, "reward_std": 0.033946847356855867, "rewards/countdown_reward/mean": 0.02109375037252903, "rewards/countdown_reward/std": 0.05644123367965222, "step": 80, "step_time": 25.39157868605107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1921875, "completions/max_length": 896.0, "completions/max_terminated_length": 862.5, "completions/mean_length": 410.4734375, "completions/mean_terminated_length": 295.13673095703126, "completions/min_length": 17.7, "completions/min_terminated_length": 17.7, "entropy": 1.8237345933914184, "epoch": 0.00439453125, "frac_reward_zero_std": 0.4125, "grad_norm": 8.0, "learning_rate": 9.258333333333333e-07, "loss": 0.581, "num_tokens": 3406093.0, "reward": 0.02140625035390258, "reward_std": 0.03330626599490642, "rewards/countdown_reward/mean": 0.02140625035390258, "rewards/countdown_reward/std": 0.04870203658938408, "step": 90, "step_time": 25.476102899201216 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1796875, "completions/max_length": 896.0, "completions/max_terminated_length": 820.8, "completions/mean_length": 390.85625, "completions/mean_terminated_length": 280.08765106201173, "completions/min_length": 11.1, "completions/min_terminated_length": 11.1, "entropy": 1.7234050035476685, "epoch": 0.0048828125, "frac_reward_zero_std": 0.36875, "grad_norm": 7.59375, "learning_rate": 9.174999999999999e-07, "loss": 0.5908, "num_tokens": 3761581.0, "reward": 0.02468750039115548, "reward_std": 0.04102207776159048, "rewards/countdown_reward/mean": 0.02468750039115548, "rewards/countdown_reward/std": 0.06275036595761777, "step": 100, "step_time": 24.768843121640383 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1859375, "completions/max_length": 896.0, "completions/max_terminated_length": 833.8, "completions/mean_length": 396.653125, "completions/mean_terminated_length": 282.8854248046875, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 1.611741542816162, "epoch": 0.00537109375, "frac_reward_zero_std": 0.3875, "grad_norm": 6.59375, "learning_rate": 9.091666666666666e-07, "loss": 0.608, "num_tokens": 4120719.0, "reward": 0.021250000596046446, "reward_std": 0.031688567250967026, "rewards/countdown_reward/mean": 0.021250000596046446, "rewards/countdown_reward/std": 0.04107333235442638, "step": 110, "step_time": 25.29070850703865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 896.0, "completions/max_terminated_length": 843.8, "completions/mean_length": 372.2671875, "completions/mean_terminated_length": 269.8088851928711, "completions/min_length": 10.6, "completions/min_terminated_length": 10.6, "entropy": 1.7046252846717835, "epoch": 0.005859375, "frac_reward_zero_std": 0.30625, "grad_norm": 8.625, "learning_rate": 9.008333333333333e-07, "loss": 0.607, "num_tokens": 4464170.0, "reward": 0.029375000670552254, "reward_std": 0.04725649729371071, "rewards/countdown_reward/mean": 0.029375000670552254, "rewards/countdown_reward/std": 0.07697650752961635, "step": 120, "step_time": 24.72708419151604 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1734375, "completions/max_length": 896.0, "completions/max_terminated_length": 825.5, "completions/mean_length": 393.475, "completions/mean_terminated_length": 288.0066772460938, "completions/min_length": 16.6, "completions/min_terminated_length": 16.6, "entropy": 1.7251879692077636, "epoch": 0.00634765625, "frac_reward_zero_std": 0.3375, "grad_norm": 8.0, "learning_rate": 8.924999999999999e-07, "loss": 0.6226, "num_tokens": 4821278.0, "reward": 0.027343750931322575, "reward_std": 0.042874642089009284, "rewards/countdown_reward/mean": 0.027343750186264514, "rewards/countdown_reward/std": 0.06810757033526897, "step": 130, "step_time": 24.933662101067604 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1640625, "completions/max_length": 896.0, "completions/max_terminated_length": 822.1, "completions/mean_length": 373.771875, "completions/mean_terminated_length": 270.4602966308594, "completions/min_length": 11.4, "completions/min_terminated_length": 11.4, "entropy": 1.6498929500579833, "epoch": 0.0068359375, "frac_reward_zero_std": 0.30625, "grad_norm": 7.09375, "learning_rate": 8.841666666666666e-07, "loss": 0.6355, "num_tokens": 5165796.0, "reward": 0.0340625012293458, "reward_std": 0.05313007272779942, "rewards/countdown_reward/mean": 0.0340625012293458, "rewards/countdown_reward/std": 0.09085043743252755, "step": 140, "step_time": 25.311085244268178 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.159375, "completions/max_length": 896.0, "completions/max_terminated_length": 854.9, "completions/mean_length": 383.7890625, "completions/mean_terminated_length": 286.57145538330076, "completions/min_length": 12.2, "completions/min_terminated_length": 12.2, "entropy": 1.4584159970283508, "epoch": 0.00732421875, "frac_reward_zero_std": 0.31875, "grad_norm": 8.875, "learning_rate": 8.758333333333333e-07, "loss": 0.606, "num_tokens": 5516781.0, "reward": 0.033593750186264516, "reward_std": 0.04987223707139492, "rewards/countdown_reward/mean": 0.033593750186264516, "rewards/countdown_reward/std": 0.08711243607103825, "step": 150, "step_time": 25.26378392558545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.14375, "completions/max_length": 896.0, "completions/max_terminated_length": 839.4, "completions/mean_length": 369.4796875, "completions/mean_terminated_length": 282.17943420410154, "completions/min_length": 15.6, "completions/min_terminated_length": 15.6, "entropy": 1.4469455242156983, "epoch": 0.0078125, "frac_reward_zero_std": 0.25, "grad_norm": 7.03125, "learning_rate": 8.675000000000001e-07, "loss": 0.6147, "num_tokens": 5858584.0, "reward": 0.038437500968575476, "reward_std": 0.058644673228263854, "rewards/countdown_reward/mean": 0.038437500968575476, "rewards/countdown_reward/std": 0.09334000907838344, "step": 160, "step_time": 25.13749391809106 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.16875, "completions/max_length": 896.0, "completions/max_terminated_length": 783.2, "completions/mean_length": 370.765625, "completions/mean_terminated_length": 263.9585296630859, "completions/min_length": 18.4, "completions/min_terminated_length": 18.4, "entropy": 1.333325695991516, "epoch": 0.00830078125, "frac_reward_zero_std": 0.23125, "grad_norm": 7.65625, "learning_rate": 8.591666666666666e-07, "loss": 0.603, "num_tokens": 6201090.0, "reward": 0.030000000447034835, "reward_std": 0.04308707043528557, "rewards/countdown_reward/mean": 0.030000000447034835, "rewards/countdown_reward/std": 0.05366706922650337, "step": 170, "step_time": 24.69953544661403 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1578125, "completions/max_length": 896.0, "completions/max_terminated_length": 830.3, "completions/mean_length": 369.7328125, "completions/mean_terminated_length": 271.72164764404295, "completions/min_length": 12.8, "completions/min_terminated_length": 12.8, "entropy": 1.4970128297805787, "epoch": 0.0087890625, "frac_reward_zero_std": 0.31875, "grad_norm": 9.6875, "learning_rate": 8.508333333333333e-07, "loss": 0.5897, "num_tokens": 6543051.0, "reward": 0.0342187512665987, "reward_std": 0.0469283003360033, "rewards/countdown_reward/mean": 0.0342187512665987, "rewards/countdown_reward/std": 0.07865068539977074, "step": 180, "step_time": 25.08571103066206 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 896.0, "completions/max_terminated_length": 803.2, "completions/mean_length": 360.8234375, "completions/mean_terminated_length": 261.6930465698242, "completions/min_length": 10.2, "completions/min_terminated_length": 10.2, "entropy": 1.4709661722183227, "epoch": 0.00927734375, "frac_reward_zero_std": 0.28125, "grad_norm": 8.375, "learning_rate": 8.425e-07, "loss": 0.6063, "num_tokens": 6879306.0, "reward": 0.029843750409781933, "reward_std": 0.04312321580946445, "rewards/countdown_reward/mean": 0.029843750409781933, "rewards/countdown_reward/std": 0.06135669276118279, "step": 190, "step_time": 24.323589626327156 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.128125, "completions/max_length": 896.0, "completions/max_terminated_length": 826.7, "completions/mean_length": 354.4125, "completions/mean_terminated_length": 275.29358673095703, "completions/min_length": 7.2, "completions/min_terminated_length": 7.2, "entropy": 1.3901047825813293, "epoch": 0.009765625, "frac_reward_zero_std": 0.325, "grad_norm": 6.96875, "learning_rate": 8.341666666666666e-07, "loss": 0.5764, "num_tokens": 7211414.0, "reward": 0.02937500085681677, "reward_std": 0.04360318519175053, "rewards/countdown_reward/mean": 0.02937500085681677, "rewards/countdown_reward/std": 0.06871780268847942, "step": 200, "step_time": 25.020597010850906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1296875, "completions/max_length": 896.0, "completions/max_terminated_length": 856.5, "completions/mean_length": 367.153125, "completions/mean_terminated_length": 288.6918060302734, "completions/min_length": 14.2, "completions/min_terminated_length": 14.2, "entropy": 1.29588423371315, "epoch": 0.01025390625, "frac_reward_zero_std": 0.24375, "grad_norm": 7.21875, "learning_rate": 8.258333333333333e-07, "loss": 0.5916, "num_tokens": 7551672.0, "reward": 0.038281251676380634, "reward_std": 0.05550050716847181, "rewards/countdown_reward/mean": 0.038281251303851606, "rewards/countdown_reward/std": 0.08810224570333958, "step": 210, "step_time": 25.429031091928483 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.146875, "completions/max_length": 896.0, "completions/max_terminated_length": 810.0, "completions/mean_length": 368.0078125, "completions/mean_terminated_length": 277.7821350097656, "completions/min_length": 13.4, "completions/min_terminated_length": 13.4, "entropy": 1.3307440638542176, "epoch": 0.0107421875, "frac_reward_zero_std": 0.2125, "grad_norm": 7.8125, "learning_rate": 8.175e-07, "loss": 0.6016, "num_tokens": 7892517.0, "reward": 0.03625000100582838, "reward_std": 0.049447467923164366, "rewards/countdown_reward/mean": 0.03625000100582838, "rewards/countdown_reward/std": 0.07171430811285973, "step": 220, "step_time": 24.9820388328284 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1421875, "completions/max_length": 896.0, "completions/max_terminated_length": 828.7, "completions/mean_length": 343.7046875, "completions/mean_terminated_length": 251.54532012939453, "completions/min_length": 13.4, "completions/min_terminated_length": 13.4, "entropy": 1.2655625283718108, "epoch": 0.01123046875, "frac_reward_zero_std": 0.25625, "grad_norm": 6.5625, "learning_rate": 8.091666666666666e-07, "loss": 0.5733, "num_tokens": 8217764.0, "reward": 0.04484375156462193, "reward_std": 0.06598710045218467, "rewards/countdown_reward/mean": 0.04484375156462193, "rewards/countdown_reward/std": 0.11363061256706715, "step": 230, "step_time": 25.037102556042374 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1453125, "completions/max_length": 896.0, "completions/max_terminated_length": 799.1, "completions/mean_length": 359.0484375, "completions/mean_terminated_length": 268.4431884765625, "completions/min_length": 14.8, "completions/min_terminated_length": 14.8, "entropy": 1.1599660754203795, "epoch": 0.01171875, "frac_reward_zero_std": 0.30625, "grad_norm": 6.09375, "learning_rate": 8.008333333333332e-07, "loss": 0.591, "num_tokens": 8552847.0, "reward": 0.04109375048428774, "reward_std": 0.05635511502623558, "rewards/countdown_reward/mean": 0.04109375048428774, "rewards/countdown_reward/std": 0.09851273857057094, "step": 240, "step_time": 24.375826775096357 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1484375, "completions/max_length": 896.0, "completions/max_terminated_length": 836.6, "completions/mean_length": 362.7484375, "completions/mean_terminated_length": 269.93778533935546, "completions/min_length": 12.1, "completions/min_terminated_length": 12.1, "entropy": 1.2053649842739105, "epoch": 0.01220703125, "frac_reward_zero_std": 0.225, "grad_norm": 6.5625, "learning_rate": 7.924999999999999e-07, "loss": 0.5986, "num_tokens": 8890306.0, "reward": 0.034375001303851606, "reward_std": 0.04907280057668686, "rewards/countdown_reward/mean": 0.034375001303851606, "rewards/countdown_reward/std": 0.06699181348085403, "step": 250, "step_time": 24.675566471740602 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 896.0, "completions/max_terminated_length": 810.0, "completions/mean_length": 335.265625, "completions/mean_terminated_length": 266.33894805908204, "completions/min_length": 8.8, "completions/min_terminated_length": 8.8, "entropy": 1.1610565066337586, "epoch": 0.0126953125, "frac_reward_zero_std": 0.19375, "grad_norm": 8.0625, "learning_rate": 7.841666666666666e-07, "loss": 0.5831, "num_tokens": 9210128.0, "reward": 0.0448437511920929, "reward_std": 0.06717852056026459, "rewards/countdown_reward/mean": 0.0448437511920929, "rewards/countdown_reward/std": 0.10714449398219586, "step": 260, "step_time": 24.935650954023004 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 896.0, "completions/max_terminated_length": 816.2, "completions/mean_length": 333.3921875, "completions/mean_terminated_length": 258.5399505615234, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 1.262440013885498, "epoch": 0.01318359375, "frac_reward_zero_std": 0.18125, "grad_norm": 8.375, "learning_rate": 7.758333333333334e-07, "loss": 0.602, "num_tokens": 9528815.0, "reward": 0.049687500856816766, "reward_std": 0.07145854532718658, "rewards/countdown_reward/mean": 0.0496875012293458, "rewards/countdown_reward/std": 0.11678044833242893, "step": 270, "step_time": 25.409300882928072 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1203125, "completions/max_length": 896.0, "completions/max_terminated_length": 821.7, "completions/mean_length": 348.8171875, "completions/mean_terminated_length": 273.87911224365234, "completions/min_length": 18.6, "completions/min_terminated_length": 18.6, "entropy": 1.242245727777481, "epoch": 0.013671875, "frac_reward_zero_std": 0.20625, "grad_norm": 8.5, "learning_rate": 7.675e-07, "loss": 0.6079, "num_tokens": 9857346.0, "reward": 0.036718750931322576, "reward_std": 0.052498215809464455, "rewards/countdown_reward/mean": 0.036718750931322576, "rewards/countdown_reward/std": 0.07587309628725052, "step": 280, "step_time": 25.247560876980423 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 896.0, "completions/max_terminated_length": 818.8, "completions/mean_length": 324.959375, "completions/mean_terminated_length": 270.7231903076172, "completions/min_length": 9.5, "completions/min_terminated_length": 9.5, "entropy": 1.1028066635131837, "epoch": 0.01416015625, "frac_reward_zero_std": 0.20625, "grad_norm": 6.03125, "learning_rate": 7.591666666666667e-07, "loss": 0.5969, "num_tokens": 10170660.0, "reward": 0.047343751043081285, "reward_std": 0.06618047691881657, "rewards/countdown_reward/mean": 0.047343751043081285, "rewards/countdown_reward/std": 0.10932432785630226, "step": 290, "step_time": 24.50255256164819 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1296875, "completions/max_length": 896.0, "completions/max_terminated_length": 832.1, "completions/mean_length": 350.475, "completions/mean_terminated_length": 269.4090042114258, "completions/min_length": 22.6, "completions/min_terminated_length": 22.6, "entropy": 1.0962098002433778, "epoch": 0.0146484375, "frac_reward_zero_std": 0.1125, "grad_norm": 8.1875, "learning_rate": 7.508333333333333e-07, "loss": 0.5732, "num_tokens": 10500276.0, "reward": 0.047968750819563866, "reward_std": 0.06838538646697997, "rewards/countdown_reward/mean": 0.047968750819563866, "rewards/countdown_reward/std": 0.110048907995224, "step": 300, "step_time": 24.8101988658309 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1046875, "completions/max_length": 896.0, "completions/max_terminated_length": 848.4, "completions/mean_length": 328.85, "completions/mean_terminated_length": 262.10556640625, "completions/min_length": 8.6, "completions/min_terminated_length": 8.6, "entropy": 1.1127773702144623, "epoch": 0.01513671875, "frac_reward_zero_std": 0.15, "grad_norm": 9.25, "learning_rate": 7.425e-07, "loss": 0.5921, "num_tokens": 10816000.0, "reward": 0.046562501415610316, "reward_std": 0.06419679261744023, "rewards/countdown_reward/mean": 0.046562501415610316, "rewards/countdown_reward/std": 0.1015817079693079, "step": 310, "step_time": 25.099181182123722 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 896.0, "completions/max_terminated_length": 823.4, "completions/mean_length": 353.1859375, "completions/mean_terminated_length": 281.2956909179687, "completions/min_length": 10.3, "completions/min_terminated_length": 10.3, "entropy": 1.069280081987381, "epoch": 0.015625, "frac_reward_zero_std": 0.2375, "grad_norm": 7.125, "learning_rate": 7.341666666666666e-07, "loss": 0.5345, "num_tokens": 11147303.0, "reward": 0.04828125108033419, "reward_std": 0.06203695759177208, "rewards/countdown_reward/mean": 0.04828125108033419, "rewards/countdown_reward/std": 0.10523762851953507, "step": 320, "step_time": 25.224967550113796 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1125, "completions/max_length": 896.0, "completions/max_terminated_length": 816.0, "completions/mean_length": 332.334375, "completions/mean_terminated_length": 261.2976959228516, "completions/min_length": 7.2, "completions/min_terminated_length": 7.2, "entropy": 1.0088206768035888, "epoch": 0.01611328125, "frac_reward_zero_std": 0.16875, "grad_norm": 7.4375, "learning_rate": 7.258333333333333e-07, "loss": 0.5954, "num_tokens": 11465289.0, "reward": 0.05437500104308128, "reward_std": 0.07940737679600715, "rewards/countdown_reward/mean": 0.054375000298023224, "rewards/countdown_reward/std": 0.12203546613454819, "step": 330, "step_time": 24.570236332528292 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1265625, "completions/max_length": 896.0, "completions/max_terminated_length": 840.9, "completions/mean_length": 340.7734375, "completions/mean_terminated_length": 259.5488845825195, "completions/min_length": 15.4, "completions/min_terminated_length": 15.4, "entropy": 1.0386349737644196, "epoch": 0.0166015625, "frac_reward_zero_std": 0.18125, "grad_norm": 6.5625, "learning_rate": 7.175e-07, "loss": 0.5598, "num_tokens": 11788684.0, "reward": 0.055781250819563866, "reward_std": 0.07818296775221825, "rewards/countdown_reward/mean": 0.055781250819563866, "rewards/countdown_reward/std": 0.13453179076313973, "step": 340, "step_time": 25.34376360643655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10625, "completions/max_length": 896.0, "completions/max_terminated_length": 791.9, "completions/mean_length": 327.5953125, "completions/mean_terminated_length": 259.9016632080078, "completions/min_length": 15.1, "completions/min_terminated_length": 15.1, "entropy": 1.0399574160575866, "epoch": 0.01708984375, "frac_reward_zero_std": 0.2, "grad_norm": 7.4375, "learning_rate": 7.091666666666666e-07, "loss": 0.5306, "num_tokens": 12103613.0, "reward": 0.05937500223517418, "reward_std": 0.08099231384694576, "rewards/countdown_reward/mean": 0.05937500223517418, "rewards/countdown_reward/std": 0.1481408253312111, "step": 350, "step_time": 24.45040359124541 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0796875, "completions/max_length": 896.0, "completions/max_terminated_length": 782.6, "completions/mean_length": 333.4796875, "completions/mean_terminated_length": 284.84088439941405, "completions/min_length": 17.1, "completions/min_terminated_length": 17.1, "entropy": 1.0049890100955963, "epoch": 0.017578125, "frac_reward_zero_std": 0.10625, "grad_norm": 7.09375, "learning_rate": 7.008333333333333e-07, "loss": 0.5597, "num_tokens": 12422372.0, "reward": 0.0653125025331974, "reward_std": 0.08555707633495331, "rewards/countdown_reward/mean": 0.0653125025331974, "rewards/countdown_reward/std": 0.14168082512915134, "step": 360, "step_time": 24.008703147247434 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1390625, "completions/max_length": 896.0, "completions/max_terminated_length": 834.4, "completions/mean_length": 355.415625, "completions/mean_terminated_length": 267.82156524658205, "completions/min_length": 12.2, "completions/min_terminated_length": 12.2, "entropy": 0.9507689893245697, "epoch": 0.01806640625, "frac_reward_zero_std": 0.14375, "grad_norm": 7.875, "learning_rate": 6.924999999999999e-07, "loss": 0.5827, "num_tokens": 12755154.0, "reward": 0.05843750089406967, "reward_std": 0.08364246040582657, "rewards/countdown_reward/mean": 0.05843750089406967, "rewards/countdown_reward/std": 0.14659380502998828, "step": 370, "step_time": 24.557909776829185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1421875, "completions/max_length": 896.0, "completions/max_terminated_length": 840.7, "completions/mean_length": 358.5796875, "completions/mean_terminated_length": 269.2084762573242, "completions/min_length": 17.6, "completions/min_terminated_length": 17.6, "entropy": 0.9881624519824982, "epoch": 0.0185546875, "frac_reward_zero_std": 0.18125, "grad_norm": 7.375, "learning_rate": 6.841666666666666e-07, "loss": 0.5613, "num_tokens": 13089945.0, "reward": 0.05500000081956387, "reward_std": 0.07594744227826596, "rewards/countdown_reward/mean": 0.05500000081956387, "rewards/countdown_reward/std": 0.12179435268044472, "step": 380, "step_time": 24.971423621475697 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1375, "completions/max_length": 896.0, "completions/max_terminated_length": 810.6, "completions/mean_length": 352.5265625, "completions/mean_terminated_length": 266.0350082397461, "completions/min_length": 13.8, "completions/min_terminated_length": 13.8, "entropy": 1.0581223964691162, "epoch": 0.01904296875, "frac_reward_zero_std": 0.25, "grad_norm": 7.0, "learning_rate": 6.758333333333333e-07, "loss": 0.5536, "num_tokens": 13420854.0, "reward": 0.04625000040978193, "reward_std": 0.05869964323937893, "rewards/countdown_reward/mean": 0.04625000040978193, "rewards/countdown_reward/std": 0.0977620642632246, "step": 390, "step_time": 25.002157350629567 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.121875, "completions/max_length": 896.0, "completions/max_terminated_length": 824.3, "completions/mean_length": 341.946875, "completions/mean_terminated_length": 265.08824615478517, "completions/min_length": 12.5, "completions/min_terminated_length": 12.5, "entropy": 0.9753122925758362, "epoch": 0.01953125, "frac_reward_zero_std": 0.18125, "grad_norm": 7.75, "learning_rate": 6.675e-07, "loss": 0.5432, "num_tokens": 13744968.0, "reward": 0.05421875119209289, "reward_std": 0.07552922628819943, "rewards/countdown_reward/mean": 0.05421875119209289, "rewards/countdown_reward/std": 0.12587119974195957, "step": 400, "step_time": 24.589017517305912 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.15625, "completions/max_length": 896.0, "completions/max_terminated_length": 796.9, "completions/mean_length": 379.1328125, "completions/mean_terminated_length": 283.77355804443357, "completions/min_length": 18.4, "completions/min_terminated_length": 18.4, "entropy": 0.9261811673641205, "epoch": 0.02001953125, "frac_reward_zero_std": 0.16875, "grad_norm": 6.71875, "learning_rate": 6.591666666666667e-07, "loss": 0.5754, "num_tokens": 14092993.0, "reward": 0.04218750074505806, "reward_std": 0.056520643085241316, "rewards/countdown_reward/mean": 0.04218750074505806, "rewards/countdown_reward/std": 0.08452619351446629, "step": 410, "step_time": 25.90334438290447 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.103125, "completions/max_length": 896.0, "completions/max_terminated_length": 805.6, "completions/mean_length": 322.4546875, "completions/mean_terminated_length": 256.5689697265625, "completions/min_length": 16.1, "completions/min_terminated_length": 16.1, "entropy": 1.067062246799469, "epoch": 0.0205078125, "frac_reward_zero_std": 0.1125, "grad_norm": 6.96875, "learning_rate": 6.508333333333334e-07, "loss": 0.574, "num_tokens": 14404592.0, "reward": 0.06187500171363354, "reward_std": 0.08447934240102768, "rewards/countdown_reward/mean": 0.061875000968575476, "rewards/countdown_reward/std": 0.14208688028156757, "step": 420, "step_time": 24.62650573514402 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10625, "completions/max_length": 896.0, "completions/max_terminated_length": 811.1, "completions/mean_length": 337.4625, "completions/mean_terminated_length": 271.038835144043, "completions/min_length": 14.2, "completions/min_terminated_length": 14.2, "entropy": 0.9988112270832061, "epoch": 0.02099609375, "frac_reward_zero_std": 0.13125, "grad_norm": 7.875, "learning_rate": 6.424999999999999e-07, "loss": 0.6009, "num_tokens": 14725940.0, "reward": 0.06312500201165676, "reward_std": 0.08316206149756908, "rewards/countdown_reward/mean": 0.06312500201165676, "rewards/countdown_reward/std": 0.14305311515927316, "step": 430, "step_time": 25.187859536148608 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.103125, "completions/max_length": 896.0, "completions/max_terminated_length": 818.6, "completions/mean_length": 331.878125, "completions/mean_terminated_length": 267.05482482910156, "completions/min_length": 12.8, "completions/min_terminated_length": 12.8, "entropy": 0.9937775075435639, "epoch": 0.021484375, "frac_reward_zero_std": 0.13125, "grad_norm": 6.375, "learning_rate": 6.341666666666666e-07, "loss": 0.5633, "num_tokens": 15043654.0, "reward": 0.05875000134110451, "reward_std": 0.081893440335989, "rewards/countdown_reward/mean": 0.05875000134110451, "rewards/countdown_reward/std": 0.13321504928171635, "step": 440, "step_time": 25.278656247630714 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1265625, "completions/max_length": 896.0, "completions/max_terminated_length": 847.5, "completions/mean_length": 353.665625, "completions/mean_terminated_length": 275.2277572631836, "completions/min_length": 11.8, "completions/min_terminated_length": 11.8, "entropy": 0.9645709276199341, "epoch": 0.02197265625, "frac_reward_zero_std": 0.175, "grad_norm": 7.625, "learning_rate": 6.258333333333333e-07, "loss": 0.5508, "num_tokens": 15375264.0, "reward": 0.05500000044703483, "reward_std": 0.0756346419453621, "rewards/countdown_reward/mean": 0.05500000044703483, "rewards/countdown_reward/std": 0.13179100640118122, "step": 450, "step_time": 24.5269087491557 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.103125, "completions/max_length": 896.0, "completions/max_terminated_length": 832.6, "completions/mean_length": 335.703125, "completions/mean_terminated_length": 271.37211456298826, "completions/min_length": 17.9, "completions/min_terminated_length": 17.9, "entropy": 0.9477737307548523, "epoch": 0.0224609375, "frac_reward_zero_std": 0.11875, "grad_norm": 7.84375, "learning_rate": 6.175e-07, "loss": 0.5824, "num_tokens": 15695462.0, "reward": 0.05609375089406967, "reward_std": 0.075543162971735, "rewards/countdown_reward/mean": 0.05609375014901161, "rewards/countdown_reward/std": 0.11948777660727501, "step": 460, "step_time": 24.531771176680923 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 896.0, "completions/max_terminated_length": 825.8, "completions/mean_length": 320.25625, "completions/mean_terminated_length": 266.3700347900391, "completions/min_length": 7.5, "completions/min_terminated_length": 7.5, "entropy": 0.9824447631835938, "epoch": 0.02294921875, "frac_reward_zero_std": 0.1375, "grad_norm": 7.46875, "learning_rate": 6.091666666666666e-07, "loss": 0.583, "num_tokens": 16005782.0, "reward": 0.06640625223517418, "reward_std": 0.0941702265292406, "rewards/countdown_reward/mean": 0.06640625186264515, "rewards/countdown_reward/std": 0.159518039226532, "step": 470, "step_time": 24.515448566526175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1078125, "completions/max_length": 896.0, "completions/max_terminated_length": 807.3, "completions/mean_length": 333.115625, "completions/mean_terminated_length": 265.3373336791992, "completions/min_length": 9.6, "completions/min_terminated_length": 9.6, "entropy": 0.9896134197711944, "epoch": 0.0234375, "frac_reward_zero_std": 0.1375, "grad_norm": 6.25, "learning_rate": 6.008333333333333e-07, "loss": 0.5722, "num_tokens": 16324248.0, "reward": 0.06281250193715096, "reward_std": 0.08813615441322327, "rewards/countdown_reward/mean": 0.06281250268220902, "rewards/countdown_reward/std": 0.14971866346895696, "step": 480, "step_time": 25.304411934316157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1140625, "completions/max_length": 896.0, "completions/max_terminated_length": 820.0, "completions/mean_length": 346.325, "completions/mean_terminated_length": 275.4749725341797, "completions/min_length": 11.5, "completions/min_terminated_length": 11.5, "entropy": 0.8737943172454834, "epoch": 0.02392578125, "frac_reward_zero_std": 0.15625, "grad_norm": 8.125, "learning_rate": 5.925e-07, "loss": 0.5619, "num_tokens": 16651208.0, "reward": 0.07375000230967999, "reward_std": 0.10199141055345536, "rewards/countdown_reward/mean": 0.07375000230967999, "rewards/countdown_reward/std": 0.18162324875593186, "step": 490, "step_time": 25.553995211422443 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1203125, "completions/max_length": 896.0, "completions/max_terminated_length": 840.1, "completions/mean_length": 338.578125, "completions/mean_terminated_length": 261.7823287963867, "completions/min_length": 13.1, "completions/min_terminated_length": 13.1, "entropy": 0.8656957924365998, "epoch": 0.0244140625, "frac_reward_zero_std": 0.15, "grad_norm": 7.03125, "learning_rate": 5.841666666666665e-07, "loss": 0.5924, "num_tokens": 16973142.0, "reward": 0.05671875178813934, "reward_std": 0.0722871594130993, "rewards/countdown_reward/mean": 0.056718751043081286, "rewards/countdown_reward/std": 0.11478379778563977, "step": 500, "step_time": 24.83757957611233 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 896.0, "completions/max_terminated_length": 797.0, "completions/mean_length": 335.265625, "completions/mean_terminated_length": 260.7118545532227, "completions/min_length": 13.3, "completions/min_terminated_length": 13.3, "entropy": 0.9005569100379944, "epoch": 0.02490234375, "frac_reward_zero_std": 0.09375, "grad_norm": 6.5625, "learning_rate": 5.758333333333333e-07, "loss": 0.5692, "num_tokens": 17293028.0, "reward": 0.06281250156462193, "reward_std": 0.08693316914141178, "rewards/countdown_reward/mean": 0.06281250156462193, "rewards/countdown_reward/std": 0.13972996436059476, "step": 510, "step_time": 23.704434148594736 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1140625, "completions/max_length": 896.0, "completions/max_terminated_length": 819.2, "completions/mean_length": 349.234375, "completions/mean_terminated_length": 279.2843215942383, "completions/min_length": 12.2, "completions/min_terminated_length": 12.2, "entropy": 0.8960488080978394, "epoch": 0.025390625, "frac_reward_zero_std": 0.125, "grad_norm": 7.09375, "learning_rate": 5.675e-07, "loss": 0.5634, "num_tokens": 17621774.0, "reward": 0.0639062501490116, "reward_std": 0.08138733208179474, "rewards/countdown_reward/mean": 0.0639062501490116, "rewards/countdown_reward/std": 0.14010513089597226, "step": 520, "step_time": 23.974389219842852 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0953125, "completions/max_length": 896.0, "completions/max_terminated_length": 790.3, "completions/mean_length": 335.065625, "completions/mean_terminated_length": 275.59424896240233, "completions/min_length": 7.6, "completions/min_terminated_length": 7.6, "entropy": 0.845199978351593, "epoch": 0.02587890625, "frac_reward_zero_std": 0.10625, "grad_norm": 7.875, "learning_rate": 5.591666666666667e-07, "loss": 0.6123, "num_tokens": 17941572.0, "reward": 0.07703125141561032, "reward_std": 0.09908397942781448, "rewards/countdown_reward/mean": 0.07703125141561032, "rewards/countdown_reward/std": 0.17507924139499664, "step": 530, "step_time": 25.00715642571449 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1046875, "completions/max_length": 896.0, "completions/max_terminated_length": 816.4, "completions/mean_length": 329.7875, "completions/mean_terminated_length": 264.1236404418945, "completions/min_length": 10.4, "completions/min_terminated_length": 10.4, "entropy": 0.8763042867183686, "epoch": 0.0263671875, "frac_reward_zero_std": 0.15, "grad_norm": 7.375, "learning_rate": 5.508333333333333e-07, "loss": 0.5897, "num_tokens": 18257940.0, "reward": 0.06953125149011612, "reward_std": 0.09387525022029877, "rewards/countdown_reward/mean": 0.06953125186264515, "rewards/countdown_reward/std": 0.16201929412782193, "step": 540, "step_time": 24.02766567338258 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1296875, "completions/max_length": 896.0, "completions/max_terminated_length": 826.7, "completions/mean_length": 343.840625, "completions/mean_terminated_length": 261.8536666870117, "completions/min_length": 17.3, "completions/min_terminated_length": 17.3, "entropy": 0.8685924887657166, "epoch": 0.02685546875, "frac_reward_zero_std": 0.14375, "grad_norm": 5.65625, "learning_rate": 5.425e-07, "loss": 0.5696, "num_tokens": 18583274.0, "reward": 0.06546875163912773, "reward_std": 0.08565356433391572, "rewards/countdown_reward/mean": 0.06546875163912773, "rewards/countdown_reward/std": 0.14541278220713139, "step": 550, "step_time": 25.308295039273798 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 896.0, "completions/max_terminated_length": 821.7, "completions/mean_length": 331.296875, "completions/mean_terminated_length": 256.722509765625, "completions/min_length": 11.9, "completions/min_terminated_length": 11.9, "entropy": 0.9458018183708191, "epoch": 0.02734375, "frac_reward_zero_std": 0.2, "grad_norm": 6.625, "learning_rate": 5.341666666666667e-07, "loss": 0.5655, "num_tokens": 18900604.0, "reward": 0.06906250193715095, "reward_std": 0.09439494982361793, "rewards/countdown_reward/mean": 0.06906250230967999, "rewards/countdown_reward/std": 0.16526274979114533, "step": 560, "step_time": 25.389427276328206 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.103125, "completions/max_length": 896.0, "completions/max_terminated_length": 752.2, "completions/mean_length": 312.3, "completions/mean_terminated_length": 245.36475524902343, "completions/min_length": 9.3, "completions/min_terminated_length": 9.3, "entropy": 0.9667274177074432, "epoch": 0.02783203125, "frac_reward_zero_std": 0.15, "grad_norm": 7.5, "learning_rate": 5.258333333333333e-07, "loss": 0.5631, "num_tokens": 19205788.0, "reward": 0.07328125201165676, "reward_std": 0.10153508856892586, "rewards/countdown_reward/mean": 0.07328125201165676, "rewards/countdown_reward/std": 0.17713838517665864, "step": 570, "step_time": 25.02820492014289 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.096875, "completions/max_length": 896.0, "completions/max_terminated_length": 789.7, "completions/mean_length": 309.346875, "completions/mean_terminated_length": 245.8749206542969, "completions/min_length": 14.5, "completions/min_terminated_length": 14.5, "entropy": 0.9526776611804962, "epoch": 0.0283203125, "frac_reward_zero_std": 0.1, "grad_norm": 7.375, "learning_rate": 5.174999999999999e-07, "loss": 0.5834, "num_tokens": 19509066.0, "reward": 0.08109375275671482, "reward_std": 0.10205521881580353, "rewards/countdown_reward/mean": 0.08109375126659871, "rewards/countdown_reward/std": 0.1746360369026661, "step": 580, "step_time": 24.938805465959014 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 896.0, "completions/max_terminated_length": 794.1, "completions/mean_length": 338.95, "completions/mean_terminated_length": 264.69911804199216, "completions/min_length": 13.8, "completions/min_terminated_length": 13.8, "entropy": 0.8329543888568878, "epoch": 0.02880859375, "frac_reward_zero_std": 0.16875, "grad_norm": 7.75, "learning_rate": 5.091666666666666e-07, "loss": 0.5415, "num_tokens": 19831366.0, "reward": 0.07531250007450581, "reward_std": 0.1019394800066948, "rewards/countdown_reward/mean": 0.07531250305473805, "rewards/countdown_reward/std": 0.17538663893938064, "step": 590, "step_time": 24.95913271959871 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1078125, "completions/max_length": 896.0, "completions/max_terminated_length": 824.8, "completions/mean_length": 326.5609375, "completions/mean_terminated_length": 257.61290283203124, "completions/min_length": 10.8, "completions/min_terminated_length": 10.8, "entropy": 0.8435170650482178, "epoch": 0.029296875, "frac_reward_zero_std": 0.1125, "grad_norm": 6.625, "learning_rate": 5.008333333333333e-07, "loss": 0.6227, "num_tokens": 20145645.0, "reward": 0.07000000104308128, "reward_std": 0.08532712273299695, "rewards/countdown_reward/mean": 0.07000000178813934, "rewards/countdown_reward/std": 0.14291742257773876, "step": 600, "step_time": 24.51931029241532 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 896.0, "completions/max_terminated_length": 812.6, "completions/mean_length": 334.1265625, "completions/mean_terminated_length": 271.1082305908203, "completions/min_length": 15.8, "completions/min_terminated_length": 15.8, "entropy": 0.9080998659133911, "epoch": 0.02978515625, "frac_reward_zero_std": 0.14375, "grad_norm": 6.625, "learning_rate": 4.924999999999999e-07, "loss": 0.5485, "num_tokens": 20464802.0, "reward": 0.06468750238418579, "reward_std": 0.08556361570954323, "rewards/countdown_reward/mean": 0.06468750238418579, "rewards/countdown_reward/std": 0.14579989947378635, "step": 610, "step_time": 23.895136954635383 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 896.0, "completions/max_terminated_length": 814.9, "completions/mean_length": 347.065625, "completions/mean_terminated_length": 279.41527404785154, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.8951960742473603, "epoch": 0.0302734375, "frac_reward_zero_std": 0.125, "grad_norm": 6.125, "learning_rate": 4.841666666666667e-07, "loss": 0.5455, "num_tokens": 20792212.0, "reward": 0.08046875149011612, "reward_std": 0.10444493368268012, "rewards/countdown_reward/mean": 0.08046875223517418, "rewards/countdown_reward/std": 0.186547714471817, "step": 620, "step_time": 25.0146680328995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 896.0, "completions/max_terminated_length": 778.6, "completions/mean_length": 327.28125, "completions/mean_terminated_length": 257.6626953125, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.834244841337204, "epoch": 0.03076171875, "frac_reward_zero_std": 0.175, "grad_norm": 8.9375, "learning_rate": 4.758333333333333e-07, "loss": 0.5799, "num_tokens": 21106940.0, "reward": 0.06859375014901162, "reward_std": 0.08284640572965145, "rewards/countdown_reward/mean": 0.06859375052154064, "rewards/countdown_reward/std": 0.149060770124197, "step": 630, "step_time": 24.58156735803932 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1125, "completions/max_length": 896.0, "completions/max_terminated_length": 734.5, "completions/mean_length": 312.059375, "completions/mean_terminated_length": 237.9209426879883, "completions/min_length": 16.4, "completions/min_terminated_length": 16.4, "entropy": 0.8390026986598969, "epoch": 0.03125, "frac_reward_zero_std": 0.13125, "grad_norm": 8.4375, "learning_rate": 4.675e-07, "loss": 0.5825, "num_tokens": 21411866.0, "reward": 0.06875000298023223, "reward_std": 0.0896192580461502, "rewards/countdown_reward/mean": 0.06875000298023223, "rewards/countdown_reward/std": 0.14610664173960686, "step": 640, "step_time": 23.93718005064875 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 896.0, "completions/max_terminated_length": 829.7, "completions/mean_length": 340.2046875, "completions/mean_terminated_length": 254.8256866455078, "completions/min_length": 13.0, "completions/min_terminated_length": 13.0, "entropy": 0.8469295740127564, "epoch": 0.03173828125, "frac_reward_zero_std": 0.08125, "grad_norm": 7.3125, "learning_rate": 4.5916666666666663e-07, "loss": 0.5898, "num_tokens": 21734885.0, "reward": 0.05859375223517418, "reward_std": 0.08388733118772507, "rewards/countdown_reward/mean": 0.05859375223517418, "rewards/countdown_reward/std": 0.13099333830177784, "step": 650, "step_time": 24.29397039692849 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1078125, "completions/max_length": 896.0, "completions/max_terminated_length": 804.8, "completions/mean_length": 320.5078125, "completions/mean_terminated_length": 250.99697570800782, "completions/min_length": 11.8, "completions/min_terminated_length": 11.8, "entropy": 0.932354724407196, "epoch": 0.0322265625, "frac_reward_zero_std": 0.125, "grad_norm": 7.21875, "learning_rate": 4.508333333333333e-07, "loss": 0.5777, "num_tokens": 22045290.0, "reward": 0.08625000230967998, "reward_std": 0.12074292115867138, "rewards/countdown_reward/mean": 0.08625000081956387, "rewards/countdown_reward/std": 0.1962367944419384, "step": 660, "step_time": 25.05079063028097 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1140625, "completions/max_length": 896.0, "completions/max_terminated_length": 815.0, "completions/mean_length": 332.846875, "completions/mean_terminated_length": 259.8363983154297, "completions/min_length": 21.7, "completions/min_terminated_length": 21.7, "entropy": 0.8404986917972564, "epoch": 0.03271484375, "frac_reward_zero_std": 0.2, "grad_norm": 7.78125, "learning_rate": 4.425e-07, "loss": 0.5502, "num_tokens": 22363676.0, "reward": 0.06921875216066838, "reward_std": 0.09303389303386211, "rewards/countdown_reward/mean": 0.06921875216066838, "rewards/countdown_reward/std": 0.1610251296311617, "step": 670, "step_time": 24.160976577736438 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1109375, "completions/max_length": 896.0, "completions/max_terminated_length": 788.3, "completions/mean_length": 331.51875, "completions/mean_terminated_length": 261.1531021118164, "completions/min_length": 20.3, "completions/min_terminated_length": 20.3, "entropy": 0.7840811610221863, "epoch": 0.033203125, "frac_reward_zero_std": 0.10625, "grad_norm": 6.40625, "learning_rate": 4.341666666666666e-07, "loss": 0.5798, "num_tokens": 22681152.0, "reward": 0.0820312526077032, "reward_std": 0.10946015641093254, "rewards/countdown_reward/mean": 0.08203125409781933, "rewards/countdown_reward/std": 0.17900386527180673, "step": 680, "step_time": 24.077951609715818 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 896.0, "completions/max_terminated_length": 795.3, "completions/mean_length": 329.9921875, "completions/mean_terminated_length": 260.6099029541016, "completions/min_length": 12.1, "completions/min_terminated_length": 12.1, "entropy": 0.8129122972488403, "epoch": 0.03369140625, "frac_reward_zero_std": 0.14375, "grad_norm": 7.6875, "learning_rate": 4.258333333333333e-07, "loss": 0.5514, "num_tokens": 22997699.0, "reward": 0.06984375342726708, "reward_std": 0.08685192093253136, "rewards/countdown_reward/mean": 0.06984375119209289, "rewards/countdown_reward/std": 0.15657551735639572, "step": 690, "step_time": 23.668710809573533 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 896.0, "completions/max_terminated_length": 817.3, "completions/mean_length": 342.33125, "completions/mean_terminated_length": 257.51468658447266, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.8417673468589782, "epoch": 0.0341796875, "frac_reward_zero_std": 0.1625, "grad_norm": 6.65625, "learning_rate": 4.1749999999999997e-07, "loss": 0.5782, "num_tokens": 23322103.0, "reward": 0.06484375111758708, "reward_std": 0.08775122947990895, "rewards/countdown_reward/mean": 0.06484375111758708, "rewards/countdown_reward/std": 0.1558416083455086, "step": 700, "step_time": 24.84407932162285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1046875, "completions/max_length": 895.9, "completions/max_terminated_length": 780.6, "completions/mean_length": 324.2390625, "completions/mean_terminated_length": 256.8675827026367, "completions/min_length": 15.9, "completions/min_terminated_length": 15.9, "entropy": 0.833120584487915, "epoch": 0.03466796875, "frac_reward_zero_std": 0.14375, "grad_norm": 8.1875, "learning_rate": 4.091666666666667e-07, "loss": 0.5803, "num_tokens": 23634956.0, "reward": 0.08078125156462193, "reward_std": 0.11274370029568673, "rewards/countdown_reward/mean": 0.08078125156462193, "rewards/countdown_reward/std": 0.1888708457350731, "step": 710, "step_time": 24.003210030682386 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 896.0, "completions/max_terminated_length": 785.0, "completions/mean_length": 325.8203125, "completions/mean_terminated_length": 256.05240631103516, "completions/min_length": 14.5, "completions/min_terminated_length": 14.5, "entropy": 0.8463104784488678, "epoch": 0.03515625, "frac_reward_zero_std": 0.11875, "grad_norm": 6.96875, "learning_rate": 4.008333333333333e-07, "loss": 0.5973, "num_tokens": 23948753.0, "reward": 0.09265625290572643, "reward_std": 0.12520211115479468, "rewards/countdown_reward/mean": 0.09265625290572643, "rewards/countdown_reward/std": 0.20838060677051545, "step": 720, "step_time": 23.73168744854629 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 896.0, "completions/max_terminated_length": 798.0, "completions/mean_length": 327.9640625, "completions/mean_terminated_length": 259.05336303710936, "completions/min_length": 9.4, "completions/min_terminated_length": 9.4, "entropy": 0.8542082130908966, "epoch": 0.03564453125, "frac_reward_zero_std": 0.1125, "grad_norm": 6.21875, "learning_rate": 3.925e-07, "loss": 0.5799, "num_tokens": 24263910.0, "reward": 0.08125000223517417, "reward_std": 0.11535628736019135, "rewards/countdown_reward/mean": 0.08125000298023224, "rewards/countdown_reward/std": 0.1932430237531662, "step": 730, "step_time": 24.364370074123144 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.115625, "completions/max_length": 896.0, "completions/max_terminated_length": 779.5, "completions/mean_length": 329.1, "completions/mean_terminated_length": 254.90389862060547, "completions/min_length": 15.4, "completions/min_terminated_length": 15.4, "entropy": 0.8964889407157898, "epoch": 0.0361328125, "frac_reward_zero_std": 0.16875, "grad_norm": 6.78125, "learning_rate": 3.8416666666666666e-07, "loss": 0.5775, "num_tokens": 24579850.0, "reward": 0.07015625201165676, "reward_std": 0.09255762472748756, "rewards/countdown_reward/mean": 0.0701562512665987, "rewards/countdown_reward/std": 0.1594569891691208, "step": 740, "step_time": 24.19482237063348 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 896.0, "completions/max_terminated_length": 798.0, "completions/mean_length": 314.7859375, "completions/mean_terminated_length": 243.06817626953125, "completions/min_length": 14.0, "completions/min_terminated_length": 14.0, "entropy": 0.851690125465393, "epoch": 0.03662109375, "frac_reward_zero_std": 0.09375, "grad_norm": 6.96875, "learning_rate": 3.758333333333333e-07, "loss": 0.5795, "num_tokens": 24886593.0, "reward": 0.08812500312924385, "reward_std": 0.1176932692527771, "rewards/countdown_reward/mean": 0.08812500350177288, "rewards/countdown_reward/std": 0.19335507303476335, "step": 750, "step_time": 24.91104617062956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1140625, "completions/max_length": 896.0, "completions/max_terminated_length": 837.0, "completions/mean_length": 330.1078125, "completions/mean_terminated_length": 257.0475051879883, "completions/min_length": 17.7, "completions/min_terminated_length": 17.7, "entropy": 0.8071902453899383, "epoch": 0.037109375, "frac_reward_zero_std": 0.11875, "grad_norm": 6.5625, "learning_rate": 3.675e-07, "loss": 0.5799, "num_tokens": 25203158.0, "reward": 0.07171875052154064, "reward_std": 0.0914872907102108, "rewards/countdown_reward/mean": 0.07171875201165676, "rewards/countdown_reward/std": 0.15805347599089145, "step": 760, "step_time": 23.224130306206643 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0890625, "completions/max_length": 896.0, "completions/max_terminated_length": 818.4, "completions/mean_length": 323.796875, "completions/mean_terminated_length": 268.398127746582, "completions/min_length": 13.4, "completions/min_terminated_length": 13.4, "entropy": 0.7983418762683868, "epoch": 0.03759765625, "frac_reward_zero_std": 0.14375, "grad_norm": 6.15625, "learning_rate": 3.591666666666667e-07, "loss": 0.5645, "num_tokens": 25515732.0, "reward": 0.06062500067055225, "reward_std": 0.07699515260756015, "rewards/countdown_reward/mean": 0.060625001415610315, "rewards/countdown_reward/std": 0.13098395690321923, "step": 770, "step_time": 23.82822528351098 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.084375, "completions/max_length": 896.0, "completions/max_terminated_length": 823.8, "completions/mean_length": 316.425, "completions/mean_terminated_length": 262.5868911743164, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 0.8246410965919495, "epoch": 0.0380859375, "frac_reward_zero_std": 0.11875, "grad_norm": 6.46875, "learning_rate": 3.508333333333333e-07, "loss": 0.5242, "num_tokens": 25823584.0, "reward": 0.0709375023841858, "reward_std": 0.0927408766001463, "rewards/countdown_reward/mean": 0.07093750163912774, "rewards/countdown_reward/std": 0.15004065483808518, "step": 780, "step_time": 23.636079471744598 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1375, "completions/max_length": 896.0, "completions/max_terminated_length": 804.0, "completions/mean_length": 336.3625, "completions/mean_terminated_length": 247.00811462402345, "completions/min_length": 20.4, "completions/min_terminated_length": 20.4, "entropy": 0.82380011677742, "epoch": 0.03857421875, "frac_reward_zero_std": 0.15, "grad_norm": 8.875, "learning_rate": 3.425e-07, "loss": 0.6088, "num_tokens": 26144180.0, "reward": 0.07593750134110451, "reward_std": 0.10216379016637803, "rewards/countdown_reward/mean": 0.07593750134110451, "rewards/countdown_reward/std": 0.17139359414577485, "step": 790, "step_time": 24.033087480813265 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1125, "completions/max_length": 896.0, "completions/max_terminated_length": 807.4, "completions/mean_length": 311.3296875, "completions/mean_terminated_length": 237.33417663574218, "completions/min_length": 10.5, "completions/min_terminated_length": 10.5, "entropy": 0.9435284376144409, "epoch": 0.0390625, "frac_reward_zero_std": 0.125, "grad_norm": 7.625, "learning_rate": 3.3416666666666666e-07, "loss": 0.61, "num_tokens": 26448743.0, "reward": 0.07546875067055225, "reward_std": 0.10579501129686833, "rewards/countdown_reward/mean": 0.07546875067055225, "rewards/countdown_reward/std": 0.17615059800446034, "step": 800, "step_time": 23.430117261596024 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.096875, "completions/max_length": 896.0, "completions/max_terminated_length": 792.5, "completions/mean_length": 322.3265625, "completions/mean_terminated_length": 260.5394287109375, "completions/min_length": 21.2, "completions/min_terminated_length": 21.2, "entropy": 0.7501200139522552, "epoch": 0.03955078125, "frac_reward_zero_std": 0.10625, "grad_norm": 8.6875, "learning_rate": 3.258333333333333e-07, "loss": 0.6012, "num_tokens": 26760296.0, "reward": 0.0714062511920929, "reward_std": 0.09685868918895721, "rewards/countdown_reward/mean": 0.0714062511920929, "rewards/countdown_reward/std": 0.15936528667807578, "step": 810, "step_time": 23.618092795275153 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.10625, "completions/max_length": 896.0, "completions/max_terminated_length": 804.9, "completions/mean_length": 322.203125, "completions/mean_terminated_length": 253.8511978149414, "completions/min_length": 11.0, "completions/min_terminated_length": 11.0, "entropy": 0.8496071636676789, "epoch": 0.0400390625, "frac_reward_zero_std": 0.13125, "grad_norm": 8.9375, "learning_rate": 3.175e-07, "loss": 0.5544, "num_tokens": 27071754.0, "reward": 0.07453125230967998, "reward_std": 0.10312404222786427, "rewards/countdown_reward/mean": 0.07453125230967998, "rewards/countdown_reward/std": 0.16781147792935372, "step": 820, "step_time": 23.847494168020784 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1265625, "completions/max_length": 896.0, "completions/max_terminated_length": 795.4, "completions/mean_length": 331.4765625, "completions/mean_terminated_length": 250.37493743896485, "completions/min_length": 15.9, "completions/min_terminated_length": 15.9, "entropy": 0.8474254429340362, "epoch": 0.04052734375, "frac_reward_zero_std": 0.1125, "grad_norm": 6.375, "learning_rate": 3.0916666666666664e-07, "loss": 0.5568, "num_tokens": 27389203.0, "reward": 0.07343750260770321, "reward_std": 0.10714667774736882, "rewards/countdown_reward/mean": 0.07343750186264515, "rewards/countdown_reward/std": 0.17177996933460235, "step": 830, "step_time": 24.81580931842327 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1125, "completions/max_length": 896.0, "completions/max_terminated_length": 859.0, "completions/mean_length": 341.5890625, "completions/mean_terminated_length": 271.4262008666992, "completions/min_length": 14.2, "completions/min_terminated_length": 14.2, "entropy": 0.8895777881145477, "epoch": 0.041015625, "frac_reward_zero_std": 0.14375, "grad_norm": 7.625, "learning_rate": 3.0083333333333335e-07, "loss": 0.5817, "num_tokens": 27713132.0, "reward": 0.07921875193715096, "reward_std": 0.09963819682598114, "rewards/countdown_reward/mean": 0.07921875268220901, "rewards/countdown_reward/std": 0.16696702390909196, "step": 840, "step_time": 24.31707044541836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1328125, "completions/max_length": 896.0, "completions/max_terminated_length": 796.1, "completions/mean_length": 329.953125, "completions/mean_terminated_length": 242.70757904052735, "completions/min_length": 14.1, "completions/min_terminated_length": 14.1, "entropy": 0.8365299642086029, "epoch": 0.04150390625, "frac_reward_zero_std": 0.125, "grad_norm": 6.65625, "learning_rate": 2.9249999999999995e-07, "loss": 0.6128, "num_tokens": 28029622.0, "reward": 0.06796875149011612, "reward_std": 0.08987491652369499, "rewards/countdown_reward/mean": 0.06796875149011612, "rewards/countdown_reward/std": 0.1492488581687212, "step": 850, "step_time": 24.40504252873361 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1171875, "completions/max_length": 896.0, "completions/max_terminated_length": 782.2, "completions/mean_length": 337.1125, "completions/mean_terminated_length": 263.23750457763674, "completions/min_length": 16.8, "completions/min_terminated_length": 16.8, "entropy": 0.7966890335083008, "epoch": 0.0419921875, "frac_reward_zero_std": 0.08125, "grad_norm": 6.21875, "learning_rate": 2.8416666666666666e-07, "loss": 0.558, "num_tokens": 28350594.0, "reward": 0.06953125149011612, "reward_std": 0.09415114447474479, "rewards/countdown_reward/mean": 0.06953125149011612, "rewards/countdown_reward/std": 0.1647113800048828, "step": 860, "step_time": 24.430388952977957 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.125, "completions/max_length": 896.0, "completions/max_terminated_length": 814.9, "completions/mean_length": 343.6203125, "completions/mean_terminated_length": 264.50641174316405, "completions/min_length": 20.6, "completions/min_terminated_length": 20.6, "entropy": 0.8357939124107361, "epoch": 0.04248046875, "frac_reward_zero_std": 0.11875, "grad_norm": 6.5625, "learning_rate": 2.758333333333333e-07, "loss": 0.5955, "num_tokens": 28675731.0, "reward": 0.08375000208616257, "reward_std": 0.11131763160228729, "rewards/countdown_reward/mean": 0.08375000283122062, "rewards/countdown_reward/std": 0.1969392865896225, "step": 870, "step_time": 24.29804723057896 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0921875, "completions/max_length": 896.0, "completions/max_terminated_length": 814.7, "completions/mean_length": 326.309375, "completions/mean_terminated_length": 268.2035568237305, "completions/min_length": 13.9, "completions/min_terminated_length": 13.9, "entropy": 0.8172239482402801, "epoch": 0.04296875, "frac_reward_zero_std": 0.14375, "grad_norm": 7.4375, "learning_rate": 2.675e-07, "loss": 0.5207, "num_tokens": 28989933.0, "reward": 0.07593750171363353, "reward_std": 0.10525959581136704, "rewards/countdown_reward/mean": 0.07593750171363353, "rewards/countdown_reward/std": 0.17581590488553048, "step": 880, "step_time": 23.323533967137337 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.090625, "completions/max_length": 896.0, "completions/max_terminated_length": 845.2, "completions/mean_length": 328.065625, "completions/mean_terminated_length": 271.5744354248047, "completions/min_length": 17.9, "completions/min_terminated_length": 17.9, "entropy": 0.8774175584316254, "epoch": 0.04345703125, "frac_reward_zero_std": 0.15625, "grad_norm": 7.0, "learning_rate": 2.5916666666666664e-07, "loss": 0.5655, "num_tokens": 29305191.0, "reward": 0.07421875074505806, "reward_std": 0.09655480198562146, "rewards/countdown_reward/mean": 0.07421875223517418, "rewards/countdown_reward/std": 0.16562305130064486, "step": 890, "step_time": 23.691951220855117 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1109375, "completions/max_length": 896.0, "completions/max_terminated_length": 821.4, "completions/mean_length": 340.890625, "completions/mean_terminated_length": 272.02513885498047, "completions/min_length": 18.8, "completions/min_terminated_length": 18.8, "entropy": 0.8373542129993439, "epoch": 0.0439453125, "frac_reward_zero_std": 0.075, "grad_norm": 7.46875, "learning_rate": 2.5083333333333335e-07, "loss": 0.5753, "num_tokens": 29628661.0, "reward": 0.06375000216066837, "reward_std": 0.08433812372386455, "rewards/countdown_reward/mean": 0.06375000178813935, "rewards/countdown_reward/std": 0.12408021315932274, "step": 900, "step_time": 23.220538873411716 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.11875, "completions/max_length": 896.0, "completions/max_terminated_length": 773.6, "completions/mean_length": 333.51875, "completions/mean_terminated_length": 257.4912963867188, "completions/min_length": 21.3, "completions/min_terminated_length": 21.3, "entropy": 0.7894237220287323, "epoch": 0.04443359375, "frac_reward_zero_std": 0.1125, "grad_norm": 8.0, "learning_rate": 2.425e-07, "loss": 0.5637, "num_tokens": 29947305.0, "reward": 0.07984375320374966, "reward_std": 0.10594515353441239, "rewards/countdown_reward/mean": 0.07984375022351742, "rewards/countdown_reward/std": 0.17900798059999942, "step": 910, "step_time": 22.890889762155712 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1109375, "completions/max_length": 896.0, "completions/max_terminated_length": 799.0, "completions/mean_length": 325.265625, "completions/mean_terminated_length": 254.04365234375, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 0.7824168801307678, "epoch": 0.044921875, "frac_reward_zero_std": 0.13125, "grad_norm": 6.40625, "learning_rate": 2.3416666666666664e-07, "loss": 0.5601, "num_tokens": 30260763.0, "reward": 0.07328125163912773, "reward_std": 0.09717957191169262, "rewards/countdown_reward/mean": 0.07328125163912773, "rewards/countdown_reward/std": 0.15873004980385302, "step": 920, "step_time": 24.33758472185582 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.109375, "completions/max_length": 896.0, "completions/max_terminated_length": 787.3, "completions/mean_length": 341.2375, "completions/mean_terminated_length": 273.2398178100586, "completions/min_length": 19.3, "completions/min_terminated_length": 19.3, "entropy": 0.7732154250144958, "epoch": 0.04541015625, "frac_reward_zero_std": 0.1375, "grad_norm": 6.71875, "learning_rate": 2.2583333333333332e-07, "loss": 0.5364, "num_tokens": 30584435.0, "reward": 0.07312500365078449, "reward_std": 0.10188451074063778, "rewards/countdown_reward/mean": 0.07312500439584255, "rewards/countdown_reward/std": 0.1675596535205841, "step": 930, "step_time": 23.733158153668047 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 896.0, "completions/max_terminated_length": 839.1, "completions/mean_length": 334.6609375, "completions/mean_terminated_length": 271.5134994506836, "completions/min_length": 12.5, "completions/min_terminated_length": 12.5, "entropy": 0.887994259595871, "epoch": 0.0458984375, "frac_reward_zero_std": 0.1625, "grad_norm": 6.21875, "learning_rate": 2.1749999999999998e-07, "loss": 0.5417, "num_tokens": 30903926.0, "reward": 0.08062500096857547, "reward_std": 0.10788949579000473, "rewards/countdown_reward/mean": 0.08062500171363354, "rewards/countdown_reward/std": 0.17690875120460986, "step": 940, "step_time": 24.03638415466994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.128125, "completions/max_length": 896.0, "completions/max_terminated_length": 804.5, "completions/mean_length": 345.1921875, "completions/mean_terminated_length": 263.70355224609375, "completions/min_length": 7.1, "completions/min_terminated_length": 7.1, "entropy": 0.8267131209373474, "epoch": 0.04638671875, "frac_reward_zero_std": 0.13125, "grad_norm": 6.4375, "learning_rate": 2.0916666666666667e-07, "loss": 0.5589, "num_tokens": 31230117.0, "reward": 0.07000000141561032, "reward_std": 0.09300350323319435, "rewards/countdown_reward/mean": 0.07000000365078449, "rewards/countdown_reward/std": 0.1617738675326109, "step": 950, "step_time": 24.041075802221894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1, "completions/max_length": 896.0, "completions/max_terminated_length": 799.8, "completions/mean_length": 327.09375, "completions/mean_terminated_length": 263.74878082275393, "completions/min_length": 10.3, "completions/min_terminated_length": 10.3, "entropy": 0.8282470464706421, "epoch": 0.046875, "frac_reward_zero_std": 0.1375, "grad_norm": 8.0625, "learning_rate": 2.0083333333333333e-07, "loss": 0.5816, "num_tokens": 31544761.0, "reward": 0.07656250335276127, "reward_std": 0.10762294754385948, "rewards/countdown_reward/mean": 0.07656250260770321, "rewards/countdown_reward/std": 0.17815349996089935, "step": 960, "step_time": 23.31188938189298 } ], "logging_steps": 10, "max_steps": 1200, "num_input_tokens_seen": 31544761, "num_train_epochs": 1, "save_steps": 5, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 8, "trial_name": null, "trial_params": null }