{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.0439453125, "eval_steps": 10000, "global_step": 900, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.225, "completions/max_length": 896.0, "completions/max_terminated_length": 831.3, "completions/mean_length": 420.190625, "completions/mean_terminated_length": 282.23936767578124, "completions/min_length": 10.9, "completions/min_terminated_length": 10.9, "entropy": 2.1993084192276, "epoch": 0.00048828125, "frac_reward_zero_std": 0.5875, "grad_norm": 51.0, "learning_rate": 9.925e-07, "loss": 3.2985, "num_tokens": 374226.0, "reward": 0.013750000577419996, "reward_std": 0.02396928407251835, "rewards/countdown_reward/mean": 0.013750000577419996, "rewards/countdown_reward/std": 0.042456844821572304, "step": 10, "step_time": 26.37331925034523 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2359375, "completions/max_length": 896.0, "completions/max_terminated_length": 854.6, "completions/mean_length": 431.05625, "completions/mean_terminated_length": 286.68999633789065, "completions/min_length": 11.6, "completions/min_terminated_length": 11.6, "entropy": 2.0917250871658326, "epoch": 0.0009765625, "frac_reward_zero_std": 0.61875, "grad_norm": 43.25, "learning_rate": 9.841666666666666e-07, "loss": 3.0911, "num_tokens": 755378.0, "reward": 0.014843750512227416, "reward_std": 0.025129450578242542, "rewards/countdown_reward/mean": 0.014843750512227416, "rewards/countdown_reward/std": 0.05104010049253702, "step": 20, "step_time": 26.28988407207653 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.2, "completions/max_length": 896.0, "completions/max_terminated_length": 819.4, "completions/mean_length": 399.4953125, "completions/mean_terminated_length": 276.13221740722656, "completions/min_length": 8.4, "completions/min_terminated_length": 8.4, "entropy": 2.183394193649292, "epoch": 0.00146484375, "frac_reward_zero_std": 0.525, "grad_norm": 42.5, "learning_rate": 9.758333333333332e-07, "loss": 3.3142, "num_tokens": 1116335.0, "reward": 0.013906250474974513, "reward_std": 0.024378470983356236, "rewards/countdown_reward/mean": 0.013906250474974513, "rewards/countdown_reward/std": 0.03412806689739227, "step": 30, "step_time": 26.423166027106344 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1953125, "completions/max_length": 896.0, "completions/max_terminated_length": 823.1, "completions/mean_length": 390.1703125, "completions/mean_terminated_length": 268.4678482055664, "completions/min_length": 6.5, "completions/min_terminated_length": 6.5, "entropy": 2.1002277135849, "epoch": 0.001953125, "frac_reward_zero_std": 0.55, "grad_norm": 46.5, "learning_rate": 9.675e-07, "loss": 3.2509, "num_tokens": 1471292.0, "reward": 0.0187500006519258, "reward_std": 0.034330127947032454, "rewards/countdown_reward/mean": 0.0187500006519258, "rewards/countdown_reward/std": 0.06028934046626091, "step": 40, "step_time": 26.241700985468924 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.190625, "completions/max_length": 896.0, "completions/max_terminated_length": 825.9, "completions/mean_length": 356.4171875, "completions/mean_terminated_length": 229.09603576660157, "completions/min_length": 4.4, "completions/min_terminated_length": 4.4, "entropy": 1.8430278420448303, "epoch": 0.00244140625, "frac_reward_zero_std": 0.5, "grad_norm": 42.25, "learning_rate": 9.591666666666667e-07, "loss": 3.3029, "num_tokens": 1804635.0, "reward": 0.017187500512227415, "reward_std": 0.03120512804016471, "rewards/countdown_reward/mean": 0.017187500512227415, "rewards/countdown_reward/std": 0.053629692271351816, "step": 50, "step_time": 25.481621826719493 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.171875, "completions/max_length": 896.0, "completions/max_terminated_length": 826.4, "completions/mean_length": 356.1171875, "completions/mean_terminated_length": 244.8533950805664, "completions/min_length": 4.4, "completions/min_terminated_length": 4.4, "entropy": 1.930943250656128, "epoch": 0.0029296875, "frac_reward_zero_std": 0.575, "grad_norm": 38.25, "learning_rate": 9.508333333333333e-07, "loss": 3.2804, "num_tokens": 2137826.0, "reward": 0.015000000409781934, "reward_std": 0.024884347803890705, "rewards/countdown_reward/mean": 0.015000000409781934, "rewards/countdown_reward/std": 0.043557696603238584, "step": 60, "step_time": 25.769652919843793 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.146875, "completions/max_length": 896.0, "completions/max_terminated_length": 815.6, "completions/mean_length": 340.240625, "completions/mean_terminated_length": 244.73152618408204, "completions/min_length": 6.2, "completions/min_terminated_length": 6.2, "entropy": 2.102061462402344, "epoch": 0.00341796875, "frac_reward_zero_std": 0.48125, "grad_norm": 36.5, "learning_rate": 9.425e-07, "loss": 3.317, "num_tokens": 2460860.0, "reward": 0.01703125024214387, "reward_std": 0.02700106743723154, "rewards/countdown_reward/mean": 0.01703125024214387, "rewards/countdown_reward/std": 0.03765071779489517, "step": 70, "step_time": 26.03489726521075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1296875, "completions/max_length": 896.0, "completions/max_terminated_length": 796.7, "completions/mean_length": 316.2546875, "completions/mean_terminated_length": 229.84208068847656, "completions/min_length": 3.2, "completions/min_terminated_length": 3.2, "entropy": 1.9689730882644654, "epoch": 0.00390625, "frac_reward_zero_std": 0.53125, "grad_norm": 34.25, "learning_rate": 9.341666666666667e-07, "loss": 3.3896, "num_tokens": 2768575.0, "reward": 0.015937500260770322, "reward_std": 0.027120191417634488, "rewards/countdown_reward/mean": 0.015937500260770322, "rewards/countdown_reward/std": 0.045080750808119775, "step": 80, "step_time": 25.68721083942801 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1421875, "completions/max_length": 896.0, "completions/max_terminated_length": 798.9, "completions/mean_length": 319.41875, "completions/mean_terminated_length": 223.95854644775392, "completions/min_length": 4.3, "completions/min_terminated_length": 4.3, "entropy": 2.027487242221832, "epoch": 0.00439453125, "frac_reward_zero_std": 0.48125, "grad_norm": 40.75, "learning_rate": 9.258333333333333e-07, "loss": 3.3372, "num_tokens": 3078327.0, "reward": 0.018124999944120646, "reward_std": 0.029578702338039876, "rewards/countdown_reward/mean": 0.018124999944120646, "rewards/countdown_reward/std": 0.045779407024383545, "step": 90, "step_time": 25.728177772928028 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.09375, "completions/max_length": 896.0, "completions/max_terminated_length": 795.8, "completions/mean_length": 280.840625, "completions/mean_terminated_length": 217.41636505126954, "completions/min_length": 2.3, "completions/min_terminated_length": 2.3, "entropy": 1.8047725439071656, "epoch": 0.0048828125, "frac_reward_zero_std": 0.54375, "grad_norm": 34.0, "learning_rate": 9.174999999999999e-07, "loss": 3.3727, "num_tokens": 3363405.0, "reward": 0.01906250058673322, "reward_std": 0.031456972006708384, "rewards/countdown_reward/mean": 0.01906250058673322, "rewards/countdown_reward/std": 0.062330069951713085, "step": 100, "step_time": 25.512743291724473 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1078125, "completions/max_length": 896.0, "completions/max_terminated_length": 837.3, "completions/mean_length": 294.66875, "completions/mean_terminated_length": 221.6279312133789, "completions/min_length": 2.5, "completions/min_terminated_length": 2.5, "entropy": 1.9774394273757934, "epoch": 0.00537109375, "frac_reward_zero_std": 0.55, "grad_norm": 31.625, "learning_rate": 9.091666666666666e-07, "loss": 3.3144, "num_tokens": 3657273.0, "reward": 0.015625000558793546, "reward_std": 0.025899482332170008, "rewards/countdown_reward/mean": 0.015625000558793546, "rewards/countdown_reward/std": 0.04416137468069792, "step": 110, "step_time": 25.294754456356166 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0859375, "completions/max_length": 896.0, "completions/max_terminated_length": 796.1, "completions/mean_length": 256.9203125, "completions/mean_terminated_length": 197.1449996948242, "completions/min_length": 2.4, "completions/min_terminated_length": 2.4, "entropy": 1.8161049365997315, "epoch": 0.005859375, "frac_reward_zero_std": 0.45625, "grad_norm": 36.0, "learning_rate": 9.008333333333333e-07, "loss": 3.4464, "num_tokens": 3926902.0, "reward": 0.021406250726431608, "reward_std": 0.03369640149176121, "rewards/countdown_reward/mean": 0.021406250726431608, "rewards/countdown_reward/std": 0.05503195058554411, "step": 120, "step_time": 25.5652735712938 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0875, "completions/max_length": 896.0, "completions/max_terminated_length": 797.5, "completions/mean_length": 252.9828125, "completions/mean_terminated_length": 191.2444641113281, "completions/min_length": 2.8, "completions/min_terminated_length": 2.8, "entropy": 2.1179827094078063, "epoch": 0.00634765625, "frac_reward_zero_std": 0.44375, "grad_norm": 34.0, "learning_rate": 8.924999999999999e-07, "loss": 3.4323, "num_tokens": 4194095.0, "reward": 0.020625000726431607, "reward_std": 0.03396916929632425, "rewards/countdown_reward/mean": 0.020625000726431607, "rewards/countdown_reward/std": 0.05661728754639626, "step": 130, "step_time": 25.717643687780946 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.1015625, "completions/max_length": 896.0, "completions/max_terminated_length": 804.4, "completions/mean_length": 279.0984375, "completions/mean_terminated_length": 209.178173828125, "completions/min_length": 2.7, "completions/min_terminated_length": 2.7, "entropy": 1.8626050114631654, "epoch": 0.0068359375, "frac_reward_zero_std": 0.45, "grad_norm": 35.25, "learning_rate": 8.841666666666666e-07, "loss": 3.3101, "num_tokens": 4478022.0, "reward": 0.02187500074505806, "reward_std": 0.03677305728197098, "rewards/countdown_reward/mean": 0.02187500074505806, "rewards/countdown_reward/std": 0.06508030965924264, "step": 140, "step_time": 26.221133726648986 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0703125, "completions/max_length": 896.0, "completions/max_terminated_length": 820.3, "completions/mean_length": 237.4484375, "completions/mean_terminated_length": 187.54825592041016, "completions/min_length": 2.8, "completions/min_terminated_length": 2.8, "entropy": 1.7512487888336181, "epoch": 0.00732421875, "frac_reward_zero_std": 0.4625, "grad_norm": 33.75, "learning_rate": 8.758333333333333e-07, "loss": 3.405, "num_tokens": 4735349.0, "reward": 0.022500000707805157, "reward_std": 0.03876032810658216, "rewards/countdown_reward/mean": 0.022500001080334186, "rewards/countdown_reward/std": 0.07305908761918545, "step": 150, "step_time": 25.19085761765018 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0609375, "completions/max_length": 896.0, "completions/max_terminated_length": 761.7, "completions/mean_length": 224.9296875, "completions/mean_terminated_length": 181.37509460449218, "completions/min_length": 2.2, "completions/min_terminated_length": 2.2, "entropy": 1.9056554317474366, "epoch": 0.0078125, "frac_reward_zero_std": 0.39375, "grad_norm": 34.5, "learning_rate": 8.675000000000001e-07, "loss": 3.4145, "num_tokens": 4984640.0, "reward": 0.024218750931322575, "reward_std": 0.039488869905471805, "rewards/countdown_reward/mean": 0.024218750931322575, "rewards/countdown_reward/std": 0.06662302315235138, "step": 160, "step_time": 25.38799211550504 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0828125, "completions/max_length": 896.0, "completions/max_terminated_length": 787.5, "completions/mean_length": 240.5765625, "completions/mean_terminated_length": 181.69259643554688, "completions/min_length": 3.0, "completions/min_terminated_length": 3.0, "entropy": 1.7842302322387695, "epoch": 0.00830078125, "frac_reward_zero_std": 0.48125, "grad_norm": 36.75, "learning_rate": 8.591666666666666e-07, "loss": 3.3491, "num_tokens": 5243825.0, "reward": 0.019687500223517417, "reward_std": 0.032342858240008356, "rewards/countdown_reward/mean": 0.019687500223517417, "rewards/countdown_reward/std": 0.05566715896129608, "step": 170, "step_time": 25.79623172134161 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.075, "completions/max_length": 896.0, "completions/max_terminated_length": 767.2, "completions/mean_length": 237.4875, "completions/mean_terminated_length": 184.02896270751953, "completions/min_length": 2.6, "completions/min_terminated_length": 2.6, "entropy": 1.8088096499443054, "epoch": 0.0087890625, "frac_reward_zero_std": 0.4875, "grad_norm": 31.625, "learning_rate": 8.508333333333333e-07, "loss": 3.3281, "num_tokens": 5501149.0, "reward": 0.018593750521540643, "reward_std": 0.03193367030471563, "rewards/countdown_reward/mean": 0.018593750521540643, "rewards/countdown_reward/std": 0.05487189404666424, "step": 180, "step_time": 25.791260920278727 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 896.0, "completions/max_terminated_length": 814.5, "completions/mean_length": 214.875, "completions/mean_terminated_length": 175.40740203857422, "completions/min_length": 2.6, "completions/min_terminated_length": 2.6, "entropy": 1.8133741855621337, "epoch": 0.00927734375, "frac_reward_zero_std": 0.425, "grad_norm": 31.0, "learning_rate": 8.425e-07, "loss": 3.3775, "num_tokens": 5743997.0, "reward": 0.023125000484287738, "reward_std": 0.037732993997633454, "rewards/countdown_reward/mean": 0.023125000484287738, "rewards/countdown_reward/std": 0.0649927582591772, "step": 190, "step_time": 25.102144484221935 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.040625, "completions/max_length": 896.0, "completions/max_terminated_length": 749.3, "completions/mean_length": 203.93125, "completions/mean_terminated_length": 174.51675720214843, "completions/min_length": 2.3, "completions/min_terminated_length": 2.3, "entropy": 1.7565808653831483, "epoch": 0.009765625, "frac_reward_zero_std": 0.48125, "grad_norm": 29.0, "learning_rate": 8.341666666666666e-07, "loss": 3.3575, "num_tokens": 5979797.0, "reward": 0.02171875089406967, "reward_std": 0.03510004561394453, "rewards/countdown_reward/mean": 0.02171875089406967, "rewards/countdown_reward/std": 0.06441066060215235, "step": 200, "step_time": 25.48575629014522 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0734375, "completions/max_length": 896.0, "completions/max_terminated_length": 751.0, "completions/mean_length": 228.690625, "completions/mean_terminated_length": 175.98021697998047, "completions/min_length": 2.2, "completions/min_terminated_length": 2.2, "entropy": 1.5554191708564757, "epoch": 0.01025390625, "frac_reward_zero_std": 0.48125, "grad_norm": 29.0, "learning_rate": 8.258333333333333e-07, "loss": 3.2854, "num_tokens": 6231439.0, "reward": 0.01890625013038516, "reward_std": 0.02976522333920002, "rewards/countdown_reward/mean": 0.01890625013038516, "rewards/countdown_reward/std": 0.04676312413066626, "step": 210, "step_time": 24.455797299556433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 893.6, "completions/max_terminated_length": 788.9, "completions/mean_length": 215.5359375, "completions/mean_terminated_length": 181.67112884521484, "completions/min_length": 2.5, "completions/min_terminated_length": 2.5, "entropy": 1.8583380579948425, "epoch": 0.0107421875, "frac_reward_zero_std": 0.48125, "grad_norm": 29.0, "learning_rate": 8.175e-07, "loss": 3.2466, "num_tokens": 6474702.0, "reward": 0.016406250279396774, "reward_std": 0.026807691156864166, "rewards/countdown_reward/mean": 0.016406250279396774, "rewards/countdown_reward/std": 0.0370100524276495, "step": 220, "step_time": 25.69226715201512 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0546875, "completions/max_length": 887.2, "completions/max_terminated_length": 783.4, "completions/mean_length": 209.8546875, "completions/mean_terminated_length": 170.23359222412108, "completions/min_length": 2.4, "completions/min_terminated_length": 2.4, "entropy": 1.6902455806732177, "epoch": 0.01123046875, "frac_reward_zero_std": 0.45625, "grad_norm": 31.75, "learning_rate": 8.091666666666666e-07, "loss": 3.2294, "num_tokens": 6714285.0, "reward": 0.023593750782310963, "reward_std": 0.03908653613179922, "rewards/countdown_reward/mean": 0.023593750782310963, "rewards/countdown_reward/std": 0.06996353417634964, "step": 230, "step_time": 23.97437844881788 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0484375, "completions/max_length": 871.1, "completions/max_terminated_length": 739.8, "completions/mean_length": 190.453125, "completions/mean_terminated_length": 154.4923553466797, "completions/min_length": 2.3, "completions/min_terminated_length": 2.3, "entropy": 1.681441056728363, "epoch": 0.01171875, "frac_reward_zero_std": 0.3875, "grad_norm": 31.875, "learning_rate": 8.008333333333332e-07, "loss": 3.3578, "num_tokens": 6941467.0, "reward": 0.022968750819563864, "reward_std": 0.03726522326469421, "rewards/countdown_reward/mean": 0.022968750819563864, "rewards/countdown_reward/std": 0.05802568718791008, "step": 240, "step_time": 24.719267312716692 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0359375, "completions/max_length": 896.0, "completions/max_terminated_length": 733.8, "completions/mean_length": 199.7328125, "completions/mean_terminated_length": 173.6397819519043, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.587300443649292, "epoch": 0.01220703125, "frac_reward_zero_std": 0.5, "grad_norm": 31.375, "learning_rate": 7.924999999999999e-07, "loss": 3.1798, "num_tokens": 7174596.0, "reward": 0.0209375006146729, "reward_std": 0.03683012835681439, "rewards/countdown_reward/mean": 0.02093750098720193, "rewards/countdown_reward/std": 0.0683549227192998, "step": 250, "step_time": 24.908830169588327 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.040625, "completions/max_length": 896.0, "completions/max_terminated_length": 766.2, "completions/mean_length": 181.3484375, "completions/mean_terminated_length": 151.05394134521484, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.646515953540802, "epoch": 0.0126953125, "frac_reward_zero_std": 0.45625, "grad_norm": 27.125, "learning_rate": 7.841666666666666e-07, "loss": 3.2693, "num_tokens": 7395911.0, "reward": 0.021250000223517418, "reward_std": 0.031118766590952873, "rewards/countdown_reward/mean": 0.021250000223517418, "rewards/countdown_reward/std": 0.04835026822984219, "step": 260, "step_time": 25.037881007045506 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0484375, "completions/max_length": 896.0, "completions/max_terminated_length": 760.8, "completions/mean_length": 186.371875, "completions/mean_terminated_length": 150.27118148803712, "completions/min_length": 2.4, "completions/min_terminated_length": 2.4, "entropy": 1.649923610687256, "epoch": 0.01318359375, "frac_reward_zero_std": 0.44375, "grad_norm": 32.25, "learning_rate": 7.758333333333334e-07, "loss": 3.2149, "num_tokens": 7620505.0, "reward": 0.02187500074505806, "reward_std": 0.036899036914110186, "rewards/countdown_reward/mean": 0.02187500074505806, "rewards/countdown_reward/std": 0.06475385762751103, "step": 270, "step_time": 24.620592466928066 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0375, "completions/max_length": 892.1, "completions/max_terminated_length": 790.9, "completions/mean_length": 194.4109375, "completions/mean_terminated_length": 166.99260406494142, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.684264349937439, "epoch": 0.013671875, "frac_reward_zero_std": 0.46875, "grad_norm": 27.375, "learning_rate": 7.675e-07, "loss": 3.1732, "num_tokens": 7850216.0, "reward": 0.020937500707805156, "reward_std": 0.03325792122632265, "rewards/countdown_reward/mean": 0.020937500707805156, "rewards/countdown_reward/std": 0.056190946325659755, "step": 280, "step_time": 24.59303784724325 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0421875, "completions/max_length": 878.3, "completions/max_terminated_length": 751.0, "completions/mean_length": 176.66875, "completions/mean_terminated_length": 144.96968765258788, "completions/min_length": 2.3, "completions/min_terminated_length": 2.3, "entropy": 1.6441535234451294, "epoch": 0.01416015625, "frac_reward_zero_std": 0.4875, "grad_norm": 27.875, "learning_rate": 7.591666666666667e-07, "loss": 3.2064, "num_tokens": 8068624.0, "reward": 0.016250000149011613, "reward_std": 0.02659187875688076, "rewards/countdown_reward/mean": 0.016250000149011613, "rewards/countdown_reward/std": 0.03684515804052353, "step": 290, "step_time": 24.65147315589711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0484375, "completions/max_length": 896.0, "completions/max_terminated_length": 760.6, "completions/mean_length": 190.5, "completions/mean_terminated_length": 154.45074005126952, "completions/min_length": 2.2, "completions/min_terminated_length": 2.2, "entropy": 1.58493891954422, "epoch": 0.0146484375, "frac_reward_zero_std": 0.43125, "grad_norm": 27.0, "learning_rate": 7.508333333333333e-07, "loss": 3.2148, "num_tokens": 8295856.0, "reward": 0.02250000089406967, "reward_std": 0.037026658095419406, "rewards/countdown_reward/mean": 0.02250000089406967, "rewards/countdown_reward/std": 0.06564349606633187, "step": 300, "step_time": 23.980915648117662 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0453125, "completions/max_length": 896.0, "completions/max_terminated_length": 744.0, "completions/mean_length": 192.096875, "completions/mean_terminated_length": 158.2847587585449, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.484564447402954, "epoch": 0.01513671875, "frac_reward_zero_std": 0.3875, "grad_norm": 24.5, "learning_rate": 7.425e-07, "loss": 3.1605, "num_tokens": 8524058.0, "reward": 0.025625001173466444, "reward_std": 0.040091433003544806, "rewards/countdown_reward/mean": 0.025625001173466444, "rewards/countdown_reward/std": 0.06666186973452567, "step": 310, "step_time": 24.08509969925508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0515625, "completions/max_length": 896.0, "completions/max_terminated_length": 739.2, "completions/mean_length": 193.2640625, "completions/mean_terminated_length": 154.72355041503906, "completions/min_length": 2.3, "completions/min_terminated_length": 2.3, "entropy": 1.5967578291893005, "epoch": 0.015625, "frac_reward_zero_std": 0.46875, "grad_norm": 25.875, "learning_rate": 7.341666666666666e-07, "loss": 3.1335, "num_tokens": 8753011.0, "reward": 0.02000000011175871, "reward_std": 0.033064546436071394, "rewards/countdown_reward/mean": 0.02000000011175871, "rewards/countdown_reward/std": 0.05598658174276352, "step": 320, "step_time": 24.42959037479013 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.034375, "completions/max_length": 890.6, "completions/max_terminated_length": 752.9, "completions/mean_length": 174.7015625, "completions/mean_terminated_length": 149.13903732299804, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.6641492128372193, "epoch": 0.01611328125, "frac_reward_zero_std": 0.4, "grad_norm": 26.625, "learning_rate": 7.258333333333333e-07, "loss": 3.106, "num_tokens": 8970112.0, "reward": 0.02000000048428774, "reward_std": 0.033689546026289464, "rewards/countdown_reward/mean": 0.02000000048428774, "rewards/countdown_reward/std": 0.04762300215661526, "step": 330, "step_time": 25.179076719470324 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0265625, "completions/max_length": 896.0, "completions/max_terminated_length": 753.0, "completions/mean_length": 170.725, "completions/mean_terminated_length": 150.9649917602539, "completions/min_length": 2.3, "completions/min_terminated_length": 2.3, "entropy": 1.5165095686912538, "epoch": 0.0166015625, "frac_reward_zero_std": 0.36875, "grad_norm": 24.0, "learning_rate": 7.175e-07, "loss": 3.1329, "num_tokens": 9184676.0, "reward": 0.025468750111758708, "reward_std": 0.041015223041176795, "rewards/countdown_reward/mean": 0.025468750111758708, "rewards/countdown_reward/std": 0.0677075818181038, "step": 340, "step_time": 24.313945209607482 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0421875, "completions/max_length": 875.0, "completions/max_terminated_length": 744.0, "completions/mean_length": 179.93125, "completions/mean_terminated_length": 148.47922058105468, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.639288032054901, "epoch": 0.01708984375, "frac_reward_zero_std": 0.3375, "grad_norm": 27.375, "learning_rate": 7.091666666666666e-07, "loss": 3.1388, "num_tokens": 9405100.0, "reward": 0.027656251192092897, "reward_std": 0.04279166404157877, "rewards/countdown_reward/mean": 0.027656251192092897, "rewards/countdown_reward/std": 0.06816818602383137, "step": 350, "step_time": 23.679154459945856 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 879.8, "completions/max_terminated_length": 725.6, "completions/mean_length": 183.7296875, "completions/mean_terminated_length": 160.65014038085937, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.466630494594574, "epoch": 0.017578125, "frac_reward_zero_std": 0.43125, "grad_norm": 24.5, "learning_rate": 7.008333333333333e-07, "loss": 3.0263, "num_tokens": 9628019.0, "reward": 0.02000000011175871, "reward_std": 0.03202350363135338, "rewards/countdown_reward/mean": 0.02000000011175871, "rewards/countdown_reward/std": 0.04808597713708877, "step": 360, "step_time": 23.761195236165076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.034375, "completions/max_length": 876.3, "completions/max_terminated_length": 726.7, "completions/mean_length": 175.615625, "completions/mean_terminated_length": 150.09845123291015, "completions/min_length": 2.4, "completions/min_terminated_length": 2.4, "entropy": 1.5843002676963807, "epoch": 0.01806640625, "frac_reward_zero_std": 0.4, "grad_norm": 20.875, "learning_rate": 6.924999999999999e-07, "loss": 3.0925, "num_tokens": 9845729.0, "reward": 0.022031250596046447, "reward_std": 0.03412464167922735, "rewards/countdown_reward/mean": 0.022031250596046447, "rewards/countdown_reward/std": 0.048822575435042384, "step": 370, "step_time": 24.44293014323339 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0375, "completions/max_length": 875.9, "completions/max_terminated_length": 743.0, "completions/mean_length": 181.234375, "completions/mean_terminated_length": 153.66319046020507, "completions/min_length": 2.2, "completions/min_terminated_length": 2.2, "entropy": 1.6312729477882386, "epoch": 0.0185546875, "frac_reward_zero_std": 0.425, "grad_norm": 24.125, "learning_rate": 6.841666666666666e-07, "loss": 3.0691, "num_tokens": 10067019.0, "reward": 0.02109375027939677, "reward_std": 0.03301281854510307, "rewards/countdown_reward/mean": 0.02109375027939677, "rewards/countdown_reward/std": 0.048004529997706415, "step": 380, "step_time": 24.042471435014157 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05625, "completions/max_length": 896.0, "completions/max_terminated_length": 777.3, "completions/mean_length": 196.8, "completions/mean_terminated_length": 155.13842849731446, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.6682270526885987, "epoch": 0.01904296875, "frac_reward_zero_std": 0.4375, "grad_norm": 20.25, "learning_rate": 6.758333333333333e-07, "loss": 3.0205, "num_tokens": 10298263.0, "reward": 0.022812500596046448, "reward_std": 0.034675389900803565, "rewards/countdown_reward/mean": 0.022812500596046448, "rewards/countdown_reward/std": 0.0573184922337532, "step": 390, "step_time": 25.246092740632594 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0453125, "completions/max_length": 896.0, "completions/max_terminated_length": 755.9, "completions/mean_length": 188.30625, "completions/mean_terminated_length": 154.49720764160156, "completions/min_length": 2.4, "completions/min_terminated_length": 2.4, "entropy": 1.5790095448493957, "epoch": 0.01953125, "frac_reward_zero_std": 0.38125, "grad_norm": 21.0, "learning_rate": 6.675e-07, "loss": 3.0411, "num_tokens": 10524047.0, "reward": 0.027500000782310963, "reward_std": 0.04049376659095287, "rewards/countdown_reward/mean": 0.027500000782310963, "rewards/countdown_reward/std": 0.06835579425096512, "step": 400, "step_time": 24.696903565526007 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.040625, "completions/max_length": 894.1, "completions/max_terminated_length": 715.8, "completions/mean_length": 175.3625, "completions/mean_terminated_length": 144.90719833374024, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.4519409537315369, "epoch": 0.02001953125, "frac_reward_zero_std": 0.375, "grad_norm": 21.5, "learning_rate": 6.591666666666667e-07, "loss": 3.0936, "num_tokens": 10741659.0, "reward": 0.02453125063329935, "reward_std": 0.0378902230411768, "rewards/countdown_reward/mean": 0.02453125063329935, "rewards/countdown_reward/std": 0.05529664158821106, "step": 410, "step_time": 24.79300628369674 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0375, "completions/max_length": 896.0, "completions/max_terminated_length": 754.5, "completions/mean_length": 179.746875, "completions/mean_terminated_length": 151.92101821899413, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.4282978773117065, "epoch": 0.0205078125, "frac_reward_zero_std": 0.43125, "grad_norm": 20.75, "learning_rate": 6.508333333333334e-07, "loss": 3.0339, "num_tokens": 10961925.0, "reward": 0.02203125050291419, "reward_std": 0.0350292643532157, "rewards/countdown_reward/mean": 0.02203125050291419, "rewards/countdown_reward/std": 0.057419099286198615, "step": 420, "step_time": 25.018830474000424 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0359375, "completions/max_length": 892.1, "completions/max_terminated_length": 715.5, "completions/mean_length": 176.246875, "completions/mean_terminated_length": 149.36392211914062, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.447169542312622, "epoch": 0.02099609375, "frac_reward_zero_std": 0.3875, "grad_norm": 17.125, "learning_rate": 6.424999999999999e-07, "loss": 3.0323, "num_tokens": 11180095.0, "reward": 0.029531250428408384, "reward_std": 0.04363781940191984, "rewards/countdown_reward/mean": 0.029531250428408384, "rewards/countdown_reward/std": 0.08126966096460819, "step": 430, "step_time": 24.760067723039537 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.034375, "completions/max_length": 843.2, "completions/max_terminated_length": 705.3, "completions/mean_length": 169.8796875, "completions/mean_terminated_length": 144.09583892822266, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.4899834871292115, "epoch": 0.021484375, "frac_reward_zero_std": 0.39375, "grad_norm": 19.625, "learning_rate": 6.341666666666666e-07, "loss": 3.0319, "num_tokens": 11394130.0, "reward": 0.020000000298023225, "reward_std": 0.031327723525464535, "rewards/countdown_reward/mean": 0.020000000298023225, "rewards/countdown_reward/std": 0.040029546990990636, "step": 440, "step_time": 23.167645210307093 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 890.5, "completions/max_terminated_length": 686.4, "completions/mean_length": 167.6125, "completions/mean_terminated_length": 144.18032836914062, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.4972158551216126, "epoch": 0.02197265625, "frac_reward_zero_std": 0.40625, "grad_norm": 18.0, "learning_rate": 6.258333333333333e-07, "loss": 3.0398, "num_tokens": 11606666.0, "reward": 0.022187501098960637, "reward_std": 0.033618765883147717, "rewards/countdown_reward/mean": 0.022187501098960637, "rewards/countdown_reward/std": 0.04948996976017952, "step": 450, "step_time": 23.767644193395974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.028125, "completions/max_length": 884.0, "completions/max_terminated_length": 801.4, "completions/mean_length": 170.1875, "completions/mean_terminated_length": 149.28519134521486, "completions/min_length": 2.5, "completions/min_terminated_length": 2.5, "entropy": 1.553570568561554, "epoch": 0.0224609375, "frac_reward_zero_std": 0.4375, "grad_norm": 19.5, "learning_rate": 6.175e-07, "loss": 2.9874, "num_tokens": 11820934.0, "reward": 0.022812499850988387, "reward_std": 0.03734285905957222, "rewards/countdown_reward/mean": 0.022812499850988387, "rewards/countdown_reward/std": 0.055606366321444514, "step": 460, "step_time": 23.249832791183145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 896.0, "completions/max_terminated_length": 736.2, "completions/mean_length": 177.8421875, "completions/mean_terminated_length": 148.76157455444337, "completions/min_length": 2.2, "completions/min_terminated_length": 2.2, "entropy": 1.4015629768371582, "epoch": 0.02294921875, "frac_reward_zero_std": 0.41875, "grad_norm": 18.625, "learning_rate": 6.091666666666666e-07, "loss": 3.0384, "num_tokens": 12040109.0, "reward": 0.025000000558793544, "reward_std": 0.03585460986942053, "rewards/countdown_reward/mean": 0.025000000558793544, "rewards/countdown_reward/std": 0.05452777072787285, "step": 470, "step_time": 24.88592515876517 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0328125, "completions/max_length": 896.0, "completions/max_terminated_length": 788.3, "completions/mean_length": 181.4109375, "completions/mean_terminated_length": 157.1363540649414, "completions/min_length": 2.3, "completions/min_terminated_length": 2.3, "entropy": 1.392818546295166, "epoch": 0.0234375, "frac_reward_zero_std": 0.39375, "grad_norm": 17.875, "learning_rate": 6.008333333333333e-07, "loss": 3.0153, "num_tokens": 12261484.0, "reward": 0.02171875089406967, "reward_std": 0.034195422381162646, "rewards/countdown_reward/mean": 0.02171875089406967, "rewards/countdown_reward/std": 0.04822017289698124, "step": 480, "step_time": 23.11841878872365 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0375, "completions/max_length": 896.0, "completions/max_terminated_length": 728.2, "completions/mean_length": 167.95, "completions/mean_terminated_length": 139.61820602416992, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.3758840322494508, "epoch": 0.02392578125, "frac_reward_zero_std": 0.38125, "grad_norm": 17.75, "learning_rate": 5.925e-07, "loss": 2.9443, "num_tokens": 12474284.0, "reward": 0.0234375, "reward_std": 0.03491025473922491, "rewards/countdown_reward/mean": 0.0234375, "rewards/countdown_reward/std": 0.05023438818752766, "step": 490, "step_time": 24.034089656639843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0296875, "completions/max_length": 860.2, "completions/max_terminated_length": 727.3, "completions/mean_length": 170.3015625, "completions/mean_terminated_length": 147.81097869873048, "completions/min_length": 2.3, "completions/min_terminated_length": 2.3, "entropy": 1.4730456352233887, "epoch": 0.0244140625, "frac_reward_zero_std": 0.375, "grad_norm": 17.5, "learning_rate": 5.841666666666665e-07, "loss": 2.9886, "num_tokens": 12688521.0, "reward": 0.026250000484287737, "reward_std": 0.043715454265475275, "rewards/countdown_reward/mean": 0.026250000484287737, "rewards/countdown_reward/std": 0.07578264400362969, "step": 500, "step_time": 24.09979841997847 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 871.9, "completions/max_terminated_length": 712.6, "completions/mean_length": 146.7375, "completions/mean_terminated_length": 131.22517929077148, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.4819509744644166, "epoch": 0.02490234375, "frac_reward_zero_std": 0.3875, "grad_norm": 18.5, "learning_rate": 5.758333333333333e-07, "loss": 2.9644, "num_tokens": 12887749.0, "reward": 0.026718751154839994, "reward_std": 0.044659606739878656, "rewards/countdown_reward/mean": 0.026718751154839994, "rewards/countdown_reward/std": 0.0841249592602253, "step": 510, "step_time": 22.108612711634485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 885.2, "completions/max_terminated_length": 671.0, "completions/mean_length": 166.2734375, "completions/mean_terminated_length": 142.84731750488282, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.3339287400245667, "epoch": 0.025390625, "frac_reward_zero_std": 0.38125, "grad_norm": 16.5, "learning_rate": 5.675e-07, "loss": 3.0275, "num_tokens": 13099400.0, "reward": 0.024218750465661288, "reward_std": 0.035345350950956346, "rewards/countdown_reward/mean": 0.024218750465661288, "rewards/countdown_reward/std": 0.05006552264094353, "step": 520, "step_time": 24.082430380955337 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0359375, "completions/max_length": 896.0, "completions/max_terminated_length": 747.6, "completions/mean_length": 192.66875, "completions/mean_terminated_length": 166.5578598022461, "completions/min_length": 2.3, "completions/min_terminated_length": 2.3, "entropy": 1.35913827419281, "epoch": 0.02587890625, "frac_reward_zero_std": 0.39375, "grad_norm": 17.625, "learning_rate": 5.591666666666667e-07, "loss": 2.9141, "num_tokens": 13328064.0, "reward": 0.027187501080334187, "reward_std": 0.04525890182703733, "rewards/countdown_reward/mean": 0.027187501080334187, "rewards/countdown_reward/std": 0.08057694993913174, "step": 530, "step_time": 24.278202325198798 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0421875, "completions/max_length": 896.0, "completions/max_terminated_length": 722.5, "completions/mean_length": 186.5375, "completions/mean_terminated_length": 155.06928482055665, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.4033898711204529, "epoch": 0.0263671875, "frac_reward_zero_std": 0.33125, "grad_norm": 18.25, "learning_rate": 5.508333333333333e-07, "loss": 3.0438, "num_tokens": 13552752.0, "reward": 0.033437500894069674, "reward_std": 0.05141034312546253, "rewards/countdown_reward/mean": 0.033437500894069674, "rewards/countdown_reward/std": 0.09178748689591884, "step": 540, "step_time": 23.97597706336528 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.034375, "completions/max_length": 870.8, "completions/max_terminated_length": 708.9, "completions/mean_length": 169.334375, "completions/mean_terminated_length": 143.47926712036133, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.3962953448295594, "epoch": 0.02685546875, "frac_reward_zero_std": 0.375, "grad_norm": 17.5, "learning_rate": 5.425e-07, "loss": 2.9595, "num_tokens": 13766402.0, "reward": 0.025156250596046446, "reward_std": 0.04347373377531767, "rewards/countdown_reward/mean": 0.025156250596046446, "rewards/countdown_reward/std": 0.07100300677120686, "step": 550, "step_time": 23.625368633586913 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0296875, "completions/max_length": 896.0, "completions/max_terminated_length": 774.4, "completions/mean_length": 179.059375, "completions/mean_terminated_length": 157.06670761108398, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.4758158206939698, "epoch": 0.02734375, "frac_reward_zero_std": 0.375, "grad_norm": 17.125, "learning_rate": 5.341666666666667e-07, "loss": 2.9016, "num_tokens": 13986300.0, "reward": 0.023593750223517417, "reward_std": 0.03541613183915615, "rewards/countdown_reward/mean": 0.023593750223517417, "rewards/countdown_reward/std": 0.05010475628077984, "step": 560, "step_time": 23.27777968607843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 872.1, "completions/max_terminated_length": 723.9, "completions/mean_length": 172.4828125, "completions/mean_terminated_length": 156.1008285522461, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.399899423122406, "epoch": 0.02783203125, "frac_reward_zero_std": 0.375, "grad_norm": 16.5, "learning_rate": 5.258333333333333e-07, "loss": 2.8929, "num_tokens": 14202001.0, "reward": 0.02578125027939677, "reward_std": 0.040999642387032506, "rewards/countdown_reward/mean": 0.02578125027939677, "rewards/countdown_reward/std": 0.06623750440776348, "step": 570, "step_time": 21.645063481107353 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 896.0, "completions/max_terminated_length": 699.9, "completions/mean_length": 171.2828125, "completions/mean_terminated_length": 148.0937744140625, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.3453419864177705, "epoch": 0.0283203125, "frac_reward_zero_std": 0.35625, "grad_norm": 16.25, "learning_rate": 5.174999999999999e-07, "loss": 2.9563, "num_tokens": 14416918.0, "reward": 0.03000000100582838, "reward_std": 0.04752065353095532, "rewards/countdown_reward/mean": 0.03000000100582838, "rewards/countdown_reward/std": 0.08137238956987858, "step": 580, "step_time": 24.35591931855306 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 853.2, "completions/max_terminated_length": 771.5, "completions/mean_length": 167.721875, "completions/mean_terminated_length": 156.036767578125, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.3564936101436615, "epoch": 0.02880859375, "frac_reward_zero_std": 0.28125, "grad_norm": 15.75, "learning_rate": 5.091666666666666e-07, "loss": 2.9336, "num_tokens": 14629632.0, "reward": 0.028125000838190316, "reward_std": 0.04535558801144361, "rewards/countdown_reward/mean": 0.028125000838190316, "rewards/countdown_reward/std": 0.06475923582911491, "step": 590, "step_time": 22.915118097048254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 852.4, "completions/max_terminated_length": 645.1, "completions/mean_length": 158.3421875, "completions/mean_terminated_length": 145.2704948425293, "completions/min_length": 2.5, "completions/min_terminated_length": 2.5, "entropy": 1.3537224888801576, "epoch": 0.029296875, "frac_reward_zero_std": 0.3, "grad_norm": 15.75, "learning_rate": 5.008333333333333e-07, "loss": 2.8229, "num_tokens": 14836251.0, "reward": 0.029062500782310964, "reward_std": 0.04508804976940155, "rewards/countdown_reward/mean": 0.029062500782310964, "rewards/countdown_reward/std": 0.06894496046006679, "step": 600, "step_time": 21.979122698400168 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 896.0, "completions/max_terminated_length": 703.0, "completions/mean_length": 165.3109375, "completions/mean_terminated_length": 147.77641372680665, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.354167377948761, "epoch": 0.02978515625, "frac_reward_zero_std": 0.3375, "grad_norm": 15.875, "learning_rate": 4.924999999999999e-07, "loss": 2.9045, "num_tokens": 15047366.0, "reward": 0.028125000558793543, "reward_std": 0.04263280890882015, "rewards/countdown_reward/mean": 0.028125000558793543, "rewards/countdown_reward/std": 0.06515576653182506, "step": 610, "step_time": 22.39595678895712 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 880.9, "completions/max_terminated_length": 765.5, "completions/mean_length": 177.5421875, "completions/mean_terminated_length": 154.2152961730957, "completions/min_length": 2.2, "completions/min_terminated_length": 2.2, "entropy": 1.4923561334609985, "epoch": 0.0302734375, "frac_reward_zero_std": 0.35625, "grad_norm": 16.875, "learning_rate": 4.841666666666667e-07, "loss": 2.9357, "num_tokens": 15266281.0, "reward": 0.029375001043081283, "reward_std": 0.04742396511137485, "rewards/countdown_reward/mean": 0.029375001043081283, "rewards/countdown_reward/std": 0.0852317038923502, "step": 620, "step_time": 23.26461230944842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 787.4, "completions/max_terminated_length": 704.6, "completions/mean_length": 157.803125, "completions/mean_terminated_length": 141.41823120117186, "completions/min_length": 2.2, "completions/min_terminated_length": 2.2, "entropy": 1.372353744506836, "epoch": 0.03076171875, "frac_reward_zero_std": 0.3375, "grad_norm": 14.8125, "learning_rate": 4.758333333333333e-07, "loss": 2.8891, "num_tokens": 15472543.0, "reward": 0.03250000141561031, "reward_std": 0.048145537823438646, "rewards/countdown_reward/mean": 0.03250000104308128, "rewards/countdown_reward/std": 0.0815532322973013, "step": 630, "step_time": 21.110069372318684 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.028125, "completions/max_length": 864.6, "completions/max_terminated_length": 715.8, "completions/mean_length": 165.796875, "completions/mean_terminated_length": 144.79628677368163, "completions/min_length": 2.2, "completions/min_terminated_length": 2.2, "entropy": 1.4008338689804076, "epoch": 0.03125, "frac_reward_zero_std": 0.3875, "grad_norm": 18.0, "learning_rate": 4.675e-07, "loss": 2.9206, "num_tokens": 15683861.0, "reward": 0.024062500242143868, "reward_std": 0.03746545407921076, "rewards/countdown_reward/mean": 0.024062500242143868, "rewards/countdown_reward/std": 0.058644356206059456, "step": 640, "step_time": 23.97576075810939 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.040625, "completions/max_length": 843.0, "completions/max_terminated_length": 708.4, "completions/mean_length": 167.309375, "completions/mean_terminated_length": 136.137442779541, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.401304829120636, "epoch": 0.03173828125, "frac_reward_zero_std": 0.3375, "grad_norm": 15.3125, "learning_rate": 4.5916666666666663e-07, "loss": 2.9608, "num_tokens": 15896227.0, "reward": 0.027656250447034837, "reward_std": 0.042487775534391405, "rewards/countdown_reward/mean": 0.027656250447034837, "rewards/countdown_reward/std": 0.068609619140625, "step": 650, "step_time": 23.319154789857567 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 882.8, "completions/max_terminated_length": 715.8, "completions/mean_length": 159.71875, "completions/mean_terminated_length": 141.96146469116212, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.3945164203643798, "epoch": 0.0322265625, "frac_reward_zero_std": 0.33125, "grad_norm": 15.0, "learning_rate": 4.508333333333333e-07, "loss": 3.004, "num_tokens": 16103727.0, "reward": 0.03546875081956387, "reward_std": 0.05444376692175865, "rewards/countdown_reward/mean": 0.03546875081956387, "rewards/countdown_reward/std": 0.09976205304265022, "step": 660, "step_time": 23.23933917991817 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0265625, "completions/max_length": 862.2, "completions/max_terminated_length": 725.1, "completions/mean_length": 166.8390625, "completions/mean_terminated_length": 147.04438171386718, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.3718195676803588, "epoch": 0.03271484375, "frac_reward_zero_std": 0.38125, "grad_norm": 14.9375, "learning_rate": 4.425e-07, "loss": 2.8863, "num_tokens": 16315868.0, "reward": 0.027656250074505805, "reward_std": 0.04069399684667587, "rewards/countdown_reward/mean": 0.027656250074505805, "rewards/countdown_reward/std": 0.06433716192841529, "step": 670, "step_time": 21.503668100293726 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0203125, "completions/max_length": 796.9, "completions/max_terminated_length": 688.5, "completions/mean_length": 159.5125, "completions/mean_terminated_length": 144.52471160888672, "completions/min_length": 2.3, "completions/min_terminated_length": 2.3, "entropy": 1.3942294120788574, "epoch": 0.033203125, "frac_reward_zero_std": 0.40625, "grad_norm": 15.8125, "learning_rate": 4.341666666666666e-07, "loss": 2.8667, "num_tokens": 16523260.0, "reward": 0.02468750039115548, "reward_std": 0.03920227698981762, "rewards/countdown_reward/mean": 0.02468750039115548, "rewards/countdown_reward/std": 0.06662776544690133, "step": 680, "step_time": 21.104468582198024 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0328125, "completions/max_length": 864.9, "completions/max_terminated_length": 687.0, "completions/mean_length": 173.3265625, "completions/mean_terminated_length": 148.59743118286133, "completions/min_length": 2.2, "completions/min_terminated_length": 2.2, "entropy": 1.3705597281455995, "epoch": 0.03369140625, "frac_reward_zero_std": 0.34375, "grad_norm": 15.1875, "learning_rate": 4.258333333333333e-07, "loss": 2.9367, "num_tokens": 16739541.0, "reward": 0.03312500063329935, "reward_std": 0.047729380615055564, "rewards/countdown_reward/mean": 0.033125000260770324, "rewards/countdown_reward/std": 0.08274909816682338, "step": 690, "step_time": 23.566480154637247 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.046875, "completions/max_length": 896.0, "completions/max_terminated_length": 764.4, "completions/mean_length": 188.371875, "completions/mean_terminated_length": 153.54863739013672, "completions/min_length": 2.2, "completions/min_terminated_length": 2.2, "entropy": 1.2602599263191223, "epoch": 0.0341796875, "frac_reward_zero_std": 0.35, "grad_norm": 17.0, "learning_rate": 4.1749999999999997e-07, "loss": 2.9923, "num_tokens": 16965411.0, "reward": 0.024843750521540642, "reward_std": 0.03672132939100266, "rewards/countdown_reward/mean": 0.024843750521540642, "rewards/countdown_reward/std": 0.051072873175144196, "step": 700, "step_time": 24.180580911412836 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 860.7, "completions/max_terminated_length": 729.0, "completions/mean_length": 166.2421875, "completions/mean_terminated_length": 142.77251892089845, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.3303504228591918, "epoch": 0.03466796875, "frac_reward_zero_std": 0.34375, "grad_norm": 15.0625, "learning_rate": 4.091666666666667e-07, "loss": 2.9875, "num_tokens": 17177146.0, "reward": 0.02968750111758709, "reward_std": 0.04518800638616085, "rewards/countdown_reward/mean": 0.02968750111758709, "rewards/countdown_reward/std": 0.07328724712133408, "step": 710, "step_time": 21.380438385810702 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.021875, "completions/max_length": 853.4, "completions/max_terminated_length": 760.4, "completions/mean_length": 164.23125, "completions/mean_terminated_length": 147.90616302490236, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.338031566143036, "epoch": 0.03515625, "frac_reward_zero_std": 0.4125, "grad_norm": 16.25, "learning_rate": 4.008333333333333e-07, "loss": 2.9689, "num_tokens": 17387526.0, "reward": 0.024218750931322575, "reward_std": 0.03631899692118168, "rewards/countdown_reward/mean": 0.024218750931322575, "rewards/countdown_reward/std": 0.05812408849596977, "step": 720, "step_time": 23.478064949344844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 884.2, "completions/max_terminated_length": 677.0, "completions/mean_length": 166.3078125, "completions/mean_terminated_length": 142.8493621826172, "completions/min_length": 2.2, "completions/min_terminated_length": 2.2, "entropy": 1.3246945977210998, "epoch": 0.03564453125, "frac_reward_zero_std": 0.3875, "grad_norm": 15.9375, "learning_rate": 3.925e-07, "loss": 2.9074, "num_tokens": 17599223.0, "reward": 0.03140625096857548, "reward_std": 0.05138966292142868, "rewards/countdown_reward/mean": 0.03140625096857548, "rewards/countdown_reward/std": 0.09857280850410462, "step": 730, "step_time": 23.050611491780728 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0390625, "completions/max_length": 863.3, "completions/max_terminated_length": 724.8, "completions/mean_length": 176.9421875, "completions/mean_terminated_length": 147.8658477783203, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.3500581979751587, "epoch": 0.0361328125, "frac_reward_zero_std": 0.375, "grad_norm": 18.125, "learning_rate": 3.8416666666666666e-07, "loss": 2.8916, "num_tokens": 17817782.0, "reward": 0.027343750931322575, "reward_std": 0.04105484038591385, "rewards/countdown_reward/mean": 0.027343750931322575, "rewards/countdown_reward/std": 0.06861551590263844, "step": 740, "step_time": 23.008771714475007 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0296875, "completions/max_length": 893.9, "completions/max_terminated_length": 750.1, "completions/mean_length": 168.1890625, "completions/mean_terminated_length": 145.98768768310546, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.3319650530815124, "epoch": 0.03662109375, "frac_reward_zero_std": 0.3375, "grad_norm": 13.8125, "learning_rate": 3.758333333333333e-07, "loss": 2.961, "num_tokens": 18030703.0, "reward": 0.03687500152736902, "reward_std": 0.059189964458346365, "rewards/countdown_reward/mean": 0.03687500152736902, "rewards/countdown_reward/std": 0.11248447522521018, "step": 750, "step_time": 23.7137832201086 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0171875, "completions/max_length": 832.8, "completions/max_terminated_length": 683.5, "completions/mean_length": 157.309375, "completions/mean_terminated_length": 144.43923110961913, "completions/min_length": 2.2, "completions/min_terminated_length": 2.2, "entropy": 1.3674452185630799, "epoch": 0.037109375, "frac_reward_zero_std": 0.31875, "grad_norm": 16.625, "learning_rate": 3.675e-07, "loss": 2.9114, "num_tokens": 18236677.0, "reward": 0.03484374973922968, "reward_std": 0.05516568552702665, "rewards/countdown_reward/mean": 0.03484374973922968, "rewards/countdown_reward/std": 0.0996122632175684, "step": 760, "step_time": 21.48599229361862 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0328125, "completions/max_length": 867.0, "completions/max_terminated_length": 741.3, "completions/mean_length": 172.5, "completions/mean_terminated_length": 148.0032516479492, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.2775516033172607, "epoch": 0.03759765625, "frac_reward_zero_std": 0.31875, "grad_norm": 15.375, "learning_rate": 3.591666666666667e-07, "loss": 2.9307, "num_tokens": 18452421.0, "reward": 0.029062500596046446, "reward_std": 0.046534808725118636, "rewards/countdown_reward/mean": 0.029062500596046446, "rewards/countdown_reward/std": 0.0676502026617527, "step": 770, "step_time": 22.376655898522586 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.028125, "completions/max_length": 880.1, "completions/max_terminated_length": 747.3, "completions/mean_length": 174.7765625, "completions/mean_terminated_length": 153.97996292114257, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.2681300044059753, "epoch": 0.0380859375, "frac_reward_zero_std": 0.35, "grad_norm": 14.5, "learning_rate": 3.508333333333333e-07, "loss": 2.9511, "num_tokens": 18669618.0, "reward": 0.027187500149011612, "reward_std": 0.042049411498010156, "rewards/countdown_reward/mean": 0.027187500149011612, "rewards/countdown_reward/std": 0.06341570504009723, "step": 780, "step_time": 23.92856244612485 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0359375, "completions/max_length": 882.5, "completions/max_terminated_length": 786.5, "completions/mean_length": 194.9734375, "completions/mean_terminated_length": 169.04163208007813, "completions/min_length": 2.3, "completions/min_terminated_length": 2.3, "entropy": 1.4257143497467042, "epoch": 0.03857421875, "frac_reward_zero_std": 0.40625, "grad_norm": 14.6875, "learning_rate": 3.425e-07, "loss": 2.8597, "num_tokens": 18899725.0, "reward": 0.021875000186264516, "reward_std": 0.03395031839609146, "rewards/countdown_reward/mean": 0.021875000186264516, "rewards/countdown_reward/std": 0.04871735982596874, "step": 790, "step_time": 23.72505459850654 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0328125, "completions/max_length": 877.0, "completions/max_terminated_length": 767.9, "completions/mean_length": 181.9609375, "completions/mean_terminated_length": 157.7393310546875, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.366630494594574, "epoch": 0.0390625, "frac_reward_zero_std": 0.29375, "grad_norm": 14.0, "learning_rate": 3.3416666666666666e-07, "loss": 2.8585, "num_tokens": 19121492.0, "reward": 0.03921875096857548, "reward_std": 0.05894474759697914, "rewards/countdown_reward/mean": 0.03921875096857548, "rewards/countdown_reward/std": 0.10378477610647678, "step": 800, "step_time": 22.305985555145888 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.03125, "completions/max_length": 896.0, "completions/max_terminated_length": 758.8, "completions/mean_length": 161.7453125, "completions/mean_terminated_length": 137.96148147583008, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.3022151470184327, "epoch": 0.03955078125, "frac_reward_zero_std": 0.35, "grad_norm": 15.125, "learning_rate": 3.258333333333333e-07, "loss": 2.9625, "num_tokens": 19330273.0, "reward": 0.027187501080334187, "reward_std": 0.03924376629292965, "rewards/countdown_reward/mean": 0.027187501080334187, "rewards/countdown_reward/std": 0.05998440831899643, "step": 810, "step_time": 23.078460136428475 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 848.0, "completions/max_terminated_length": 774.0, "completions/mean_length": 168.28125, "completions/mean_terminated_length": 149.5424331665039, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.3011632323265077, "epoch": 0.0400390625, "frac_reward_zero_std": 0.31875, "grad_norm": 15.1875, "learning_rate": 3.175e-07, "loss": 2.9429, "num_tokens": 19543221.0, "reward": 0.031093750149011612, "reward_std": 0.05001041404902935, "rewards/countdown_reward/mean": 0.031093750149011612, "rewards/countdown_reward/std": 0.08130740188062191, "step": 820, "step_time": 22.261277247965335 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 872.6, "completions/max_terminated_length": 714.1, "completions/mean_length": 177.6390625, "completions/mean_terminated_length": 159.28287200927736, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.3941667795181274, "epoch": 0.04052734375, "frac_reward_zero_std": 0.36875, "grad_norm": 16.0, "learning_rate": 3.0916666666666664e-07, "loss": 2.8208, "num_tokens": 19762214.0, "reward": 0.030468749813735485, "reward_std": 0.04448548592627048, "rewards/countdown_reward/mean": 0.030468750558793545, "rewards/countdown_reward/std": 0.08046761713922024, "step": 830, "step_time": 23.017085282318295 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0421875, "completions/max_length": 896.0, "completions/max_terminated_length": 698.3, "completions/mean_length": 173.278125, "completions/mean_terminated_length": 141.3477424621582, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.4728692293167114, "epoch": 0.041015625, "frac_reward_zero_std": 0.38125, "grad_norm": 15.3125, "learning_rate": 3.0083333333333335e-07, "loss": 2.9588, "num_tokens": 19978424.0, "reward": 0.03437500149011612, "reward_std": 0.054521631076931955, "rewards/countdown_reward/mean": 0.03437500149011612, "rewards/countdown_reward/std": 0.09451144561171532, "step": 840, "step_time": 24.67629501139745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 862.3, "completions/max_terminated_length": 762.3, "completions/mean_length": 180.55625, "completions/mean_terminated_length": 163.30989685058594, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.379426610469818, "epoch": 0.04150390625, "frac_reward_zero_std": 0.28125, "grad_norm": 15.8125, "learning_rate": 2.9249999999999995e-07, "loss": 2.88, "num_tokens": 20199300.0, "reward": 0.02671875022351742, "reward_std": 0.0400551725178957, "rewards/countdown_reward/mean": 0.02671875022351742, "rewards/countdown_reward/std": 0.052150234952569006, "step": 850, "step_time": 23.329633852746337 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 842.4, "completions/max_terminated_length": 708.4, "completions/mean_length": 169.18125, "completions/mean_terminated_length": 150.7229766845703, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.2809597492218017, "epoch": 0.0419921875, "frac_reward_zero_std": 0.4, "grad_norm": 14.1875, "learning_rate": 2.8416666666666666e-07, "loss": 2.9573, "num_tokens": 20412796.0, "reward": 0.026718750596046448, "reward_std": 0.0368887985125184, "rewards/countdown_reward/mean": 0.026718750596046448, "rewards/countdown_reward/std": 0.0598111879080534, "step": 860, "step_time": 21.771882350370287 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 852.3, "completions/max_terminated_length": 707.3, "completions/mean_length": 158.690625, "completions/mean_terminated_length": 141.1696647644043, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.2831413984298705, "epoch": 0.04248046875, "frac_reward_zero_std": 0.35, "grad_norm": 16.875, "learning_rate": 2.758333333333333e-07, "loss": 2.9342, "num_tokens": 20619578.0, "reward": 0.02828125059604645, "reward_std": 0.04511048551648855, "rewards/countdown_reward/mean": 0.02828125059604645, "rewards/countdown_reward/std": 0.07287927865982055, "step": 870, "step_time": 22.934940588101746 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0234375, "completions/max_length": 885.4, "completions/max_terminated_length": 759.3, "completions/mean_length": 172.7078125, "completions/mean_terminated_length": 155.47078475952148, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.3583064079284668, "epoch": 0.04296875, "frac_reward_zero_std": 0.33125, "grad_norm": 14.0625, "learning_rate": 2.675e-07, "loss": 2.8697, "num_tokens": 20835475.0, "reward": 0.03328125048428774, "reward_std": 0.04892939515411854, "rewards/countdown_reward/mean": 0.03328125048428774, "rewards/countdown_reward/std": 0.08635270148515702, "step": 880, "step_time": 23.812089944351463 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.015625, "completions/max_length": 846.1, "completions/max_terminated_length": 757.6, "completions/mean_length": 155.54375, "completions/mean_terminated_length": 143.83542022705078, "completions/min_length": 2.0, "completions/min_terminated_length": 2.0, "entropy": 1.3820024251937866, "epoch": 0.04345703125, "frac_reward_zero_std": 0.38125, "grad_norm": 15.0625, "learning_rate": 2.5916666666666664e-07, "loss": 2.9181, "num_tokens": 21040319.0, "reward": 0.02937500085681677, "reward_std": 0.04575780797749758, "rewards/countdown_reward/mean": 0.02937500085681677, "rewards/countdown_reward/std": 0.0852824330329895, "step": 890, "step_time": 22.35206176545471 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.028125, "completions/max_length": 880.6, "completions/max_terminated_length": 739.3, "completions/mean_length": 180.509375, "completions/mean_terminated_length": 159.6612350463867, "completions/min_length": 2.1, "completions/min_terminated_length": 2.1, "entropy": 1.2652525186538697, "epoch": 0.0439453125, "frac_reward_zero_std": 0.3, "grad_norm": 14.0625, "learning_rate": 2.5083333333333335e-07, "loss": 2.9685, "num_tokens": 21261145.0, "reward": 0.0325000012293458, "reward_std": 0.04800397753715515, "rewards/countdown_reward/mean": 0.0325000012293458, "rewards/countdown_reward/std": 0.07297105491161346, "step": 900, "step_time": 23.15838078930974 } ], "logging_steps": 10, "max_steps": 1200, "num_input_tokens_seen": 21261145, "num_train_epochs": 1, "save_steps": 100, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 8, "trial_name": null, "trial_params": null }