sidbaines commited on
Commit
e4bec94
·
verified ·
1 Parent(s): e49044c

rollouts: coin-thinking-run2/coin-thinking-phase768/rollouts/raw_rollouts.rank-0.jsonl provenance

Browse files
rollouts/coin-thinking-run2/coin-thinking-phase768/rollouts/raw_rollouts.rank-0.jsonl.meta.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "source_repo": "arcadia-impact/scimt-dispatch-rlvr-gemma4-26b-v1-runs",
3
+ "source_path": "coin-thinking-run2/coin-thinking-phase768/rollouts/raw_rollouts.rank-0.jsonl",
4
+ "declared_bytes": 35136374734,
5
+ "streamed_bytes": 35226544314,
6
+ "sha256_streamed": "4dfe31ecc350f742d5bcfd9dd14bb72a1e4aca7214d9a99d2933941fc94a6d31",
7
+ "sha256_expected": "4dfe31ecc350f742d5bcfd9dd14bb72a1e4aca7214d9a99d2933941fc94a6d31",
8
+ "sha256_match": true,
9
+ "rows": 46976,
10
+ "dropped_fields": [
11
+ "completion_ids",
12
+ "log_extra",
13
+ "log_metric",
14
+ "trainer_state"
15
+ ],
16
+ "lines_seen": 46976,
17
+ "unparseable_lines": 0,
18
+ "blank_lines": 0,
19
+ "rows_per_rollout_audit": null,
20
+ "rows_match_audit": null,
21
+ "trainer_state": "TrainerState(epoch=0.041666666666666664, global_step=32, max_steps=768, logging_steps=1, eval_steps=500, save_steps=17, train_batch_size=4, num_train_epochs=1, num_input_tokens_seen=6272881, total_flos=0.0, log_history=[{'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.1875, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 4019.0, 'completions/mean_length': 2632.40625, 'completions/mean_terminated_length': 2294.654052734375, 'completions/min_length': 806.0, 'completions/min_terminated_length': 806.0, 'entropy': 0.1282222168520093, 'epoch': 0.0013020833333333333, 'frac_reward_zero_std': 0.25, 'grad_norm': 0.0011536319507285953, 'learning_rate': 1e-05, 'loss': -0.0033152722753584385, 'num_tokens': 228250.0, 'reward': 0.765625, 'reward/selected_zero_std_group_fraction': 0.0, 'reward/zero_std_group_fraction': 0.25, 'reward_components/channel_close_count': 0.8125, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.1875, 'reward_components/format_valid': 0.765625, 'reward_components/native_boundary_valid': 0.8125, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.8125, 'reward_components/parser_unsafe': 0.046875, 'reward_components/parser_valid': 0.765625, 'reward_components/reward': 0.765625, 'reward_components/runs_correct': 1.265625, 'reward_components/runs_total': 1.625, 'reward_components/semantic_correct': 0.765625, 'reward_std': 0.42695629596710205, 'rewards/reward_func/mean': 0.765625, 'rewards/reward_func/std': 0.42695629596710205, 'sampling/importance_sampling_ratio/max': 1.5687487125396729, 'sampling/importance_sampling_ratio/mean': 0.27578580379486084, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 11.267158508300781, 'sampling/sampling_logp_difference/mean': 0.014642934314906597, 'step': 1, 'step_time': 200.55014428962022}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.015625, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3940.0, 'completions/mean_length': 1831.34375, 'completions/mean_terminated_length': 1795.39697265625, 'completions/min_length': 601.0, 'completions/min_terminated_length': 601.0, 'entropy': 0.10770621336996555, 'epoch': 0.0026041666666666665, 'frac_reward_zero_std': 0.875, 'grad_norm': 0.015225221402943134, 'learning_rate': 1e-05, 'loss': -0.0068842144683003426, 'num_tokens': 423024.0, 'reward': 0.984375, 'reward/selected_zero_std_group_fraction': 0.375, 'reward/zero_std_group_fraction': 0.5625, 'reward_components/channel_close_count': 0.984375, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.015625, 'reward_components/format_valid': 0.984375, 'reward_components/native_boundary_valid': 0.984375, 'reward_components/parser_json': 0.0625, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.921875, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.984375, 'reward_components/reward': 0.984375, 'reward_components/runs_correct': 1.359375, 'reward_components/runs_total': 1.375, 'reward_components/semantic_correct': 0.984375, 'reward_std': 0.125, 'rewards/reward_func/mean': 0.984375, 'rewards/reward_func/std': 0.125, 'sampling/importance_sampling_ratio/max': 2.610960006713867, 'sampling/importance_sampling_ratio/mean': 0.315263032913208, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 12.9798583984375, 'sampling/sampling_logp_difference/mean': 0.012177010998129845, 'step': 2, 'step_time': 142.26964644063264}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.109375, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3878.0, 'completions/mean_length': 2153.34375, 'completions/mean_terminated_length': 1914.77197265625, 'completions/min_length': 641.0, 'completions/min_terminated_length': 641.0, 'entropy': 0.12617996335029602, 'epoch': 0.00390625, 'frac_reward_zero_std': 0.5, 'grad_norm': 0.015413987450301647, 'learning_rate': 1e-05, 'loss': -0.009396588429808617, 'num_tokens': 618182.0, 'reward': 0.875, 'reward/selected_zero_std_group_fraction': 0.25, 'reward/zero_std_group_fraction': 0.5416666666666666, 'reward_components/channel_close_count': 0.890625, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.109375, 'reward_components/format_valid': 0.875, 'reward_components/native_boundary_valid': 0.890625, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.890625, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.875, 'reward_components/reward': 0.875, 'reward_components/runs_correct': 1.265625, 'reward_components/runs_total': 1.5, 'reward_components/semantic_correct': 0.875, 'reward_std': 0.3333333432674408, 'rewards/reward_func/mean': 0.875, 'rewards/reward_func/std': 0.3333333432674408, 'sampling/importance_sampling_ratio/max': 2.8673667907714844, 'sampling/importance_sampling_ratio/mean': 0.2788175046443939, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 13.147320747375488, 'sampling/sampling_logp_difference/mean': 0.015405948273837566, 'step': 3, 'step_time': 145.2710806336254}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.171875, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3812.0, 'completions/mean_length': 2255.5, 'completions/mean_terminated_length': 1873.509521484375, 'completions/min_length': 600.0, 'completions/min_terminated_length': 600.0, 'entropy': 0.11300652660429478, 'epoch': 0.005208333333333333, 'frac_reward_zero_std': 0.625, 'grad_norm': 0.0048752459697425365, 'learning_rate': 1e-05, 'loss': -0.0020500593818724155, 'num_tokens': 839974.0, 'reward': 0.734375, 'reward/selected_zero_std_group_fraction': 0.25, 'reward/zero_std_group_fraction': 0.5625, 'reward_components/channel_close_count': 0.828125, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.171875, 'reward_components/format_valid': 0.734375, 'reward_components/native_boundary_valid': 0.828125, 'reward_components/parser_json': 0.046875, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.78125, 'reward_components/parser_unsafe': 0.09375, 'reward_components/parser_valid': 0.734375, 'reward_components/reward': 0.734375, 'reward_components/runs_correct': 1.109375, 'reward_components/runs_total': 1.625, 'reward_components/semantic_correct': 0.734375, 'reward_std': 0.44515693187713623, 'rewards/reward_func/mean': 0.734375, 'rewards/reward_func/std': 0.44515693187713623, 'sampling/importance_sampling_ratio/max': 2.213904619216919, 'sampling/importance_sampling_ratio/mean': 0.32652053236961365, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 7.92255163192749, 'sampling/sampling_logp_difference/mean': 0.013877478428184986, 'step': 4, 'step_time': 156.51220831135288}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.046875, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3845.0, 'completions/mean_length': 2066.296875, 'completions/mean_terminated_length': 1966.475341796875, 'completions/min_length': 858.0, 'completions/min_terminated_length': 858.0, 'entropy': 0.08706962689757347, 'epoch': 0.006510416666666667, 'frac_reward_zero_std': 0.625, 'grad_norm': 0.0025338781997561455, 'learning_rate': 1e-05, 'loss': -0.007903359830379486, 'num_tokens': 1036921.0, 'reward': 0.953125, 'reward/selected_zero_std_group_fraction': 0.25, 'reward/zero_std_group_fraction': 0.575, 'reward_components/channel_close_count': 0.953125, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.046875, 'reward_components/format_valid': 0.953125, 'reward_components/native_boundary_valid': 0.953125, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.953125, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.953125, 'reward_components/reward': 0.953125, 'reward_components/runs_correct': 1.53125, 'reward_components/runs_total': 1.625, 'reward_components/semantic_correct': 0.953125, 'reward_std': 0.21304203569889069, 'rewards/reward_func/mean': 0.953125, 'rewards/reward_func/std': 0.21304203569889069, 'sampling/importance_sampling_ratio/max': 2.06689190864563, 'sampling/importance_sampling_ratio/mean': 0.18144401907920837, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 25.62481117248535, 'sampling/sampling_logp_difference/mean': 0.01159844920039177, 'step': 5, 'step_time': 143.6842720527202}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.203125, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3265.0, 'completions/mean_length': 2198.421875, 'completions/mean_terminated_length': 1714.7255859375, 'completions/min_length': 653.0, 'completions/min_terminated_length': 653.0, 'entropy': 0.09479253180325031, 'epoch': 0.0078125, 'frac_reward_zero_std': 0.375, 'grad_norm': 0.0012363357236608863, 'learning_rate': 1e-05, 'loss': -0.005158400163054466, 'num_tokens': 1244628.0, 'reward': 0.78125, 'reward/selected_zero_std_group_fraction': 0.20833333333333334, 'reward/zero_std_group_fraction': 0.5416666666666666, 'reward_components/channel_close_count': 0.796875, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.203125, 'reward_components/format_valid': 0.78125, 'reward_components/native_boundary_valid': 0.796875, 'reward_components/parser_json': 0.0625, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.734375, 'reward_components/parser_unsafe': 0.015625, 'reward_components/parser_valid': 0.78125, 'reward_components/reward': 0.78125, 'reward_components/runs_correct': 1.1875, 'reward_components/runs_total': 1.5, 'reward_components/semantic_correct': 0.78125, 'reward_std': 0.4166666865348816, 'rewards/reward_func/mean': 0.78125, 'rewards/reward_func/std': 0.4166666865348816, 'sampling/importance_sampling_ratio/max': 1.0, 'sampling/importance_sampling_ratio/mean': 0.2652585506439209, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 5.46408224105835, 'sampling/sampling_logp_difference/mean': 0.013486729934811592, 'step': 6, 'step_time': 153.38705213507637}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.140625, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3736.0, 'completions/mean_length': 2206.71875, 'completions/mean_terminated_length': 1897.5635986328125, 'completions/min_length': 677.0, 'completions/min_terminated_length': 677.0, 'entropy': 0.08720389613881707, 'epoch': 0.009114583333333334, 'frac_reward_zero_std': 0.625, 'grad_norm': 0.004168783780187368, 'learning_rate': 1e-05, 'loss': -0.010104609653353691, 'num_tokens': 1451650.0, 'reward': 0.859375, 'reward/selected_zero_std_group_fraction': 0.21428571428571427, 'reward/zero_std_group_fraction': 0.5535714285714286, 'reward_components/channel_close_count': 0.859375, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.140625, 'reward_components/format_valid': 0.859375, 'reward_components/native_boundary_valid': 0.859375, 'reward_components/parser_json': 0.125, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.734375, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.859375, 'reward_components/reward': 0.859375, 'reward_components/runs_correct': 1.34375, 'reward_components/runs_total': 1.625, 'reward_components/semantic_correct': 0.859375, 'reward_std': 0.3503824472427368, 'rewards/reward_func/mean': 0.859375, 'rewards/reward_func/std': 0.3503824472427368, 'sampling/importance_sampling_ratio/max': 2.483690023422241, 'sampling/importance_sampling_ratio/mean': 0.3499862253665924, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 3.714539051055908, 'sampling/sampling_logp_difference/mean': 0.010350207798182964, 'step': 7, 'step_time': 152.70913899922743}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.078125, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3866.0, 'completions/mean_length': 2072.078125, 'completions/mean_terminated_length': 1900.559326171875, 'completions/min_length': 739.0, 'completions/min_terminated_length': 739.0, 'entropy': 0.10795088950544596, 'epoch': 0.010416666666666666, 'frac_reward_zero_std': 0.75, 'grad_norm': 0.008111380971968174, 'learning_rate': 1e-05, 'loss': -0.002255987608805299, 'num_tokens': 1655495.0, 'reward': 0.796875, 'reward/selected_zero_std_group_fraction': 0.25, 'reward/zero_std_group_fraction': 0.578125, 'reward_components/channel_close_count': 0.921875, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.078125, 'reward_components/format_valid': 0.796875, 'reward_components/native_boundary_valid': 0.921875, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.921875, 'reward_components/parser_unsafe': 0.125, 'reward_components/parser_valid': 0.796875, 'reward_components/reward': 0.796875, 'reward_components/runs_correct': 1.40625, 'reward_components/runs_total': 1.625, 'reward_components/semantic_correct': 0.796875, 'reward_std': 0.40550529956817627, 'rewards/reward_func/mean': 0.796875, 'rewards/reward_func/std': 0.40550529956817627, 'sampling/importance_sampling_ratio/max': 2.2298293113708496, 'sampling/importance_sampling_ratio/mean': 0.3002867102622986, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 12.510236740112305, 'sampling/sampling_logp_difference/mean': 0.011226527392864227, 'step': 8, 'step_time': 149.86075648618862}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.15625, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3638.0, 'completions/mean_length': 2010.03125, 'completions/mean_terminated_length': 1623.74072265625, 'completions/min_length': 629.0, 'completions/min_terminated_length': 629.0, 'entropy': 0.0870186563115567, 'epoch': 0.01171875, 'frac_reward_zero_std': 0.875, 'grad_norm': 0.0017528084572404623, 'learning_rate': 1e-05, 'loss': -0.006248719058930874, 'num_tokens': 1864585.0, 'reward': 0.84375, 'reward/selected_zero_std_group_fraction': 0.3055555555555556, 'reward/zero_std_group_fraction': 0.6111111111111112, 'reward_components/channel_close_count': 0.84375, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.15625, 'reward_components/format_valid': 0.84375, 'reward_components/native_boundary_valid': 0.84375, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.84375, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.84375, 'reward_components/reward': 0.84375, 'reward_components/runs_correct': 1.3125, 'reward_components/runs_total': 1.625, 'reward_components/semantic_correct': 0.84375, 'reward_std': 0.36596253514289856, 'rewards/reward_func/mean': 0.84375, 'rewards/reward_func/std': 0.36596253514289856, 'sampling/importance_sampling_ratio/max': 1.813738226890564, 'sampling/importance_sampling_ratio/mean': 0.38007521629333496, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 6.474149227142334, 'sampling/sampling_logp_difference/mean': 0.010790170170366764, 'step': 9, 'step_time': 150.01978925429285}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.140625, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 2771.0, 'completions/mean_length': 1648.953125, 'completions/mean_terminated_length': 1248.5272216796875, 'completions/min_length': 650.0, 'completions/min_terminated_length': 650.0, 'entropy': 0.08642252301797271, 'epoch': 0.013020833333333334, 'frac_reward_zero_std': 0.875, 'grad_norm': 0.0029048214200884104, 'learning_rate': 1e-05, 'loss': -0.004058819729834795, 'num_tokens': 2024262.0, 'reward': 0.859375, 'reward/selected_zero_std_group_fraction': 0.35, 'reward/zero_std_group_fraction': 0.6375, 'reward_components/channel_close_count': 0.859375, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.140625, 'reward_components/format_valid': 0.859375, 'reward_components/native_boundary_valid': 0.859375, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.859375, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.859375, 'reward_components/reward': 0.859375, 'reward_components/runs_correct': 0.984375, 'reward_components/runs_total': 1.125, 'reward_components/semantic_correct': 0.859375, 'reward_std': 0.3503824472427368, 'rewards/reward_func/mean': 0.859375, 'rewards/reward_func/std': 0.3503824472427368, 'sampling/importance_sampling_ratio/max': 1.9908063411712646, 'sampling/importance_sampling_ratio/mean': 0.39528805017471313, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 3.9151744842529297, 'sampling/sampling_logp_difference/mean': 0.012164212763309479, 'step': 10, 'step_time': 129.30503477202728}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.140625, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 4036.0, 'completions/mean_length': 1976.546875, 'completions/mean_terminated_length': 1629.7271728515625, 'completions/min_length': 581.0, 'completions/min_terminated_length': 581.0, 'entropy': 0.11516128480434418, 'epoch': 0.014322916666666666, 'frac_reward_zero_std': 0.375, 'grad_norm': 0.03148887678980827, 'learning_rate': 1e-05, 'loss': -0.06983760744333267, 'num_tokens': 2210473.0, 'reward': 0.796875, 'reward/selected_zero_std_group_fraction': 0.3181818181818182, 'reward/zero_std_group_fraction': 0.6136363636363636, 'reward_components/channel_close_count': 0.84375, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.140625, 'reward_components/format_valid': 0.796875, 'reward_components/native_boundary_valid': 0.84375, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.84375, 'reward_components/parser_unsafe': 0.015625, 'reward_components/parser_valid': 0.796875, 'reward_components/reward': 0.796875, 'reward_components/runs_correct': 1.046875, 'reward_components/runs_total': 1.375, 'reward_components/semantic_correct': 0.796875, 'reward_std': 0.40550529956817627, 'rewards/reward_func/mean': 0.796875, 'rewards/reward_func/std': 0.40550529956817627, 'sampling/importance_sampling_ratio/max': 2.955448627471924, 'sampling/importance_sampling_ratio/mean': 0.39765244722366333, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 15.707569122314453, 'sampling/sampling_logp_difference/mean': 0.013812672346830368, 'step': 11, 'step_time': 139.653326690197}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.078125, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3948.0, 'completions/mean_length': 1791.203125, 'completions/mean_terminated_length': 1595.88134765625, 'completions/min_length': 680.0, 'completions/min_terminated_length': 680.0, 'entropy': 0.07880649995058775, 'epoch': 0.015625, 'frac_reward_zero_std': 0.625, 'grad_norm': 0.025353381410241127, 'learning_rate': 1e-05, 'loss': -0.014870086684823036, 'num_tokens': 2404982.0, 'reward': 0.90625, 'reward/selected_zero_std_group_fraction': 0.3125, 'reward/zero_std_group_fraction': 0.6145833333333334, 'reward_components/channel_close_count': 0.921875, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.078125, 'reward_components/format_valid': 0.90625, 'reward_components/native_boundary_valid': 0.921875, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.921875, 'reward_components/parser_unsafe': 0.015625, 'reward_components/parser_valid': 0.90625, 'reward_components/reward': 0.90625, 'reward_components/runs_correct': 1.203125, 'reward_components/runs_total': 1.375, 'reward_components/semantic_correct': 0.90625, 'reward_std': 0.29378482699394226, 'rewards/reward_func/mean': 0.90625, 'rewards/reward_func/std': 0.29378482699394226, 'sampling/importance_sampling_ratio/max': 2.6387836933135986, 'sampling/importance_sampling_ratio/mean': 0.3010123074054718, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 26.78571319580078, 'sampling/sampling_logp_difference/mean': 0.010469737462699413, 'step': 12, 'step_time': 137.0080711999908}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.0, 'completions/max_length': 3022.0, 'completions/max_terminated_length': 3022.0, 'completions/mean_length': 1364.96875, 'completions/mean_terminated_length': 1364.96875, 'completions/min_length': 532.0, 'completions/min_terminated_length': 532.0, 'entropy': 0.07851186720654368, 'epoch': 0.016927083333333332, 'frac_reward_zero_std': 1.0, 'grad_norm': 0.0, 'learning_rate': 1e-05, 'loss': 0.0, 'num_tokens': 2546612.0, 'reward': 1.0, 'reward/selected_zero_std_group_fraction': 0.36538461538461536, 'reward/zero_std_group_fraction': 0.6442307692307693, 'reward_components/channel_close_count': 1.0, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.0, 'reward_components/format_valid': 1.0, 'reward_components/native_boundary_valid': 1.0, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 1.0, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 1.0, 'reward_components/reward': 1.0, 'reward_components/runs_correct': 1.375, 'reward_components/runs_total': 1.375, 'reward_components/semantic_correct': 1.0, 'reward_std': 0.0, 'rewards/reward_func/mean': 1.0, 'rewards/reward_func/std': 0.0, 'sampling/importance_sampling_ratio/max': 2.8008604049682617, 'sampling/importance_sampling_ratio/mean': 0.37771734595298767, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 3.568187952041626, 'sampling/sampling_logp_difference/mean': 0.009501718915998936, 'step': 13, 'step_time': 93.79989521810785}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.0625, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3501.0, 'completions/mean_length': 1657.578125, 'completions/mean_terminated_length': 1495.0167236328125, 'completions/min_length': 372.0, 'completions/min_terminated_length': 372.0, 'entropy': 0.08700296236202121, 'epoch': 0.018229166666666668, 'frac_reward_zero_std': 0.625, 'grad_norm': 0.022519465535879135, 'learning_rate': 1e-05, 'loss': -0.016026651486754417, 'num_tokens': 2731545.0, 'reward': 0.90625, 'reward/selected_zero_std_group_fraction': 0.35714285714285715, 'reward/zero_std_group_fraction': 0.6428571428571429, 'reward_components/channel_close_count': 0.9375, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.0625, 'reward_components/format_valid': 0.90625, 'reward_components/native_boundary_valid': 0.9375, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.9375, 'reward_components/parser_unsafe': 0.03125, 'reward_components/parser_valid': 0.90625, 'reward_components/reward': 0.90625, 'reward_components/runs_correct': 1.21875, 'reward_components/runs_total': 1.375, 'reward_components/semantic_correct': 0.90625, 'reward_std': 0.29378482699394226, 'rewards/reward_func/mean': 0.90625, 'rewards/reward_func/std': 0.29378482699394226, 'sampling/importance_sampling_ratio/max': 2.862748861312866, 'sampling/importance_sampling_ratio/mean': 0.417085200548172, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 2.9688491821289062, 'sampling/sampling_logp_difference/mean': 0.010627579875290394, 'step': 14, 'step_time': 136.54511701548472}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.03125, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 4020.0, 'completions/mean_length': 1676.0625, 'completions/mean_terminated_length': 1598.0, 'completions/min_length': 682.0, 'completions/min_terminated_length': 682.0, 'entropy': 0.09528498537838459, 'epoch': 0.01953125, 'frac_reward_zero_std': 0.75, 'grad_norm': 0.0033239691983908415, 'learning_rate': 1e-05, 'loss': -0.006476776208728552, 'num_tokens': 2896925.0, 'reward': 0.859375, 'reward/selected_zero_std_group_fraction': 0.36666666666666664, 'reward/zero_std_group_fraction': 0.65, 'reward_components/channel_close_count': 0.96875, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.03125, 'reward_components/format_valid': 0.859375, 'reward_components/native_boundary_valid': 0.96875, 'reward_components/parser_json': 0.09375, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.875, 'reward_components/parser_unsafe': 0.109375, 'reward_components/parser_valid': 0.859375, 'reward_components/reward': 0.859375, 'reward_components/runs_correct': 1.21875, 'reward_components/runs_total': 1.5, 'reward_components/semantic_correct': 0.859375, 'reward_std': 0.3503824472427368, 'rewards/reward_func/mean': 0.859375, 'rewards/reward_func/std': 0.3503824472427368, 'sampling/importance_sampling_ratio/max': 2.320100784301758, 'sampling/importance_sampling_ratio/mean': 0.27009284496307373, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 12.702853202819824, 'sampling/sampling_logp_difference/mean': 0.013368000276386738, 'step': 15, 'step_time': 127.97649346617982}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.015625, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3962.0, 'completions/mean_length': 1511.53125, 'completions/mean_terminated_length': 1470.508056640625, 'completions/min_length': 653.0, 'completions/min_terminated_length': 653.0, 'entropy': 0.08486554026603699, 'epoch': 0.020833333333333332, 'frac_reward_zero_std': 0.875, 'grad_norm': 0.0001440451160306111, 'learning_rate': 1e-05, 'loss': -0.00045549581409431994, 'num_tokens': 3062719.0, 'reward': 0.984375, 'reward/selected_zero_std_group_fraction': 0.390625, 'reward/zero_std_group_fraction': 0.6640625, 'reward_components/channel_close_count': 1.0, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.015625, 'reward_components/format_valid': 0.984375, 'reward_components/native_boundary_valid': 1.0, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 1.0, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.984375, 'reward_components/reward': 0.984375, 'reward_components/runs_correct': 1.34375, 'reward_components/runs_total': 1.375, 'reward_components/semantic_correct': 0.984375, 'reward_std': 0.125, 'rewards/reward_func/mean': 0.984375, 'rewards/reward_func/std': 0.125, 'sampling/importance_sampling_ratio/max': 2.442641496658325, 'sampling/importance_sampling_ratio/mean': 0.31286460161209106, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 4.551835536956787, 'sampling/sampling_logp_difference/mean': 0.010723333805799484, 'step': 16, 'step_time': 136.00184198096395}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.0625, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3941.0, 'completions/mean_length': 1615.609375, 'completions/mean_terminated_length': 1450.2501220703125, 'completions/min_length': 608.0, 'completions/min_terminated_length': 608.0, 'entropy': 0.08994154632091522, 'epoch': 0.022135416666666668, 'frac_reward_zero_std': 0.5, 'grad_norm': 0.04282917454838753, 'learning_rate': 1e-05, 'loss': 0.005546635948121548, 'num_tokens': 3222630.0, 'reward': 0.84375, 'reward/selected_zero_std_group_fraction': 0.0, 'reward/zero_std_group_fraction': 0.5, 'reward_components/channel_close_count': 0.9375, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.0625, 'reward_components/format_valid': 0.84375, 'reward_components/native_boundary_valid': 0.9375, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.9375, 'reward_components/parser_unsafe': 0.09375, 'reward_components/parser_valid': 0.84375, 'reward_components/reward': 0.84375, 'reward_components/runs_correct': 1.046875, 'reward_components/runs_total': 1.25, 'reward_components/semantic_correct': 0.84375, 'reward_std': 0.36596253514289856, 'rewards/reward_func/mean': 0.84375, 'rewards/reward_func/std': 0.36596253514289856, 'sampling/importance_sampling_ratio/max': 2.3637678623199463, 'sampling/importance_sampling_ratio/mean': 0.2872670590877533, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 12.142862319946289, 'sampling/sampling_logp_difference/mean': 0.011082040145993233, 'step': 17, 'step_time': 163.51387257687747}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.140625, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3999.0, 'completions/mean_length': 2165.5, 'completions/mean_terminated_length': 1849.5999755859375, 'completions/min_length': 674.0, 'completions/min_terminated_length': 674.0, 'entropy': 0.12831554375588894, 'epoch': 0.0234375, 'frac_reward_zero_std': 0.625, 'grad_norm': 0.0067041851580142975, 'learning_rate': 1e-05, 'loss': -0.0075442735105752945, 'num_tokens': 3424262.0, 'reward': 0.84375, 'reward/selected_zero_std_group_fraction': 0.125, 'reward/zero_std_group_fraction': 0.5625, 'reward_components/channel_close_count': 0.859375, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.140625, 'reward_components/format_valid': 0.84375, 'reward_components/native_boundary_valid': 0.859375, 'reward_components/parser_json': 0.125, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.734375, 'reward_components/parser_unsafe': 0.015625, 'reward_components/parser_valid': 0.84375, 'reward_components/reward': 0.84375, 'reward_components/runs_correct': 1.265625, 'reward_components/runs_total': 1.5, 'reward_components/semantic_correct': 0.84375, 'reward_std': 0.36596253514289856, 'rewards/reward_func/mean': 0.84375, 'rewards/reward_func/std': 0.36596253514289856, 'sampling/importance_sampling_ratio/max': 1.0, 'sampling/importance_sampling_ratio/mean': 0.2487977147102356, 'sampling/importance_sampling_ratio/min': 7.1923109828953e-12, 'sampling/sampling_logp_difference/max': 22.121578216552734, 'sampling/sampling_logp_difference/mean': 0.012574245221912861, 'step': 18, 'step_time': 149.08230208978057}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.09375, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3817.0, 'completions/mean_length': 1678.09375, 'completions/mean_terminated_length': 1427.9654541015625, 'completions/min_length': 609.0, 'completions/min_terminated_length': 609.0, 'entropy': 0.10531319305300713, 'epoch': 0.024739583333333332, 'frac_reward_zero_std': 0.625, 'grad_norm': 0.0022952959407120943, 'learning_rate': 1e-05, 'loss': -0.0024729480501264334, 'num_tokens': 3610892.0, 'reward': 0.828125, 'reward/selected_zero_std_group_fraction': 0.16666666666666666, 'reward/zero_std_group_fraction': 0.5833333333333334, 'reward_components/channel_close_count': 0.90625, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.09375, 'reward_components/format_valid': 0.828125, 'reward_components/native_boundary_valid': 0.90625, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.90625, 'reward_components/parser_unsafe': 0.078125, 'reward_components/parser_valid': 0.828125, 'reward_components/reward': 0.828125, 'reward_components/runs_correct': 1.078125, 'reward_components/runs_total': 1.25, 'reward_components/semantic_correct': 0.828125, 'reward_std': 0.38025420904159546, 'rewards/reward_func/mean': 0.828125, 'rewards/reward_func/std': 0.38025420904159546, 'sampling/importance_sampling_ratio/max': 2.4680769443511963, 'sampling/importance_sampling_ratio/mean': 0.37189191579818726, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 10.000045776367188, 'sampling/sampling_logp_difference/mean': 0.012953806668519974, 'step': 19, 'step_time': 142.12330092396587}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.171875, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3978.0, 'completions/mean_length': 2114.96875, 'completions/mean_terminated_length': 1703.8114013671875, 'completions/min_length': 635.0, 'completions/min_terminated_length': 635.0, 'entropy': 0.10196862556040287, 'epoch': 0.026041666666666668, 'frac_reward_zero_std': 0.625, 'grad_norm': 0.01444269996136427, 'learning_rate': 1e-05, 'loss': -0.024201350286602974, 'num_tokens': 3812234.0, 'reward': 0.828125, 'reward/selected_zero_std_group_fraction': 0.1875, 'reward/zero_std_group_fraction': 0.59375, 'reward_components/channel_close_count': 0.828125, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.171875, 'reward_components/format_valid': 0.828125, 'reward_components/native_boundary_valid': 0.828125, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.828125, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.828125, 'reward_components/reward': 0.828125, 'reward_components/runs_correct': 1.28125, 'reward_components/runs_total': 1.625, 'reward_components/semantic_correct': 0.828125, 'reward_std': 0.38025420904159546, 'rewards/reward_func/mean': 0.828125, 'rewards/reward_func/std': 0.38025420904159546, 'sampling/importance_sampling_ratio/max': 2.895143747329712, 'sampling/importance_sampling_ratio/mean': 0.4145027697086334, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 15.089095115661621, 'sampling/sampling_logp_difference/mean': 0.012161580845713615, 'step': 20, 'step_time': 148.25134309753776}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.15625, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3581.0, 'completions/mean_length': 1848.359375, 'completions/mean_terminated_length': 1432.129638671875, 'completions/min_length': 688.0, 'completions/min_terminated_length': 688.0, 'entropy': 0.10191419580951333, 'epoch': 0.02734375, 'frac_reward_zero_std': 0.5, 'grad_norm': 0.004159783944487572, 'learning_rate': 1e-05, 'loss': -0.00925515778362751, 'num_tokens': 3981473.0, 'reward': 0.8125, 'reward/selected_zero_std_group_fraction': 0.15, 'reward/zero_std_group_fraction': 0.575, 'reward_components/channel_close_count': 0.84375, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.15625, 'reward_components/format_valid': 0.8125, 'reward_components/native_boundary_valid': 0.84375, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.84375, 'reward_components/parser_unsafe': 0.03125, 'reward_components/parser_valid': 0.8125, 'reward_components/reward': 0.8125, 'reward_components/runs_correct': 1.046875, 'reward_components/runs_total': 1.25, 'reward_components/semantic_correct': 0.8125, 'reward_std': 0.39339789748191833, 'rewards/reward_func/mean': 0.8125, 'rewards/reward_func/std': 0.39339789748191833, 'sampling/importance_sampling_ratio/max': 2.27093243598938, 'sampling/importance_sampling_ratio/mean': 0.34236621856689453, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 5.04082727432251, 'sampling/sampling_logp_difference/mean': 0.01293655950576067, 'step': 21, 'step_time': 137.55480616306886}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.125, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 4069.0, 'completions/mean_length': 2542.875, 'completions/mean_terminated_length': 2321.0, 'completions/min_length': 907.0, 'completions/min_terminated_length': 907.0, 'entropy': 0.10400373674929142, 'epoch': 0.028645833333333332, 'frac_reward_zero_std': 0.625, 'grad_norm': 0.03445667773485184, 'learning_rate': 1e-05, 'loss': -0.02254950813949108, 'num_tokens': 4212249.0, 'reward': 0.875, 'reward/selected_zero_std_group_fraction': 0.16666666666666666, 'reward/zero_std_group_fraction': 0.5833333333333334, 'reward_components/channel_close_count': 0.875, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.125, 'reward_components/format_valid': 0.875, 'reward_components/native_boundary_valid': 0.875, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.875, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.875, 'reward_components/reward': 0.875, 'reward_components/runs_correct': 1.75, 'reward_components/runs_total': 2.0, 'reward_components/semantic_correct': 0.875, 'reward_std': 0.3333333432674408, 'rewards/reward_func/mean': 0.875, 'rewards/reward_func/std': 0.3333333432674408, 'sampling/importance_sampling_ratio/max': 2.4763169288635254, 'sampling/importance_sampling_ratio/mean': 0.281249463558197, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 17.946428298950195, 'sampling/sampling_logp_difference/mean': 0.012010331265628338, 'step': 22, 'step_time': 160.82924399897456}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.125, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3977.0, 'completions/mean_length': 1940.03125, 'completions/mean_terminated_length': 1632.0357666015625, 'completions/min_length': 586.0, 'completions/min_terminated_length': 586.0, 'entropy': 0.10038873367011547, 'epoch': 0.029947916666666668, 'frac_reward_zero_std': 0.625, 'grad_norm': 0.7102032899856567, 'learning_rate': 1e-05, 'loss': -0.011978881433606148, 'num_tokens': 4405851.0, 'reward': 0.8125, 'reward/selected_zero_std_group_fraction': 0.17857142857142858, 'reward/zero_std_group_fraction': 0.5892857142857143, 'reward_components/channel_close_count': 0.875, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.125, 'reward_components/format_valid': 0.8125, 'reward_components/native_boundary_valid': 0.875, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.875, 'reward_components/parser_unsafe': 0.0625, 'reward_components/parser_valid': 0.8125, 'reward_components/reward': 0.8125, 'reward_components/runs_correct': 1.171875, 'reward_components/runs_total': 1.5, 'reward_components/semantic_correct': 0.8125, 'reward_std': 0.39339789748191833, 'rewards/reward_func/mean': 0.8125, 'rewards/reward_func/std': 0.39339789748191833, 'sampling/importance_sampling_ratio/max': 1.4011995792388916, 'sampling/importance_sampling_ratio/mean': 0.3341714143753052, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 15.6409273147583, 'sampling/sampling_logp_difference/mean': 0.010782009921967983, 'step': 23, 'step_time': 145.83045846642926}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.03125, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3806.0, 'completions/mean_length': 2056.171875, 'completions/mean_terminated_length': 1990.370849609375, 'completions/min_length': 695.0, 'completions/min_terminated_length': 695.0, 'entropy': 0.1051472770050168, 'epoch': 0.03125, 'frac_reward_zero_std': 0.75, 'grad_norm': 0.008990166708827019, 'learning_rate': 1e-05, 'loss': -0.0021168089006096125, 'num_tokens': 4605414.0, 'reward': 0.96875, 'reward/selected_zero_std_group_fraction': 0.21875, 'reward/zero_std_group_fraction': 0.609375, 'reward_components/channel_close_count': 0.96875, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.03125, 'reward_components/format_valid': 0.96875, 'reward_components/native_boundary_valid': 0.96875, 'reward_components/parser_json': 0.109375, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.859375, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.96875, 'reward_components/reward': 0.96875, 'reward_components/runs_correct': 1.8125, 'reward_components/runs_total': 1.875, 'reward_components/semantic_correct': 0.96875, 'reward_std': 0.17536810040473938, 'rewards/reward_func/mean': 0.96875, 'rewards/reward_func/std': 0.17536810040473938, 'sampling/importance_sampling_ratio/max': 2.7678279876708984, 'sampling/importance_sampling_ratio/mean': 0.23847268521785736, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 15.094564437866211, 'sampling/sampling_logp_difference/mean': 0.012344561517238617, 'step': 24, 'step_time': 144.50246571889147}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.015625, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3653.0, 'completions/mean_length': 2069.171875, 'completions/mean_terminated_length': 2037.0001220703125, 'completions/min_length': 888.0, 'completions/min_terminated_length': 888.0, 'entropy': 0.08877563942223787, 'epoch': 0.032552083333333336, 'frac_reward_zero_std': 0.875, 'grad_norm': 0.0003568730317056179, 'learning_rate': 1e-05, 'loss': -0.0006816749810241163, 'num_tokens': 4816817.0, 'reward': 0.984375, 'reward/selected_zero_std_group_fraction': 0.2777777777777778, 'reward/zero_std_group_fraction': 0.6388888888888888, 'reward_components/channel_close_count': 0.984375, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.015625, 'reward_components/format_valid': 0.984375, 'reward_components/native_boundary_valid': 0.984375, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.984375, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.984375, 'reward_components/reward': 0.984375, 'reward_components/runs_correct': 1.84375, 'reward_components/runs_total': 1.875, 'reward_components/semantic_correct': 0.984375, 'reward_std': 0.125, 'rewards/reward_func/mean': 0.984375, 'rewards/reward_func/std': 0.125, 'sampling/importance_sampling_ratio/max': 1.5636930465698242, 'sampling/importance_sampling_ratio/mean': 0.2087942510843277, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 4.127016067504883, 'sampling/sampling_logp_difference/mean': 0.010643535293638706, 'step': 25, 'step_time': 148.26050570281222}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.109375, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 4081.0, 'completions/mean_length': 1636.640625, 'completions/mean_terminated_length': 1440.7193603515625, 'completions/min_length': 606.0, 'completions/min_terminated_length': 606.0, 'entropy': 0.10319297481328249, 'epoch': 0.033854166666666664, 'frac_reward_zero_std': 0.375, 'grad_norm': 0.09031198918819427, 'learning_rate': 1e-05, 'loss': -0.014390656724572182, 'num_tokens': 4978202.0, 'reward': 0.796875, 'reward/selected_zero_std_group_fraction': 0.25, 'reward/zero_std_group_fraction': 0.6125, 'reward_components/channel_close_count': 0.921875, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.078125, 'reward_components/format_valid': 0.796875, 'reward_components/native_boundary_valid': 0.890625, 'reward_components/parser_json': 0.125, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.765625, 'reward_components/parser_unsafe': 0.09375, 'reward_components/parser_valid': 0.796875, 'reward_components/reward': 0.796875, 'reward_components/runs_correct': 1.03125, 'reward_components/runs_total': 1.375, 'reward_components/semantic_correct': 0.796875, 'reward_std': 0.40550529956817627, 'rewards/reward_func/mean': 0.796875, 'rewards/reward_func/std': 0.40550529956817627, 'sampling/importance_sampling_ratio/max': 2.638819456100464, 'sampling/importance_sampling_ratio/mean': 0.35338953137397766, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 3.432163715362549, 'sampling/sampling_logp_difference/mean': 0.01307997852563858, 'step': 26, 'step_time': 132.963375385385}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.171875, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3999.0, 'completions/mean_length': 2254.4375, 'completions/mean_terminated_length': 1872.2264404296875, 'completions/min_length': 589.0, 'completions/min_terminated_length': 589.0, 'entropy': 0.10867640702053905, 'epoch': 0.03515625, 'frac_reward_zero_std': 0.375, 'grad_norm': 0.0060604410246014595, 'learning_rate': 1e-05, 'loss': -0.010539907962083817, 'num_tokens': 5198326.0, 'reward': 0.765625, 'reward/selected_zero_std_group_fraction': 0.22727272727272727, 'reward/zero_std_group_fraction': 0.5909090909090909, 'reward_components/channel_close_count': 0.828125, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.171875, 'reward_components/format_valid': 0.765625, 'reward_components/native_boundary_valid': 0.828125, 'reward_components/parser_json': 0.1875, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.640625, 'reward_components/parser_unsafe': 0.0625, 'reward_components/parser_valid': 0.765625, 'reward_components/reward': 0.765625, 'reward_components/runs_correct': 1.203125, 'reward_components/runs_total': 1.5, 'reward_components/semantic_correct': 0.765625, 'reward_std': 0.42695629596710205, 'rewards/reward_func/mean': 0.765625, 'rewards/reward_func/std': 0.42695629596710205, 'sampling/importance_sampling_ratio/max': 1.9732943773269653, 'sampling/importance_sampling_ratio/mean': 0.26364538073539734, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 3.868208646774292, 'sampling/sampling_logp_difference/mean': 0.013032162562012672, 'step': 27, 'step_time': 155.20538381813094}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.125, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3343.0, 'completions/mean_length': 2022.46875, 'completions/mean_terminated_length': 1726.2501220703125, 'completions/min_length': 802.0, 'completions/min_terminated_length': 802.0, 'entropy': 0.08392351027578115, 'epoch': 0.036458333333333336, 'frac_reward_zero_std': 0.75, 'grad_norm': 0.0033873559441417456, 'learning_rate': 1e-05, 'loss': -0.01038953848183155, 'num_tokens': 5395476.0, 'reward': 0.859375, 'reward/selected_zero_std_group_fraction': 0.25, 'reward/zero_std_group_fraction': 0.6041666666666666, 'reward_components/channel_close_count': 0.875, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.125, 'reward_components/format_valid': 0.875, 'reward_components/native_boundary_valid': 0.875, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.875, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.875, 'reward_components/reward': 0.859375, 'reward_components/runs_correct': 1.453125, 'reward_components/runs_total': 1.625, 'reward_components/semantic_correct': 0.8671875, 'reward_std': 0.3503824472427368, 'rewards/reward_func/mean': 0.859375, 'rewards/reward_func/std': 0.3503824472427368, 'sampling/importance_sampling_ratio/max': 1.5085198879241943, 'sampling/importance_sampling_ratio/mean': 0.26240062713623047, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 20.089284896850586, 'sampling/sampling_logp_difference/mean': 0.010678061284124851, 'step': 28, 'step_time': 146.1634053401649}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.046875, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3729.0, 'completions/mean_length': 1768.109375, 'completions/mean_terminated_length': 1653.622802734375, 'completions/min_length': 633.0, 'completions/min_terminated_length': 633.0, 'entropy': 0.09789686091244221, 'epoch': 0.037760416666666664, 'frac_reward_zero_std': 0.75, 'grad_norm': 0.0009929501684382558, 'learning_rate': 1e-05, 'loss': -0.0021212436258792877, 'num_tokens': 5582875.0, 'reward': 0.953125, 'reward/selected_zero_std_group_fraction': 0.2692307692307692, 'reward/zero_std_group_fraction': 0.6153846153846154, 'reward_components/channel_close_count': 0.953125, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.046875, 'reward_components/format_valid': 0.953125, 'reward_components/native_boundary_valid': 0.953125, 'reward_components/parser_json': 0.109375, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.84375, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.953125, 'reward_components/reward': 0.953125, 'reward_components/runs_correct': 1.3125, 'reward_components/runs_total': 1.375, 'reward_components/semantic_correct': 0.953125, 'reward_std': 0.21304203569889069, 'rewards/reward_func/mean': 0.953125, 'rewards/reward_func/std': 0.21304203569889069, 'sampling/importance_sampling_ratio/max': 2.259580373764038, 'sampling/importance_sampling_ratio/mean': 0.2557954788208008, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 9.817340850830078, 'sampling/sampling_logp_difference/mean': 0.011634938418865204, 'step': 29, 'step_time': 145.3711199723184}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.0625, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3717.0, 'completions/mean_length': 2064.203125, 'completions/mean_terminated_length': 1928.7501220703125, 'completions/min_length': 826.0, 'completions/min_terminated_length': 826.0, 'entropy': 0.09083252307027578, 'epoch': 0.0390625, 'frac_reward_zero_std': 0.625, 'grad_norm': 0.0027994767297059298, 'learning_rate': 1e-05, 'loss': 0.003185940207913518, 'num_tokens': 5792808.0, 'reward': 0.796875, 'reward/selected_zero_std_group_fraction': 0.26785714285714285, 'reward/zero_std_group_fraction': 0.6160714285714286, 'reward_components/channel_close_count': 0.9375, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.0625, 'reward_components/format_valid': 0.796875, 'reward_components/native_boundary_valid': 0.9375, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.9375, 'reward_components/parser_unsafe': 0.140625, 'reward_components/parser_valid': 0.796875, 'reward_components/reward': 0.796875, 'reward_components/runs_correct': 1.34375, 'reward_components/runs_total': 1.75, 'reward_components/semantic_correct': 0.796875, 'reward_std': 0.40550529956817627, 'rewards/reward_func/mean': 0.796875, 'rewards/reward_func/std': 0.40550529956817627, 'sampling/importance_sampling_ratio/max': 2.066760540008545, 'sampling/importance_sampling_ratio/mean': 0.22840645909309387, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 7.2408766746521, 'sampling/sampling_logp_difference/mean': 0.01208382099866867, 'step': 30, 'step_time': 150.06700502010062}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.03125, 'completions/max_length': 4096.0, 'completions/max_terminated_length': 3760.0, 'completions/mean_length': 1482.9375, 'completions/mean_terminated_length': 1398.6451416015625, 'completions/min_length': 576.0, 'completions/min_terminated_length': 576.0, 'entropy': 0.09557253075763583, 'epoch': 0.040364583333333336, 'frac_reward_zero_std': 0.75, 'grad_norm': 0.0006523437332361937, 'learning_rate': 1e-05, 'loss': -0.002152282977476716, 'num_tokens': 5951140.0, 'reward': 0.96875, 'reward/selected_zero_std_group_fraction': 0.2833333333333333, 'reward/zero_std_group_fraction': 0.625, 'reward_components/channel_close_count': 0.96875, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.03125, 'reward_components/format_valid': 0.96875, 'reward_components/native_boundary_valid': 0.96875, 'reward_components/parser_json': 0.0, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.96875, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 0.96875, 'reward_components/reward': 0.96875, 'reward_components/runs_correct': 1.328125, 'reward_components/runs_total': 1.375, 'reward_components/semantic_correct': 0.96875, 'reward_std': 0.17536810040473938, 'rewards/reward_func/mean': 0.96875, 'rewards/reward_func/std': 0.17536810040473938, 'sampling/importance_sampling_ratio/max': 2.6156203746795654, 'sampling/importance_sampling_ratio/mean': 0.3359881639480591, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 2.7988429069519043, 'sampling/sampling_logp_difference/mean': 0.011432373896241188, 'step': 31, 'step_time': 133.95856479695067}, {'clip_ratio/high_max': 0.0, 'clip_ratio/high_mean': 0.0, 'clip_ratio/low_mean': 0.0, 'clip_ratio/low_min': 0.0, 'clip_ratio/region_mean': 0.0, 'completions/clipped_ratio': 0.0, 'completions/max_length': 3307.0, 'completions/max_terminated_length': 3307.0, 'completions/mean_length': 1282.3125, 'completions/mean_terminated_length': 1282.3125, 'completions/min_length': 624.0, 'completions/min_terminated_length': 624.0, 'entropy': 0.08730019442737103, 'epoch': 0.041666666666666664, 'frac_reward_zero_std': 1.0, 'grad_norm': 0.0, 'learning_rate': 1e-05, 'loss': 0.0, 'num_tokens': 6089592.0, 'reward': 1.0, 'reward/selected_zero_std_group_fraction': 0.328125, 'reward/zero_std_group_fraction': 0.6484375, 'reward_components/channel_close_count': 1.0, 'reward_components/channel_open_count': 1.0, 'reward_components/completion_truncated': 0.0, 'reward_components/format_valid': 1.0, 'reward_components/native_boundary_valid': 1.0, 'reward_components/parser_json': 0.125, 'reward_components/parser_labelled_records': 0.0, 'reward_components/parser_natural': 0.875, 'reward_components/parser_unsafe': 0.0, 'reward_components/parser_valid': 1.0, 'reward_components/reward': 1.0, 'reward_components/runs_correct': 1.25, 'reward_components/runs_total': 1.25, 'reward_components/semantic_correct': 1.0, 'reward_std': 0.0, 'rewards/reward_func/mean': 1.0, 'rewards/reward_func/std': 0.0, 'sampling/importance_sampling_ratio/max': 2.468935012817383, 'sampling/importance_sampling_ratio/mean': 0.3145565986633301, 'sampling/importance_sampling_ratio/min': 0.0, 'sampling/sampling_logp_difference/max': 3.8094847202301025, 'sampling/sampling_logp_difference/mean': 0.01148232538253069, 'step': 32, 'step_time': 102.69295566342771}], best_metric=None, best_global_step=None, best_model_checkpoint=None, is_local_process_zero=True, is_world_process_zero=True, is_hyper_param_search=False, trial_name=None, trial_params=None, stateful_callbacks={'TrainerControl': {'args': {'should_epoch_stop': False, 'should_evaluate': False, 'should_log': False, 'should_save': True, 'should_training_stop': True}, 'attributes': {}}})",
22
+ "note": "completion_ids removed; regenerate by tokenising completion_raw_text with the gemma4-26b-a4b tokenizer (verified exact on 221 records). log_extra/log_metric were bound-method reprs. Fetched as parallel byte ranges because one connection to this repo sustains only 1.4 MiB/s, and because the Hub's declared size for some files disagrees with their stored bytes."
23
+ }